any-doctor 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTEXT.md +128 -0
- package/README.md +68 -0
- package/bin/capabilities.d.ts +15 -0
- package/bin/capabilities.js +131 -0
- package/bin/cli.d.ts +2 -0
- package/bin/cli.js +426 -0
- package/bin/clipboard.d.ts +1 -0
- package/bin/clipboard.js +10 -0
- package/bin/contract.d.ts +134 -0
- package/bin/contract.js +70 -0
- package/bin/dashboard.d.ts +108 -0
- package/bin/dashboard.js +718 -0
- package/bin/discover.d.ts +24 -0
- package/bin/discover.js +87 -0
- package/bin/doctor-loader.d.mts +1 -0
- package/bin/doctor-loader.mjs +161 -0
- package/bin/engine.d.ts +18 -0
- package/bin/engine.js +22 -0
- package/bin/fuzzy.d.ts +2 -0
- package/bin/fuzzy.js +31 -0
- package/bin/import-guard.mjs +31 -0
- package/bin/keys.d.ts +2 -0
- package/bin/keys.js +72 -0
- package/bin/palette.d.ts +7 -0
- package/bin/palette.js +14 -0
- package/bin/picker.d.ts +12 -0
- package/bin/picker.js +82 -0
- package/bin/report.d.ts +18 -0
- package/bin/report.js +159 -0
- package/bin/runner.d.ts +58 -0
- package/bin/runner.js +271 -0
- package/bin/score.d.ts +13 -0
- package/bin/score.js +39 -0
- package/bin/sdk.d.ts +5 -0
- package/bin/sdk.js +95 -0
- package/bin/search-host.d.ts +6 -0
- package/bin/search-host.js +56 -0
- package/bin/select.d.ts +35 -0
- package/bin/select.js +45 -0
- package/bin/tty.d.ts +38 -0
- package/bin/tty.js +94 -0
- package/docs/REPAIR-LOG.md +45 -0
- package/docs/RESULTS.md +70 -0
- package/docs/decisions.md +450 -0
- package/docs/example-catalog.md +122 -0
- package/docs/features.md +67 -0
- package/docs/first-shot-results.md +18 -0
- package/docs/intents.md +21 -0
- package/docs/kill-test.md +54 -0
- package/docs/research.md +66 -0
- package/docs/vision.md +83 -0
- package/doctors/AGENTS.md +103 -0
- package/doctors/api-route-files-do-import.fixtures.mjs +61 -0
- package/doctors/api-route-files-do-import.mjs +26 -0
- package/doctors/async-doctor.fixtures.mjs +147 -0
- package/doctors/async-doctor.mjs +295 -0
- package/doctors/convex-doctor.fixtures.mjs +177 -0
- package/doctors/convex-doctor.mjs +223 -0
- package/doctors/date-now-used-inside-effect.fixtures.mjs +46 -0
- package/doctors/date-now-used-inside-effect.mjs +132 -0
- package/doctors/json-parse-calls-llm-api.fixtures.mjs +37 -0
- package/doctors/json-parse-calls-llm-api.mjs +85 -0
- package/doctors/route-handlers-touch-database-before.fixtures.mjs +58 -0
- package/doctors/route-handlers-touch-database-before.mjs +98 -0
- package/doctors/z-record-called-with-single.fixtures.mjs +28 -0
- package/doctors/z-record-called-with-single.mjs +19 -0
- package/fixtures/sample-app/src/hooks/useChat.ts +15 -0
- package/fixtures/sample-app/src/lib/ai/client.ts +5 -0
- package/fixtures/sample-app/src/schemas/user.ts +6 -0
- package/fixtures/sample-app/src/services/chat.ts +17 -0
- package/fixtures/sample-app/src/services/user.ts +10 -0
- package/fixtures/sample-app/src/utils/sync.ts +16 -0
- package/package.json +42 -0
- package/skill/any-doctor.skill.md +188 -0
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import * as os from "os";
|
|
2
|
+
import * as path from "path";
|
|
3
|
+
import { SEARCH_REQUEST } from "./contract.js";
|
|
4
|
+
import { runEngineSearch } from "./engine.js";
|
|
5
|
+
// The search host: the doctor child cannot spawn (Confinement), so it asks
|
|
6
|
+
// any-doctor to run the Engine over a dedicated channel. This module is
|
|
7
|
+
// the host side — a pure function from request line to response JSON,
|
|
8
|
+
// directly testable without a real doctor child.
|
|
9
|
+
//
|
|
10
|
+
// The root check is a security decision: a run may search its target, a
|
|
11
|
+
// verify may search its fixture sandboxes under the temp dir, and meta may
|
|
12
|
+
// not search at all.
|
|
13
|
+
export function searchBase(mode) {
|
|
14
|
+
switch (mode.kind) {
|
|
15
|
+
// Verify sandboxes are seeded under the temp dir with this prefix —
|
|
16
|
+
// not the whole temp dir, and nothing else in it.
|
|
17
|
+
case "verify": return path.join(os.tmpdir(), "any-doctor-verify-");
|
|
18
|
+
case "run": return mode.root;
|
|
19
|
+
case "meta": return "";
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
// Bases are anchors: a run's target directory (anything beneath it), or
|
|
23
|
+
// verify's sandbox prefix (any any-doctor-verify-* sandbox). The prefix
|
|
24
|
+
// form ends in "-" on purpose — mkdtemp appends to it.
|
|
25
|
+
function withinBase(root, base) {
|
|
26
|
+
if (root === base)
|
|
27
|
+
return true;
|
|
28
|
+
if (base.endsWith("-"))
|
|
29
|
+
return root.startsWith(base);
|
|
30
|
+
return root.startsWith(base + path.sep);
|
|
31
|
+
}
|
|
32
|
+
// One request line in, one response body out (the SEARCH_RESULT sentinel
|
|
33
|
+
// is framing added by the transport in the runner). Returns null for lines
|
|
34
|
+
// that are not requests — the channel carries nothing else, so they are
|
|
35
|
+
// ignored rather than answered.
|
|
36
|
+
export function handleSearchLine(line, mode, engine = runEngineSearch) {
|
|
37
|
+
var _a;
|
|
38
|
+
if (!line.startsWith(SEARCH_REQUEST))
|
|
39
|
+
return null;
|
|
40
|
+
let req;
|
|
41
|
+
try {
|
|
42
|
+
req = JSON.parse(line.slice(SEARCH_REQUEST.length));
|
|
43
|
+
}
|
|
44
|
+
catch {
|
|
45
|
+
return JSON.stringify({ error: "ctx.search failed: malformed host request" });
|
|
46
|
+
}
|
|
47
|
+
const base = searchBase(mode);
|
|
48
|
+
// resolve() collapses `..` and anchors relatives — a prefix check on the
|
|
49
|
+
// raw string would let /target/../../etc through.
|
|
50
|
+
const root = typeof req.root === "string" ? path.resolve(req.root) : "";
|
|
51
|
+
if (base === "" || !withinBase(root, base)) {
|
|
52
|
+
return JSON.stringify({ error: "ctx.search failed: search root is outside the allowed target" });
|
|
53
|
+
}
|
|
54
|
+
const r = engine(String((_a = req.pattern) !== null && _a !== void 0 ? _a : ""), typeof req.language === "string" ? req.language : "TypeScript", root);
|
|
55
|
+
return r.ok ? JSON.stringify({ matches: r.matches }) : JSON.stringify({ error: r.error });
|
|
56
|
+
}
|
package/bin/select.d.ts
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import { BrokenDoctor } from "./discover.js";
|
|
2
|
+
import { TtyEnv } from "./tty.js";
|
|
3
|
+
export interface SelectionRow {
|
|
4
|
+
scope: string;
|
|
5
|
+
slug: string;
|
|
6
|
+
description: string;
|
|
7
|
+
}
|
|
8
|
+
export type Selection = {
|
|
9
|
+
kind: "doctor";
|
|
10
|
+
doctorPath: string;
|
|
11
|
+
skipped: BrokenDoctor[];
|
|
12
|
+
unsafe: string[];
|
|
13
|
+
} | {
|
|
14
|
+
kind: "not-found";
|
|
15
|
+
arg: string;
|
|
16
|
+
} | {
|
|
17
|
+
kind: "none-discovered";
|
|
18
|
+
broken: BrokenDoctor[];
|
|
19
|
+
unsafe: string[];
|
|
20
|
+
} | {
|
|
21
|
+
kind: "non-interactive";
|
|
22
|
+
rows: SelectionRow[];
|
|
23
|
+
skipped: BrokenDoctor[];
|
|
24
|
+
unsafe: string[];
|
|
25
|
+
} | {
|
|
26
|
+
kind: "cancelled";
|
|
27
|
+
};
|
|
28
|
+
export interface SelectOptions {
|
|
29
|
+
cwd: string;
|
|
30
|
+
globalDir?: string;
|
|
31
|
+
useColor: boolean;
|
|
32
|
+
allowPicker?: boolean;
|
|
33
|
+
env: TtyEnv;
|
|
34
|
+
}
|
|
35
|
+
export declare function selectDoctor(doctorArg: string | undefined, options: SelectOptions): Promise<Selection>;
|
package/bin/select.js
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import { brokenDoctors, discoverDoctors, resolveDoctorPath, unsafeSlugs } from "./discover.js";
|
|
2
|
+
import { canRunTui } from "./tty.js";
|
|
3
|
+
import { pickItemOn } from "./picker.js";
|
|
4
|
+
export async function selectDoctor(doctorArg, options) {
|
|
5
|
+
if (doctorArg !== undefined) {
|
|
6
|
+
const resolved = resolveDoctorPath(doctorArg, options.cwd, options.globalDir !== undefined ? { globalDir: options.globalDir } : undefined);
|
|
7
|
+
return resolved !== null
|
|
8
|
+
? { kind: "doctor", doctorPath: resolved, skipped: [], unsafe: [] }
|
|
9
|
+
: { kind: "not-found", arg: doctorArg };
|
|
10
|
+
}
|
|
11
|
+
const discovered = await discoverDoctors(options.cwd, options.globalDir !== undefined ? { globalDir: options.globalDir } : undefined);
|
|
12
|
+
const valid = discovered.filter(d => d.meta !== null);
|
|
13
|
+
const unsafe = unsafeSlugs(discovered);
|
|
14
|
+
const broken = brokenDoctors(discovered);
|
|
15
|
+
if (valid.length === 0)
|
|
16
|
+
return { kind: "none-discovered", broken, unsafe };
|
|
17
|
+
// The gate runs before a picker ever starts, so "cancelled" can only mean
|
|
18
|
+
// the user ended the pick — never "this isn't a terminal".
|
|
19
|
+
if (!canRunTui(options.env) || options.allowPicker === false) {
|
|
20
|
+
return {
|
|
21
|
+
kind: "non-interactive",
|
|
22
|
+
skipped: broken,
|
|
23
|
+
unsafe,
|
|
24
|
+
rows: valid.map(d => ({
|
|
25
|
+
scope: d.scope,
|
|
26
|
+
slug: d.slug,
|
|
27
|
+
description: d.meta.description,
|
|
28
|
+
})),
|
|
29
|
+
};
|
|
30
|
+
}
|
|
31
|
+
const chosen = await pickItemOn(options.env, valid.map(d => ({
|
|
32
|
+
id: d.slug,
|
|
33
|
+
label: d.meta.description,
|
|
34
|
+
sub: d.scope,
|
|
35
|
+
severity: d.meta.severity,
|
|
36
|
+
})), options.useColor, "Select a doctor");
|
|
37
|
+
if (chosen === null)
|
|
38
|
+
return { kind: "cancelled" };
|
|
39
|
+
return {
|
|
40
|
+
kind: "doctor",
|
|
41
|
+
doctorPath: valid.find(d => d.slug === chosen.id).path,
|
|
42
|
+
skipped: broken,
|
|
43
|
+
unsafe,
|
|
44
|
+
};
|
|
45
|
+
}
|
package/bin/tty.d.ts
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
export interface TtyEnv {
|
|
2
|
+
stdin: TtyStdin;
|
|
3
|
+
stdout: TtyStdout;
|
|
4
|
+
}
|
|
5
|
+
export declare function processTtyEnv(): TtyEnv;
|
|
6
|
+
export declare function canRunTui(env: {
|
|
7
|
+
stdin: {
|
|
8
|
+
readonly isTTY?: boolean;
|
|
9
|
+
};
|
|
10
|
+
stdout: {
|
|
11
|
+
readonly isTTY?: boolean;
|
|
12
|
+
};
|
|
13
|
+
}): boolean;
|
|
14
|
+
export interface TtyStdin {
|
|
15
|
+
readonly isTTY?: boolean;
|
|
16
|
+
readonly isRaw?: boolean;
|
|
17
|
+
setRawMode(mode: boolean): unknown;
|
|
18
|
+
resume(): unknown;
|
|
19
|
+
pause(): unknown;
|
|
20
|
+
on(event: "data", listener: (chunk: string | Buffer) => void): unknown;
|
|
21
|
+
removeListener(event: "data", listener: (chunk: string | Buffer) => void): unknown;
|
|
22
|
+
}
|
|
23
|
+
export interface TtyStdout {
|
|
24
|
+
readonly isTTY?: boolean;
|
|
25
|
+
columns?: number;
|
|
26
|
+
rows?: number;
|
|
27
|
+
write(s: string): unknown;
|
|
28
|
+
}
|
|
29
|
+
export declare function visibleWidth(s: string): number;
|
|
30
|
+
export declare function truncateVisible(s: string, width: number): string;
|
|
31
|
+
export declare function paintFrame(stdout: TtyStdout, frame: string, cols: number): void;
|
|
32
|
+
export interface RunTtyOptions<T> {
|
|
33
|
+
stdin: TtyStdin;
|
|
34
|
+
stdout: TtyStdout;
|
|
35
|
+
frame: () => string;
|
|
36
|
+
onKey: (key: string, finish: (result: T) => void) => void;
|
|
37
|
+
}
|
|
38
|
+
export declare function runTty<T>(options: RunTtyOptions<T>): Promise<T>;
|
package/bin/tty.js
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
import { createKeyFeed } from "./keys.js";
|
|
2
|
+
// The one adapter from Node's process stdio to the tty seam.
|
|
3
|
+
export function processTtyEnv() {
|
|
4
|
+
return {
|
|
5
|
+
stdin: process.stdin,
|
|
6
|
+
stdout: process.stdout,
|
|
7
|
+
};
|
|
8
|
+
}
|
|
9
|
+
// "Is this a real terminal" — the shared floor of every TUI decision.
|
|
10
|
+
// Deliberately excludes headless env vars and width heuristics: those are
|
|
11
|
+
// report-vs-dashboard policy and belong to the command layer.
|
|
12
|
+
export function canRunTui(env) {
|
|
13
|
+
return Boolean(env.stdin.isTTY && env.stdout.isTTY);
|
|
14
|
+
}
|
|
15
|
+
export function visibleWidth(s) {
|
|
16
|
+
return s.replace(/\x1b\[[0-9;]*m/g, "").length;
|
|
17
|
+
}
|
|
18
|
+
export function truncateVisible(s, width) {
|
|
19
|
+
if (visibleWidth(s) <= width)
|
|
20
|
+
return s;
|
|
21
|
+
let out = "";
|
|
22
|
+
let w = 0;
|
|
23
|
+
for (const ch of s.replace(/\x1b\[[0-9;]*m/g, "")) {
|
|
24
|
+
if (w + 1 > width - 1)
|
|
25
|
+
break;
|
|
26
|
+
out += ch;
|
|
27
|
+
w++;
|
|
28
|
+
}
|
|
29
|
+
return out + "…";
|
|
30
|
+
}
|
|
31
|
+
// In-place repaint, always: home the cursor and rewrite every line with a
|
|
32
|
+
// clear-to-end-of-line, then clear below the frame. Per-line \x1b[K plus
|
|
33
|
+
// the trailing \x1b[J fully own the screen, so no full-screen \x1b[2J
|
|
34
|
+
// erase is ever needed — including on the first paint, where the erase
|
|
35
|
+
// showed as a one-time blank flash while the frame streamed in. The payload
|
|
36
|
+
// is wrapped in DECSET 2026 (synchronized output): terminals that support
|
|
37
|
+
// it hold the repaint until the frame is fully transmitted, so they never
|
|
38
|
+
// paint a half-frame; terminals that don't simply ignore the mode.
|
|
39
|
+
export function paintFrame(stdout, frame, cols) {
|
|
40
|
+
const width = Math.max(10, cols - 1);
|
|
41
|
+
const lines = frame.split("\n").map(l => truncateVisible(l, width) + "\x1b[K");
|
|
42
|
+
stdout.write("\x1b[?2026h\x1b[H" + lines.join("\n") + "\x1b[J\x1b[?2026l");
|
|
43
|
+
}
|
|
44
|
+
export function runTty(options) {
|
|
45
|
+
const { stdin, stdout, frame, onKey } = options;
|
|
46
|
+
return new Promise((resolve) => {
|
|
47
|
+
const wasRaw = stdin.isRaw;
|
|
48
|
+
stdin.setRawMode(true);
|
|
49
|
+
stdin.resume();
|
|
50
|
+
stdout.write("\x1b[?25l");
|
|
51
|
+
let lastFrame;
|
|
52
|
+
const repaint = () => {
|
|
53
|
+
const f = frame();
|
|
54
|
+
if (f === lastFrame)
|
|
55
|
+
return; // identical frame: not one byte of churn
|
|
56
|
+
lastFrame = f;
|
|
57
|
+
paintFrame(stdout, f, stdout.columns || 120);
|
|
58
|
+
};
|
|
59
|
+
let settled = false;
|
|
60
|
+
const finish = (result) => {
|
|
61
|
+
if (settled)
|
|
62
|
+
return;
|
|
63
|
+
settled = true;
|
|
64
|
+
stdin.removeListener("data", feed);
|
|
65
|
+
if (wasRaw !== undefined)
|
|
66
|
+
stdin.setRawMode(wasRaw);
|
|
67
|
+
stdin.pause();
|
|
68
|
+
stdout.write("\x1b[?25h");
|
|
69
|
+
resolve(result);
|
|
70
|
+
};
|
|
71
|
+
const guarded = (key) => {
|
|
72
|
+
try {
|
|
73
|
+
onKey(key, finish);
|
|
74
|
+
if (!settled)
|
|
75
|
+
repaint();
|
|
76
|
+
}
|
|
77
|
+
catch (e) {
|
|
78
|
+
process.stderr.write("key handling error: " + String(e));
|
|
79
|
+
}
|
|
80
|
+
};
|
|
81
|
+
const feed = createKeyFeed(guarded);
|
|
82
|
+
// The first paint gets the same armor as every repaint: a throwing
|
|
83
|
+
// frame builder degrades to a blank-but-alive session (keys still work,
|
|
84
|
+
// finish still restores the tty) instead of a hung promise and a leaked
|
|
85
|
+
// raw mode.
|
|
86
|
+
try {
|
|
87
|
+
repaint();
|
|
88
|
+
}
|
|
89
|
+
catch (e) {
|
|
90
|
+
process.stderr.write("initial paint failed: " + String(e) + "\n");
|
|
91
|
+
}
|
|
92
|
+
stdin.on("data", feed);
|
|
93
|
+
});
|
|
94
|
+
}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# Repair log
|
|
2
|
+
|
|
3
|
+
Every bug in the generated rules, in order. This log IS the generation-
|
|
4
|
+
reliability data: the product's claim is that an agent can repair these from
|
|
5
|
+
fixture failures, and every repair below was driven by a fixture or scan
|
|
6
|
+
failure, never by hand-tuning against the target repo.
|
|
7
|
+
|
|
8
|
+
## Inline-fixture-caught (rule fails to parse or fixture fails)
|
|
9
|
+
|
|
10
|
+
| # | Rule | Bug | Fix | Category |
|
|
11
|
+
|---|------|-----|-----|----------|
|
|
12
|
+
| 1 | fetch-without-signal | YAML: unquoted `pattern: signal: $SIG` (colon in value) | quote the pattern | YAML quoting |
|
|
13
|
+
| 2 | fetch-without-signal | `constraints` nested inside an `any` branch | `constraints` is a TOP-LEVEL rule-config field | API shape |
|
|
14
|
+
| 3 | no-empty-catch | multi-node pattern `try $T catch ($E) { }` rejected | kind-based rule on `catch_clause` | pattern vs kind choice |
|
|
15
|
+
| 4 | fetch-without-signal | `pattern: "signal: $SIG"` can't match an object pair key | match `property_identifier` kind + regex | tree-sitter kind names |
|
|
16
|
+
| 5 | map-async | `async $$$ARGS` — rest metavariable after a keyword doesn't parse | explicit arrow shapes: `async $A => $B`, `async ($$$A) => $B` | metavariable rules |
|
|
17
|
+
| 6 | no-empty-catch | `kind: statement` is not a real tree-sitter node → every catch flagged | regex on the block text `^\{\s*\}` | tree-sitter kind names |
|
|
18
|
+
| 7 | fetch-without-signal | `has` defaulted to `stopBy: neighbor` — key is a grandchild of the options object | `stopBy: end` | **stopBy defaults** |
|
|
19
|
+
| 8 | map-async | `inside` same neighbor problem (`map` → `arguments` → `Promise.all`) | `stopBy: end` | **stopBy defaults** |
|
|
20
|
+
| 9 | fetch-without-signal | `await fetch(...)` variants double-reported (inner call already matches) | drop await variants | dedup / node identity |
|
|
21
|
+
|
|
22
|
+
## Scan-level-fixture-caught (inline tests CANNOT catch these)
|
|
23
|
+
|
|
24
|
+
| # | Rule | Bug | Fix | Category |
|
|
25
|
+
|---|------|-----|-----|----------|
|
|
26
|
+
| 10 | no-direct-ai-imports | rule-level `globs:` with `!` negation silently does NOTHING | rule-level `files:` / `ignores:` are the real fields; sgconfig has no `ignores` | silent schema acceptance |
|
|
27
|
+
| 11 | harness design | inline test cases have no file path → glob scoping untestable inline | scan-level ground truth (sample-app) is REQUIRED as a second fixture layer | harness design finding |
|
|
28
|
+
|
|
29
|
+
## Meta-findings for the skill file (the product)
|
|
30
|
+
|
|
31
|
+
1. Nearly all bugs cluster into ~6 mechanical categories — none needed
|
|
32
|
+
creativity to fix. Encoding this list in the generation skill should
|
|
33
|
+
dramatically improve first-shot yield. **Generation reliability is a
|
|
34
|
+
teachable problem, not a research problem.**
|
|
35
|
+
2. `stopBy: end` should be the DEFAULT in generated relational rules
|
|
36
|
+
(ast-grep's `neighbor` default is almost never what you want).
|
|
37
|
+
3. ast-grep silently accepts unknown fields in both sgconfig.yml and rules
|
|
38
|
+
(bogus probe fields raised no error). A misspelled `ignores` produces a
|
|
39
|
+
rule that looks right and excludes nothing. **The harness must verify
|
|
40
|
+
behavior (scan-level fixtures), not just schema.**
|
|
41
|
+
4. Two fixture layers are mandatory: inline tests (node-level truth) +
|
|
42
|
+
scan-level seeded ground truth (file-scope truth, glob/ignore behavior,
|
|
43
|
+
double-report detection).
|
|
44
|
+
5. `sg test` discards generated snapshots on failed runs; baselines are
|
|
45
|
+
written with `sg test -U` and only trustworthy after a clean verify run.
|
package/docs/RESULTS.md
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
# Kill-test prototype results (2026-08-28)
|
|
2
|
+
|
|
3
|
+
Setup: 5 rules generated from intent-only one-liners ([intents.md](intents.md)),
|
|
4
|
+
fixtures via `sg test`, evaluated on (a) a seeded sample app with known ground
|
|
5
|
+
truth and (b) a real production repo (~234 TS files),
|
|
6
|
+
with every finding hand-classified. ast-grep 0.45.2.
|
|
7
|
+
|
|
8
|
+
## Sample app (seeded ground truth)
|
|
9
|
+
|
|
10
|
+
- **Recall: 7/7 planted violations found (100%)**
|
|
11
|
+
- **Precision: 8/8 (100%)** — the 8th finding was a *real* violation planted
|
|
12
|
+
by accident (an uncounted single-arg fetch) — correctly flagged
|
|
13
|
+
- All 6 clean lookalikes (gateway imports, signal-ed fetches, 2-arg z.record,
|
|
14
|
+
logger catch, Promise.all-wrapped map, catch-with-return-null) correctly silent
|
|
15
|
+
|
|
16
|
+
## Real repo (hand-verified findings)
|
|
17
|
+
|
|
18
|
+
| Rule | Findings | TP | FP | Precision |
|
|
19
|
+
|---|---|---|---|---|
|
|
20
|
+
| fetch-without-abort-signal `[R]` | 5 | 5 | 0 | **100%** |
|
|
21
|
+
| map-async-no-promise-all `[R]` | 3* | 0 | 3* | 0% |
|
|
22
|
+
|
|
23
|
+
\* The table shows the two rules with real-repo findings. `zod-record-two-args`
|
|
24
|
+
had no z.record usage in the target (fixtures only), and the two `[P]` import/
|
|
25
|
+
catch rules fired zero times on this repo (no openai imports; no empty catches
|
|
26
|
+
among 47 catch sites — spot-checked).
|
|
27
|
+
|
|
28
|
+
**All 3 false positives are ONE rule hitting its DECLARED blind spot**: the
|
|
29
|
+
map-async rule flags `const p = xs.map(async ...)` even when `p` is later
|
|
30
|
+
wrapped in `Promise.all(p)` / `Promise.allSettled(p)` on a subsequent line —
|
|
31
|
+
a cross-statement dataflow question that within-file relational operators
|
|
32
|
+
cannot answer.
|
|
33
|
+
|
|
34
|
+
## Verdict vs kill-test criteria
|
|
35
|
+
|
|
36
|
+
- ✅ End-to-end loop works: intent → generated rule + fixtures → `sg test` →
|
|
37
|
+
scan → evidence-bearing findings. Whole thing in one working session.
|
|
38
|
+
- ✅ Syntactic + simple-relational rules: precision 100% on real code.
|
|
39
|
+
- ⚠️ 90% precision bar: **not met by the weakest rule** (map-async = 0/3 on
|
|
40
|
+
real repo). The failure is precisely the analysis-tier boundary, not a
|
|
41
|
+
generation failure — the rule does what its tier can do.
|
|
42
|
+
- ✅ Generation reliability: 11 repairs across 5 rules, every one falling
|
|
43
|
+
into ~6 mechanical categories (see [REPAIR-LOG.md](REPAIR-LOG.md)) —
|
|
44
|
+
teachable to the generation skill.
|
|
45
|
+
|
|
46
|
+
## Product implications
|
|
47
|
+
|
|
48
|
+
1. **The premise is viable.** An agent can author fixture-tested ast-grep
|
|
49
|
+
rules that find real bugs in real code, and the fixture harness catches
|
|
50
|
+
generation bugs before they reach a user's CI.
|
|
51
|
+
2. **The tier boundary is real and now measured.** Within-file relational
|
|
52
|
+
rules are safe; dataflow-adjacent rules ("is this result ever awaited?")
|
|
53
|
+
need either the semantic tier (D4 trigger evidence, as designed) or must
|
|
54
|
+
be shipped as audit-mode rules with their blind spot declared (D7).
|
|
55
|
+
3. **Two fixture layers are required**, inline (`sg test`) and scan-level
|
|
56
|
+
seeded ground truth — inline tests cannot catch glob/scope/double-report
|
|
57
|
+
bugs (empirically demonstrated).
|
|
58
|
+
4. **ast-grep's silent schema acceptance is a footgun** the harness must
|
|
59
|
+
defend against (misspelled field = rule that excludes nothing and never
|
|
60
|
+
says so).
|
|
61
|
+
5. Cloudflare Code Mode: confirmed unnecessary for any of this (D2). The
|
|
62
|
+
entire loop ran locally with the user's own agent.
|
|
63
|
+
|
|
64
|
+
## Next
|
|
65
|
+
|
|
66
|
+
- Write the generation skill encoding the REPAIR-LOG's category lessons;
|
|
67
|
+
re-run this exact experiment measuring FIRST-shot yield (no repairs).
|
|
68
|
+
That number is the product's core metric.
|
|
69
|
+
- Real kill test on the Effect codebase (dogfood), per docs/kill-test.md.
|
|
70
|
+
- map-async rule is the concrete candidate to motivate the oxlint/oxc tier.
|