vigiles 12.7.0 → 13.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -15
- package/dist/audit-report.d.ts +38 -0
- package/dist/audit-report.js +13 -2
- package/dist/audit-report.template.html +44 -29
- package/dist/audit-score.js +9 -1
- package/dist/audit-verdict.d.ts +89 -0
- package/dist/audit-verdict.js +281 -0
- package/dist/cli.js +162 -13
- package/dist/core/orphans.d.ts +16 -6
- package/dist/core/orphans.js +45 -19
- package/dist/core/rule-meta.js +2 -2
- package/dist/core/types.d.ts +10 -4
- package/dist/eval-cost.d.ts +1 -1
- package/dist/eval.d.ts +2 -2
- package/dist/eval.js +1 -1
- package/dist/leaderboard.js +9 -3
- package/dist/rule-inventory.d.ts +90 -0
- package/dist/rule-inventory.js +327 -0
- package/dist/rule-routing.d.ts +46 -0
- package/dist/rule-routing.js +135 -0
- package/dist/scaffold-test.js +1 -1
- package/dist/segment.d.ts +33 -0
- package/dist/segment.js +454 -0
- package/package.json +1 -1
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { type LinterName } from "./rule-inventory.js";
|
|
2
|
+
/** How a routed rule would be enforced (a MECHANISM ladder, not a 1-10 score). */
|
|
3
|
+
export type RuleCategory = "reuse" | "hook" | "semantic" | "unrouted";
|
|
4
|
+
export type RuleMechanism = "config-line" | "hook" | "prose" | "compile";
|
|
5
|
+
/** One segmented, deterministically-routed rule with provenance. */
|
|
6
|
+
export interface RoutedRule {
|
|
7
|
+
/** Normalized atomic rule text (from the segmenter). */
|
|
8
|
+
readonly text: string;
|
|
9
|
+
/** Verbatim source slice — for a UI highlight. */
|
|
10
|
+
readonly quote: string;
|
|
11
|
+
readonly file: string | undefined;
|
|
12
|
+
readonly lineStart: number;
|
|
13
|
+
readonly lineEnd: number;
|
|
14
|
+
/** Segmenter confidence that this IS a rule (3/3 cues → high, 2/3 → medium). */
|
|
15
|
+
readonly confidence: "high" | "medium";
|
|
16
|
+
readonly category: RuleCategory;
|
|
17
|
+
readonly mechanism: RuleMechanism;
|
|
18
|
+
/** reuse only: the off-the-shelf rule that enforces it. */
|
|
19
|
+
readonly rule?: string;
|
|
20
|
+
/** reuse only: the linter that rule belongs to. */
|
|
21
|
+
readonly linter?: LinterName;
|
|
22
|
+
}
|
|
23
|
+
export interface RuleRouting {
|
|
24
|
+
/** How many atomic rules were routed (after the confidence filter). */
|
|
25
|
+
readonly segmented: number;
|
|
26
|
+
readonly counts: Record<RuleCategory, number>;
|
|
27
|
+
readonly rules: readonly RoutedRule[];
|
|
28
|
+
}
|
|
29
|
+
export interface RouteOptions {
|
|
30
|
+
/**
|
|
31
|
+
* Minimum segmenter confidence to route. The segmenter emits `high` (3/3 cues
|
|
32
|
+
* — an imperative rule) and `medium` (2/3). For the audit PREVIEW we default to
|
|
33
|
+
* `high` only: precision over recall. A doc-heavy instruction file (e.g. a
|
|
34
|
+
* keyFiles index of "`path` — description" bullets) trips the medium tier with
|
|
35
|
+
* non-rules, which would bury the real rules and overstate "unrouted". Pass
|
|
36
|
+
* `"medium"` to include both.
|
|
37
|
+
*/
|
|
38
|
+
readonly minConfidence?: "high" | "medium";
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Segment the instruction file and route every atomic rule deterministically.
|
|
42
|
+
* Pure: the caller passes the concatenated instruction text (and an optional
|
|
43
|
+
* source path for provenance). Returns per-category counts + the routed rules.
|
|
44
|
+
*/
|
|
45
|
+
export declare function routeRules(instructionText: string, file?: string, options?: RouteOptions): RuleRouting;
|
|
46
|
+
//# sourceMappingURL=rule-routing.d.ts.map
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.routeRules = routeRules;
|
|
4
|
+
/**
|
|
5
|
+
* rule-routing.ts — the deterministic (no-model) State-B routing PREVIEW.
|
|
6
|
+
*
|
|
7
|
+
* `rule-inventory.ts` answers a narrow question ("which prose lines name an
|
|
8
|
+
* off-the-shelf lint rule, and is it enabled?"). This goes one honest step
|
|
9
|
+
* further: it SEGMENTS the whole instruction file into atomic rules
|
|
10
|
+
* ({@link segmentInstructions}) and routes each one into the class that a real
|
|
11
|
+
* enforcement path would take — WITHOUT running a model:
|
|
12
|
+
*
|
|
13
|
+
* reuse → the rule text names an off-the-shelf lint rule ({@link INTENT_MAP})
|
|
14
|
+
* → mechanism: flip one config line.
|
|
15
|
+
* hook → an ACTION rule a linter can't see (git push, rm -rf, "before you
|
|
16
|
+
* commit") → mechanism: a pre-commit / PreToolUse hook.
|
|
17
|
+
* semantic → a judgment call ("readable", "single responsibility") no checker
|
|
18
|
+
* can honestly decide → mechanism: stays prose.
|
|
19
|
+
* unrouted → none of the above fired deterministically → mechanism: the opt-in
|
|
20
|
+
* `compile` tier routes it (reuse / synthesize / hook / prose).
|
|
21
|
+
*
|
|
22
|
+
* HONESTY BY CONSTRUCTION: the deterministic tier NEVER claims a rule is
|
|
23
|
+
* "synthesizable" — deciding that a custom rule can be written (and gating it)
|
|
24
|
+
* is exactly the work the opt-in model tier does. Everything this file can't
|
|
25
|
+
* pin to a concrete cue is `unrouted` ("compile to find out"), not a promise.
|
|
26
|
+
*
|
|
27
|
+
* Pure, deterministic, dependency-free. Reuses `rule-inventory`'s hardened
|
|
28
|
+
* whole-token matcher + `INTENT_MAP`, and `segment`'s Tier-A segmenter.
|
|
29
|
+
*/
|
|
30
|
+
const segment_js_1 = require("./segment.js");
|
|
31
|
+
const rule_inventory_js_1 = require("./rule-inventory.js");
|
|
32
|
+
/** The mechanism each category maps to — a fixed, honest ladder. */
|
|
33
|
+
const MECHANISM = {
|
|
34
|
+
reuse: "config-line",
|
|
35
|
+
hook: "hook",
|
|
36
|
+
semantic: "prose",
|
|
37
|
+
unrouted: "compile",
|
|
38
|
+
};
|
|
39
|
+
/**
|
|
40
|
+
* ACTION-rule cues — things a linter never sees (git, filesystem, shell,
|
|
41
|
+
* process). A hook is the right gate, not a lint rule. Ported from the compiler's
|
|
42
|
+
* classifier; deliberately specific so it doesn't grab a lint rule that merely
|
|
43
|
+
* mentions a file.
|
|
44
|
+
*/
|
|
45
|
+
const HOOK_CUES = [
|
|
46
|
+
/\bgit\s+push\b/i,
|
|
47
|
+
// "push … to main/master/prod" — tolerate backticks/adverbs between (real
|
|
48
|
+
// phrasings: "push directly to `main`", "pushing straight to master").
|
|
49
|
+
/\bpush\w*\b[^.\n]{0,24}\b(main|master|prod)\b/i,
|
|
50
|
+
/\bforce[- ]?push/i,
|
|
51
|
+
/--no-verify/i,
|
|
52
|
+
/\bnever\s+commit\b/i,
|
|
53
|
+
// "before you/each/every commit", "before committing".
|
|
54
|
+
/\bbefore\s+(you\s+|each\s+|every\s+)?commit(ting)?\b/i,
|
|
55
|
+
/\brun\b[^.\n]{0,20}\btests?\b[^.\n]{0,14}\bbefore\b/i,
|
|
56
|
+
/\bsigned-off-by\b/i,
|
|
57
|
+
/\b(don'?t|do not|never)\s+edit\b.*\b(generated|\.pb\.|_mock|proto-gen|lock)/i,
|
|
58
|
+
/\bgenerated\s+files?\b/i,
|
|
59
|
+
/\bco[- ]?authored[- ]?by\b/i,
|
|
60
|
+
/\brm\s+-rf\b/i,
|
|
61
|
+
/\bcurl\b.*\|\s*(sh|bash)/i,
|
|
62
|
+
/\bchmod\b/i,
|
|
63
|
+
];
|
|
64
|
+
/**
|
|
65
|
+
* Judgment / no-checker cues — a rule no linter can honestly decide, so it stays
|
|
66
|
+
* labeled prose. Ported from the compiler's classifier (the static markers only —
|
|
67
|
+
* no ruleMap dependency, to keep this file model-free and dep-free).
|
|
68
|
+
*/
|
|
69
|
+
const SEMANTIC_CUES = [
|
|
70
|
+
/\bself[- ]?documenting\b/i,
|
|
71
|
+
/\bclear(er)?\s+(names?|code|over clever)/i,
|
|
72
|
+
/\breadable\b/i,
|
|
73
|
+
/\bkeep it simple\b/i,
|
|
74
|
+
/\bover[- ]?engineer/i,
|
|
75
|
+
/\bsingle responsibility\b/i,
|
|
76
|
+
/\bcomposition over inheritance\b/i,
|
|
77
|
+
/\bmeaningful\b/i,
|
|
78
|
+
/\bidiomatic\b/i,
|
|
79
|
+
/\bwhere (it )?makes sense\b/i,
|
|
80
|
+
/\bappropriate(ly)?\b/i,
|
|
81
|
+
/\bsolid\s+principles?\b/i,
|
|
82
|
+
/\bbest practices?\b/i,
|
|
83
|
+
/\bclean code\b/i,
|
|
84
|
+
];
|
|
85
|
+
/**
|
|
86
|
+
* Route one atomic rule. Order matters: an ACTION cue (git push) wins over a
|
|
87
|
+
* rule-name mention ("never commit console.log" is a hook, not a lint rule);
|
|
88
|
+
* reuse (a concrete off-the-shelf rule) wins over a soft semantic cue.
|
|
89
|
+
*/
|
|
90
|
+
function classify(text) {
|
|
91
|
+
if (HOOK_CUES.some((re) => re.test(text)))
|
|
92
|
+
return { category: "hook" };
|
|
93
|
+
for (const m of rule_inventory_js_1.INTENT_MAP) {
|
|
94
|
+
if (m.keywords.some((kw) => (0, rule_inventory_js_1.matchesWholeToken)(text, kw))) {
|
|
95
|
+
return { category: "reuse", rule: m.rule, linter: m.linter };
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
if (SEMANTIC_CUES.some((re) => re.test(text)))
|
|
99
|
+
return { category: "semantic" };
|
|
100
|
+
return { category: "unrouted" };
|
|
101
|
+
}
|
|
102
|
+
/**
|
|
103
|
+
* Segment the instruction file and route every atomic rule deterministically.
|
|
104
|
+
* Pure: the caller passes the concatenated instruction text (and an optional
|
|
105
|
+
* source path for provenance). Returns per-category counts + the routed rules.
|
|
106
|
+
*/
|
|
107
|
+
function routeRules(instructionText, file, options = {}) {
|
|
108
|
+
const minConfidence = options.minConfidence ?? "high";
|
|
109
|
+
const segments = (0, segment_js_1.segmentInstructions)(instructionText, file).filter((s) => minConfidence === "medium" || s.confidence === "high");
|
|
110
|
+
const rules = segments.map((s) => {
|
|
111
|
+
const c = classify(s.text);
|
|
112
|
+
return {
|
|
113
|
+
text: s.text,
|
|
114
|
+
quote: s.exactQuote,
|
|
115
|
+
file: s.file,
|
|
116
|
+
lineStart: s.lineStart,
|
|
117
|
+
lineEnd: s.lineEnd,
|
|
118
|
+
confidence: s.confidence,
|
|
119
|
+
category: c.category,
|
|
120
|
+
mechanism: MECHANISM[c.category],
|
|
121
|
+
...(c.rule ? { rule: c.rule } : {}),
|
|
122
|
+
...(c.linter ? { linter: c.linter } : {}),
|
|
123
|
+
};
|
|
124
|
+
});
|
|
125
|
+
const counts = {
|
|
126
|
+
reuse: 0,
|
|
127
|
+
hook: 0,
|
|
128
|
+
semantic: 0,
|
|
129
|
+
unrouted: 0,
|
|
130
|
+
};
|
|
131
|
+
for (const r of rules)
|
|
132
|
+
counts[r.category]++;
|
|
133
|
+
return { segmented: segments.length, counts, rules };
|
|
134
|
+
}
|
|
135
|
+
//# sourceMappingURL=rule-routing.js.map
|
package/dist/scaffold-test.js
CHANGED
|
@@ -210,7 +210,7 @@ function safetySection(input, sideEffecting) {
|
|
|
210
210
|
// --- Safety (deterministic) — generated from ${input.name}'s side-effecting tools: ${sideEffecting.join(", ")} ---
|
|
211
211
|
// In a real run, replace this constructed Trace with a real \`runHarness\` /
|
|
212
212
|
// \`measure\` turn (use interceptTools so a real model's attempt is DENIED, never
|
|
213
|
-
// executed — see
|
|
213
|
+
// executed — see research/eval-architecture.md). The checks below are derived from the
|
|
214
214
|
// declared tools contract — the agent's "hole" asserted to stay in its lane.
|
|
215
215
|
{
|
|
216
216
|
const trace = {
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* segment.ts — Tier-A deterministic (no-model) segmenter.
|
|
3
|
+
*
|
|
4
|
+
* Splits a CLAUDE.md / AGENTS.md into ATOMIC candidate rules with provenance.
|
|
5
|
+
* Pure, deterministic, strict TS, no external deps.
|
|
6
|
+
*
|
|
7
|
+
* Design bias: PRECISION over recall. A missed rule costs a row; a garbage
|
|
8
|
+
* atom costs credibility. When in doubt we UNDER-split and REJECT.
|
|
9
|
+
*/
|
|
10
|
+
/** A single atomic candidate rule extracted from an instructions file. */
|
|
11
|
+
export interface SegmentedRule {
|
|
12
|
+
/** Normalized rule text (bullet marker stripped, continuation joined, whitespace collapsed). */
|
|
13
|
+
text: string;
|
|
14
|
+
/** Source file path (as supplied by the caller), or undefined. */
|
|
15
|
+
file: string | undefined;
|
|
16
|
+
/** 1-based inclusive start line in the source. */
|
|
17
|
+
lineStart: number;
|
|
18
|
+
/** 1-based inclusive end line in the source. */
|
|
19
|
+
lineEnd: number;
|
|
20
|
+
/** Verbatim slice of the source spanning [lineStart..lineEnd] — for UI highlight. */
|
|
21
|
+
exactQuote: string;
|
|
22
|
+
/** 3/3 cues => "high"; 2/3 => "medium". (Rejected candidates are never emitted.) */
|
|
23
|
+
confidence: "high" | "medium";
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Split a CLAUDE.md / AGENTS.md into atomic candidate rules with provenance.
|
|
27
|
+
*
|
|
28
|
+
* Deterministic Tier-A heuristic. Code fences and tables are excluded from
|
|
29
|
+
* candidacy. Candidate units are (a) list items with attached continuation
|
|
30
|
+
* lines and (b) sentences of paragraphs under a rule-ish heading.
|
|
31
|
+
*/
|
|
32
|
+
export declare function segmentInstructions(markdown: string, file?: string): SegmentedRule[];
|
|
33
|
+
//# sourceMappingURL=segment.d.ts.map
|
package/dist/segment.js
ADDED
|
@@ -0,0 +1,454 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* segment.ts — Tier-A deterministic (no-model) segmenter.
|
|
4
|
+
*
|
|
5
|
+
* Splits a CLAUDE.md / AGENTS.md into ATOMIC candidate rules with provenance.
|
|
6
|
+
* Pure, deterministic, strict TS, no external deps.
|
|
7
|
+
*
|
|
8
|
+
* Design bias: PRECISION over recall. A missed rule costs a row; a garbage
|
|
9
|
+
* atom costs credibility. When in doubt we UNDER-split and REJECT.
|
|
10
|
+
*/
|
|
11
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
12
|
+
exports.segmentInstructions = segmentInstructions;
|
|
13
|
+
// --- Heuristic vocabulary --------------------------------------------------
|
|
14
|
+
/** Imperative/prohibitive head the candidate must START with (form cue). */
|
|
15
|
+
const FORM_HEAD = /^(?:use|avoid|prefer|never|always|don'?t|do not|no\s+\S|must|should|keep|run|write|add|remove|only)\b/i;
|
|
16
|
+
/** Rule-ish heading gate for prose-under-heading candidacy. */
|
|
17
|
+
const RULE_HEADING = /rules?|conventions?|style|guidelines?|standards?|do(?:n'?ts?)?s?|never|always|must|require/i;
|
|
18
|
+
/** Declarative subjects — these signal a statement, not an instruction. */
|
|
19
|
+
const DECLARATION = /^(?:this|these|those|it|we|our|there)\b/i;
|
|
20
|
+
/** Line consisting only of a bare URL. */
|
|
21
|
+
const URL_ONLY = /^<?https?:\/\/\S+>?$/;
|
|
22
|
+
/** Line consisting only of a markdown link. */
|
|
23
|
+
const LINK_ONLY = /^\[[^\]]*\]\([^)]*\)$/;
|
|
24
|
+
/** Verb-ish lexicon (secondary shape signal). Kept curated for precision. */
|
|
25
|
+
const VERBS = new Set([
|
|
26
|
+
"use",
|
|
27
|
+
"uses",
|
|
28
|
+
"using",
|
|
29
|
+
"used",
|
|
30
|
+
"avoid",
|
|
31
|
+
"avoids",
|
|
32
|
+
"prefer",
|
|
33
|
+
"prefers",
|
|
34
|
+
"run",
|
|
35
|
+
"runs",
|
|
36
|
+
"write",
|
|
37
|
+
"writes",
|
|
38
|
+
"writing",
|
|
39
|
+
"add",
|
|
40
|
+
"adds",
|
|
41
|
+
"remove",
|
|
42
|
+
"removes",
|
|
43
|
+
"keep",
|
|
44
|
+
"keeps",
|
|
45
|
+
"import",
|
|
46
|
+
"imports",
|
|
47
|
+
"importing",
|
|
48
|
+
"split",
|
|
49
|
+
"splits",
|
|
50
|
+
"push",
|
|
51
|
+
"pushes",
|
|
52
|
+
"commit",
|
|
53
|
+
"commits",
|
|
54
|
+
"test",
|
|
55
|
+
"tests",
|
|
56
|
+
"call",
|
|
57
|
+
"calls",
|
|
58
|
+
"set",
|
|
59
|
+
"sets",
|
|
60
|
+
"make",
|
|
61
|
+
"makes",
|
|
62
|
+
"create",
|
|
63
|
+
"creates",
|
|
64
|
+
"delete",
|
|
65
|
+
"deletes",
|
|
66
|
+
"update",
|
|
67
|
+
"updates",
|
|
68
|
+
"check",
|
|
69
|
+
"checks",
|
|
70
|
+
"ensure",
|
|
71
|
+
"ensures",
|
|
72
|
+
"document",
|
|
73
|
+
"documents",
|
|
74
|
+
"follow",
|
|
75
|
+
"follows",
|
|
76
|
+
"handle",
|
|
77
|
+
"handles",
|
|
78
|
+
"return",
|
|
79
|
+
"returns",
|
|
80
|
+
"throw",
|
|
81
|
+
"throws",
|
|
82
|
+
"catch",
|
|
83
|
+
"log",
|
|
84
|
+
"logs",
|
|
85
|
+
"prefix",
|
|
86
|
+
"name",
|
|
87
|
+
"names",
|
|
88
|
+
"store",
|
|
89
|
+
"stores",
|
|
90
|
+
"read",
|
|
91
|
+
"reads",
|
|
92
|
+
"save",
|
|
93
|
+
"saves",
|
|
94
|
+
"wrap",
|
|
95
|
+
"wraps",
|
|
96
|
+
"escape",
|
|
97
|
+
"escapes",
|
|
98
|
+
"match",
|
|
99
|
+
"matches",
|
|
100
|
+
"filter",
|
|
101
|
+
"filters",
|
|
102
|
+
"merge",
|
|
103
|
+
"merges",
|
|
104
|
+
"be",
|
|
105
|
+
"is",
|
|
106
|
+
"are",
|
|
107
|
+
"have",
|
|
108
|
+
"has",
|
|
109
|
+
"may",
|
|
110
|
+
"should",
|
|
111
|
+
"must",
|
|
112
|
+
"pin",
|
|
113
|
+
"pins",
|
|
114
|
+
"lint",
|
|
115
|
+
"format",
|
|
116
|
+
"formats",
|
|
117
|
+
"sort",
|
|
118
|
+
"group",
|
|
119
|
+
"groups",
|
|
120
|
+
"export",
|
|
121
|
+
"exports",
|
|
122
|
+
"mock",
|
|
123
|
+
"stub",
|
|
124
|
+
"assert",
|
|
125
|
+
"validate",
|
|
126
|
+
"validates",
|
|
127
|
+
"sanitize",
|
|
128
|
+
"encode",
|
|
129
|
+
"decode",
|
|
130
|
+
"hash",
|
|
131
|
+
"sign",
|
|
132
|
+
"verify",
|
|
133
|
+
"verifies",
|
|
134
|
+
"expose",
|
|
135
|
+
"hide",
|
|
136
|
+
"close",
|
|
137
|
+
"open",
|
|
138
|
+
"load",
|
|
139
|
+
"loads",
|
|
140
|
+
"fetch",
|
|
141
|
+
"fetches",
|
|
142
|
+
"render",
|
|
143
|
+
"renders",
|
|
144
|
+
"mount",
|
|
145
|
+
"bind",
|
|
146
|
+
"inject",
|
|
147
|
+
"register",
|
|
148
|
+
"resolve",
|
|
149
|
+
"reject",
|
|
150
|
+
"await",
|
|
151
|
+
"apply",
|
|
152
|
+
"applies",
|
|
153
|
+
"bump",
|
|
154
|
+
"tag",
|
|
155
|
+
"branch",
|
|
156
|
+
"rebase",
|
|
157
|
+
"squash",
|
|
158
|
+
"enforce",
|
|
159
|
+
"enforces",
|
|
160
|
+
"define",
|
|
161
|
+
"defines",
|
|
162
|
+
"declare",
|
|
163
|
+
"place",
|
|
164
|
+
"put",
|
|
165
|
+
"prefer",
|
|
166
|
+
]);
|
|
167
|
+
// --- Offset / line utilities ----------------------------------------------
|
|
168
|
+
function computeLineOffsets(lines) {
|
|
169
|
+
const offsets = new Array(lines.length);
|
|
170
|
+
let acc = 0;
|
|
171
|
+
for (let i = 0; i < lines.length; i++) {
|
|
172
|
+
offsets[i] = acc;
|
|
173
|
+
acc += lines[i].length + 1; // +1 for the '\n' consumed by split
|
|
174
|
+
}
|
|
175
|
+
return offsets;
|
|
176
|
+
}
|
|
177
|
+
function offsetToLine(lineOffsets, off) {
|
|
178
|
+
// 1-based line number containing char offset `off`.
|
|
179
|
+
let lo = 0;
|
|
180
|
+
let hi = lineOffsets.length - 1;
|
|
181
|
+
let ans = 0;
|
|
182
|
+
while (lo <= hi) {
|
|
183
|
+
const mid = (lo + hi) >> 1;
|
|
184
|
+
if (lineOffsets[mid] <= off) {
|
|
185
|
+
ans = mid;
|
|
186
|
+
lo = mid + 1;
|
|
187
|
+
}
|
|
188
|
+
else {
|
|
189
|
+
hi = mid - 1;
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
return ans + 1;
|
|
193
|
+
}
|
|
194
|
+
function normalize(s) {
|
|
195
|
+
return s.replace(/\s+/g, " ").trim();
|
|
196
|
+
}
|
|
197
|
+
function hasVerbish(text) {
|
|
198
|
+
const tokens = text
|
|
199
|
+
.toLowerCase()
|
|
200
|
+
.replace(/`[^`]*`/g, " ") // drop inline code spans
|
|
201
|
+
.split(/[^a-z']+/)
|
|
202
|
+
.filter(Boolean);
|
|
203
|
+
for (const t of tokens) {
|
|
204
|
+
if (VERBS.has(t))
|
|
205
|
+
return true;
|
|
206
|
+
}
|
|
207
|
+
return false;
|
|
208
|
+
}
|
|
209
|
+
function isLinkOnly(text) {
|
|
210
|
+
const t = text.trim();
|
|
211
|
+
return URL_ONLY.test(t) || LINK_ONLY.test(t);
|
|
212
|
+
}
|
|
213
|
+
/**
|
|
214
|
+
* Score the 3 cues. Returns confidence or null (reject).
|
|
215
|
+
* - form: starts with an imperative/prohibitive head (or "No X").
|
|
216
|
+
* - context: is a bullet OR sits under a rule-ish heading.
|
|
217
|
+
* - shape: 15–300 chars, has a verb-ish token, not link-only, not a declaration.
|
|
218
|
+
*/
|
|
219
|
+
function gate(text, isBullet, underRuleHeading) {
|
|
220
|
+
const t = text.trim();
|
|
221
|
+
const form = FORM_HEAD.test(t);
|
|
222
|
+
const context = isBullet || underRuleHeading;
|
|
223
|
+
const shape = t.length >= 15 &&
|
|
224
|
+
t.length <= 300 &&
|
|
225
|
+
hasVerbish(t) &&
|
|
226
|
+
!isLinkOnly(t) &&
|
|
227
|
+
!DECLARATION.test(t);
|
|
228
|
+
const cues = (form ? 1 : 0) + (context ? 1 : 0) + (shape ? 1 : 0);
|
|
229
|
+
if (cues >= 3)
|
|
230
|
+
return "high";
|
|
231
|
+
if (cues === 2)
|
|
232
|
+
return "medium";
|
|
233
|
+
return null;
|
|
234
|
+
}
|
|
235
|
+
// --- Atomicity split -------------------------------------------------------
|
|
236
|
+
/** Never split when an exception clause carries polarity/meaning. */
|
|
237
|
+
const HAS_EXCEPT = /\bexcept\b/i;
|
|
238
|
+
function trimSpan(src, span) {
|
|
239
|
+
let { start, end } = span;
|
|
240
|
+
while (start < end && /\s/.test(src[start]))
|
|
241
|
+
start++;
|
|
242
|
+
while (end > start && /\s/.test(src[end - 1]))
|
|
243
|
+
end--;
|
|
244
|
+
return { start, end };
|
|
245
|
+
}
|
|
246
|
+
/**
|
|
247
|
+
* Try to split a single-line bullet's content span on ';' or sentence
|
|
248
|
+
* boundaries. Returns the resulting spans ONLY IF there is >1 and every
|
|
249
|
+
* piece independently passes the gate; otherwise returns [whole].
|
|
250
|
+
*/
|
|
251
|
+
function atomize(src, contentSpan, isBullet, underRuleHeading) {
|
|
252
|
+
const whole = trimSpan(src, contentSpan);
|
|
253
|
+
const wholeText = src.slice(whole.start, whole.end);
|
|
254
|
+
if (HAS_EXCEPT.test(wholeText))
|
|
255
|
+
return [whole];
|
|
256
|
+
// Candidate cut points: ';' and sentence terminators followed by a capital.
|
|
257
|
+
const cuts = [];
|
|
258
|
+
for (let i = whole.start; i < whole.end; i++) {
|
|
259
|
+
const c = src[i];
|
|
260
|
+
if (c === ";") {
|
|
261
|
+
cuts.push(i + 1);
|
|
262
|
+
}
|
|
263
|
+
else if (c === "." || c === "!" || c === "?") {
|
|
264
|
+
// sentence boundary: terminator + whitespace + capital letter
|
|
265
|
+
const rest = src.slice(i + 1, whole.end);
|
|
266
|
+
const m = /^\s+[A-Z]/.exec(rest);
|
|
267
|
+
if (m)
|
|
268
|
+
cuts.push(i + 1);
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
if (cuts.length === 0)
|
|
272
|
+
return [whole];
|
|
273
|
+
const bounds = [whole.start, ...cuts, whole.end];
|
|
274
|
+
const pieces = [];
|
|
275
|
+
for (let i = 0; i < bounds.length - 1; i++) {
|
|
276
|
+
const piece = trimSpan(src, { start: bounds[i], end: bounds[i + 1] });
|
|
277
|
+
// strip a leading semicolon left by the cut
|
|
278
|
+
while (piece.start < piece.end &&
|
|
279
|
+
(src[piece.start] === ";" || /\s/.test(src[piece.start]))) {
|
|
280
|
+
piece.start++;
|
|
281
|
+
}
|
|
282
|
+
if (piece.start >= piece.end)
|
|
283
|
+
return [whole];
|
|
284
|
+
pieces.push(piece);
|
|
285
|
+
}
|
|
286
|
+
// Both/all halves must independently pass the gate, else keep whole.
|
|
287
|
+
for (const p of pieces) {
|
|
288
|
+
const text = normalize(src.slice(p.start, p.end));
|
|
289
|
+
if (gate(text, isBullet, underRuleHeading) === null)
|
|
290
|
+
return [whole];
|
|
291
|
+
}
|
|
292
|
+
return pieces.length > 1 ? pieces : [whole];
|
|
293
|
+
}
|
|
294
|
+
// --- Emission --------------------------------------------------------------
|
|
295
|
+
function emitFromSpan(src, lineOffsets, file, span, confidence) {
|
|
296
|
+
const exactQuote = src.slice(span.start, span.end);
|
|
297
|
+
return {
|
|
298
|
+
text: normalize(exactQuote),
|
|
299
|
+
file,
|
|
300
|
+
lineStart: offsetToLine(lineOffsets, span.start),
|
|
301
|
+
lineEnd: offsetToLine(lineOffsets, span.end - 1),
|
|
302
|
+
exactQuote,
|
|
303
|
+
confidence,
|
|
304
|
+
};
|
|
305
|
+
}
|
|
306
|
+
// --- Scanner ---------------------------------------------------------------
|
|
307
|
+
const LIST_ITEM = /^(\s*)([-*+])(\s+)(.*)$/;
|
|
308
|
+
const HEADING = /^(#{1,6})\s+(.*)$/;
|
|
309
|
+
const FENCE = /^\s*(```|~~~)/;
|
|
310
|
+
const TABLE_LINE = /^\s*\|/;
|
|
311
|
+
/**
|
|
312
|
+
* Split a CLAUDE.md / AGENTS.md into atomic candidate rules with provenance.
|
|
313
|
+
*
|
|
314
|
+
* Deterministic Tier-A heuristic. Code fences and tables are excluded from
|
|
315
|
+
* candidacy. Candidate units are (a) list items with attached continuation
|
|
316
|
+
* lines and (b) sentences of paragraphs under a rule-ish heading.
|
|
317
|
+
*/
|
|
318
|
+
function segmentInstructions(markdown, file) {
|
|
319
|
+
const lines = markdown.split("\n");
|
|
320
|
+
const lineOffsets = computeLineOffsets(lines);
|
|
321
|
+
const out = [];
|
|
322
|
+
let inFence = false;
|
|
323
|
+
let currentHeadingIsRuleish = false;
|
|
324
|
+
let i = 0;
|
|
325
|
+
const lineSpan = (a, b) => ({
|
|
326
|
+
start: lineOffsets[a],
|
|
327
|
+
end: lineOffsets[b] + lines[b].length,
|
|
328
|
+
});
|
|
329
|
+
while (i < lines.length) {
|
|
330
|
+
const line = lines[i];
|
|
331
|
+
// Code fences: toggle and skip everything inside (incl. the fence lines).
|
|
332
|
+
if (FENCE.test(line)) {
|
|
333
|
+
inFence = !inFence;
|
|
334
|
+
i++;
|
|
335
|
+
continue;
|
|
336
|
+
}
|
|
337
|
+
if (inFence) {
|
|
338
|
+
i++;
|
|
339
|
+
continue;
|
|
340
|
+
}
|
|
341
|
+
// Headings: update rule-ish context, not a candidate.
|
|
342
|
+
const h = HEADING.exec(line);
|
|
343
|
+
if (h) {
|
|
344
|
+
currentHeadingIsRuleish = RULE_HEADING.test(h[2]);
|
|
345
|
+
i++;
|
|
346
|
+
continue;
|
|
347
|
+
}
|
|
348
|
+
// Tables: excluded from candidacy.
|
|
349
|
+
if (TABLE_LINE.test(line)) {
|
|
350
|
+
i++;
|
|
351
|
+
continue;
|
|
352
|
+
}
|
|
353
|
+
// List items (with attached continuation lines).
|
|
354
|
+
const li = LIST_ITEM.exec(line);
|
|
355
|
+
if (li) {
|
|
356
|
+
const markerIndent = li[1].length;
|
|
357
|
+
const contentCol = li[1].length + li[2].length + li[3].length;
|
|
358
|
+
const startLine = i;
|
|
359
|
+
// Gather continuation lines: deeper-indented, non-blank, not a new
|
|
360
|
+
// list marker, not a heading, not a fence.
|
|
361
|
+
let endLine = i;
|
|
362
|
+
let j = i + 1;
|
|
363
|
+
while (j < lines.length) {
|
|
364
|
+
const cand = lines[j];
|
|
365
|
+
if (cand.trim() === "")
|
|
366
|
+
break;
|
|
367
|
+
if (FENCE.test(cand))
|
|
368
|
+
break;
|
|
369
|
+
if (HEADING.test(cand))
|
|
370
|
+
break;
|
|
371
|
+
const indent = cand.length - cand.trimStart().length;
|
|
372
|
+
if (indent <= markerIndent)
|
|
373
|
+
break;
|
|
374
|
+
if (LIST_ITEM.test(cand))
|
|
375
|
+
break; // nested/sibling bullet => separate candidate
|
|
376
|
+
endLine = j;
|
|
377
|
+
j++;
|
|
378
|
+
}
|
|
379
|
+
const multiLine = endLine > startLine;
|
|
380
|
+
const contentStart = lineOffsets[startLine] + contentCol;
|
|
381
|
+
const contentEnd = lineOffsets[endLine] + lines[endLine].length;
|
|
382
|
+
const contentSpan = { start: contentStart, end: contentEnd };
|
|
383
|
+
const wholeText = normalize(markdown.slice(contentStart, contentEnd));
|
|
384
|
+
const conf = gate(wholeText, true, currentHeadingIsRuleish);
|
|
385
|
+
if (conf !== null) {
|
|
386
|
+
// Only attempt splitting for single-line items (keeps offsets exact).
|
|
387
|
+
const spans = multiLine
|
|
388
|
+
? [trimSpan(markdown, contentSpan)]
|
|
389
|
+
: atomize(markdown, contentSpan, true, currentHeadingIsRuleish);
|
|
390
|
+
if (spans.length === 1) {
|
|
391
|
+
// Emit whole item; exactQuote is the full source span incl. marker.
|
|
392
|
+
out.push(emitFromSpan(markdown, lineOffsets, file, lineSpan(startLine, endLine), conf));
|
|
393
|
+
}
|
|
394
|
+
else {
|
|
395
|
+
for (const s of spans) {
|
|
396
|
+
const text = normalize(markdown.slice(s.start, s.end));
|
|
397
|
+
const c = gate(text, true, currentHeadingIsRuleish);
|
|
398
|
+
if (c !== null)
|
|
399
|
+
out.push(emitFromSpan(markdown, lineOffsets, file, s, c));
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
}
|
|
403
|
+
i = endLine + 1;
|
|
404
|
+
continue;
|
|
405
|
+
}
|
|
406
|
+
// Paragraph block: accumulate until blank / heading / list / fence / table.
|
|
407
|
+
if (line.trim() !== "") {
|
|
408
|
+
const startLine = i;
|
|
409
|
+
let endLine = i;
|
|
410
|
+
let j = i + 1;
|
|
411
|
+
while (j < lines.length) {
|
|
412
|
+
const cand = lines[j];
|
|
413
|
+
if (cand.trim() === "")
|
|
414
|
+
break;
|
|
415
|
+
if (FENCE.test(cand))
|
|
416
|
+
break;
|
|
417
|
+
if (HEADING.test(cand))
|
|
418
|
+
break;
|
|
419
|
+
if (LIST_ITEM.test(cand))
|
|
420
|
+
break;
|
|
421
|
+
if (TABLE_LINE.test(cand))
|
|
422
|
+
break;
|
|
423
|
+
endLine = j;
|
|
424
|
+
j++;
|
|
425
|
+
}
|
|
426
|
+
// Prose is only a candidate under a rule-ish heading.
|
|
427
|
+
if (currentHeadingIsRuleish) {
|
|
428
|
+
const paraStart = lineOffsets[startLine];
|
|
429
|
+
const paraEnd = lineOffsets[endLine] + lines[endLine].length;
|
|
430
|
+
const paraText = markdown.slice(paraStart, paraEnd);
|
|
431
|
+
// Sentence spans preserving absolute offsets.
|
|
432
|
+
const re = /[^.!?]+[.!?]+(\s|$)|[^.!?]+$/g;
|
|
433
|
+
let m;
|
|
434
|
+
while ((m = re.exec(paraText)) !== null) {
|
|
435
|
+
const s = trimSpan(markdown, {
|
|
436
|
+
start: paraStart + m.index,
|
|
437
|
+
end: paraStart + m.index + m[0].length,
|
|
438
|
+
});
|
|
439
|
+
if (s.start >= s.end)
|
|
440
|
+
continue;
|
|
441
|
+
const text = normalize(markdown.slice(s.start, s.end));
|
|
442
|
+
const c = gate(text, false, true);
|
|
443
|
+
if (c !== null)
|
|
444
|
+
out.push(emitFromSpan(markdown, lineOffsets, file, s, c));
|
|
445
|
+
}
|
|
446
|
+
}
|
|
447
|
+
i = endLine + 1;
|
|
448
|
+
continue;
|
|
449
|
+
}
|
|
450
|
+
i++;
|
|
451
|
+
}
|
|
452
|
+
return out;
|
|
453
|
+
}
|
|
454
|
+
//# sourceMappingURL=segment.js.map
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "vigiles",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "13.0.0",
|
|
4
4
|
"description": "Lint & test the harness your AI agent runs on — verify the references in your CLAUDE.md / AGENTS.md and test that your hooks and skills actually work.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude-code",
|