@dzhechkov/harness-core 0.8.10 → 0.8.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +120 -60
- package/dist/codex-invoke.d.ts +73 -0
- package/dist/codex-invoke.d.ts.map +1 -0
- package/dist/codex-invoke.js +80 -0
- package/dist/codex-invoke.js.map +1 -0
- package/dist/discrimination-gate.d.ts +63 -3
- package/dist/discrimination-gate.d.ts.map +1 -1
- package/dist/discrimination-gate.js +113 -16
- package/dist/discrimination-gate.js.map +1 -1
- package/dist/event-chain.d.ts +30 -0
- package/dist/event-chain.d.ts.map +1 -1
- package/dist/event-chain.js +24 -0
- package/dist/event-chain.js.map +1 -1
- package/dist/guard.d.ts +8 -0
- package/dist/guard.d.ts.map +1 -1
- package/dist/guard.js +37 -0
- package/dist/guard.js.map +1 -1
- package/dist/index.d.ts +8 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +8 -4
- package/dist/index.js.map +1 -1
- package/dist/mutation-gate.d.ts +39 -36
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +111 -5
- package/dist/mutation-gate.js.map +1 -1
- package/dist/operations.d.ts.map +1 -1
- package/dist/operations.js +8 -5
- package/dist/operations.js.map +1 -1
- package/dist/plugin.d.ts.map +1 -1
- package/dist/plugin.js +27 -5
- package/dist/plugin.js.map +1 -1
- package/dist/recommend.d.ts +4 -5
- package/dist/recommend.d.ts.map +1 -1
- package/dist/recommend.js +110 -45
- package/dist/recommend.js.map +1 -1
- package/dist/registry.d.ts +32 -1
- package/dist/registry.d.ts.map +1 -1
- package/dist/registry.js +165 -9
- package/dist/registry.js.map +1 -1
- package/dist/run-records.d.ts +3 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +18 -0
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts +95 -0
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +274 -2
- package/dist/score.js.map +1 -1
- package/dist/skill-selection.d.ts +72 -0
- package/dist/skill-selection.d.ts.map +1 -0
- package/dist/skill-selection.js +76 -0
- package/dist/skill-selection.js.map +1 -0
- package/dist/stem.d.ts +12 -0
- package/dist/stem.d.ts.map +1 -0
- package/dist/stem.js +89 -0
- package/dist/stem.js.map +1 -0
- package/dist/telemetry-vocabulary.d.ts +7 -0
- package/dist/telemetry-vocabulary.d.ts.map +1 -1
- package/dist/telemetry-vocabulary.js +29 -0
- package/dist/telemetry-vocabulary.js.map +1 -1
- package/package.json +8 -8
- package/sbom.json +209 -59
- package/src/codex-invoke.ts +138 -0
- package/src/discrimination-gate.ts +183 -19
- package/src/event-chain.ts +41 -0
- package/src/guard.ts +40 -0
- package/src/index.ts +10 -4
- package/src/mutation-gate.ts +165 -5
- package/src/operations.ts +8 -5
- package/src/plugin.ts +27 -5
- package/src/recommend.ts +116 -46
- package/src/registry.ts +144 -11
- package/src/run-records.ts +23 -0
- package/src/score.ts +361 -3
- package/src/skill-selection.ts +111 -0
- package/src/stem.ts +87 -0
- package/src/telemetry-vocabulary.ts +36 -0
package/src/registry.ts
CHANGED
|
@@ -8,9 +8,11 @@
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
import { existsSync, readdirSync, readFileSync, realpathSync, statSync, type Dirent } from 'node:fs';
|
|
11
|
-
import { basename, dirname, join, resolve } from 'node:path';
|
|
11
|
+
import { basename, dirname, join, relative, resolve } from 'node:path';
|
|
12
12
|
import { fileURLToPath } from 'node:url';
|
|
13
13
|
|
|
14
|
+
import { stems } from './stem.js';
|
|
15
|
+
|
|
14
16
|
/**
|
|
15
17
|
* Resolve every `@dzhechkov` base directory that may hold `skills-*` packs, for a given
|
|
16
18
|
* working directory. Ordered by precedence (first wins on pack-name collision):
|
|
@@ -77,6 +79,37 @@ function readSkillDiscoveryConfig(cwd: string): { skillScopes: string[]; skillDi
|
|
|
77
79
|
}
|
|
78
80
|
|
|
79
81
|
/** List `skills-*` pack directories across all base dirs, de-duplicated by pack name (first wins). */
|
|
82
|
+
/**
|
|
83
|
+
* Does this directory actually carry skills? Answers by LOOKING, so a pack is catalogued for what it
|
|
84
|
+
* contains rather than for how it is named. Bounded on purpose: only the three layouts real packs
|
|
85
|
+
* use (`skills/<id>/SKILL.md`, `<pack>/<id>/SKILL.md` for a single-skill pack, and a bare
|
|
86
|
+
* `SKILL.md`), never a full-tree walk — an unbounded scan over `node_modules` would cost more than
|
|
87
|
+
* the catalogue it builds. Templates are excluded: a template is a stamp for making skills, not an
|
|
88
|
+
* installed skill, and counting it would list the same name twice.
|
|
89
|
+
*/
|
|
90
|
+
function packCarriesSkills(dir: string): boolean {
|
|
91
|
+
const hasSkillMd = (d: string): boolean => {
|
|
92
|
+
try {
|
|
93
|
+
for (const entry of readdirSync(d, { withFileTypes: true })) {
|
|
94
|
+
if (entry.name === 'templates' || entry.name === 'node_modules') continue;
|
|
95
|
+
if (entry.isDirectory() && existsSync(join(d, entry.name, 'SKILL.md'))) return true;
|
|
96
|
+
}
|
|
97
|
+
} catch { /* unreadable dir is simply not a skill carrier */ }
|
|
98
|
+
return false;
|
|
99
|
+
};
|
|
100
|
+
try {
|
|
101
|
+
if (existsSync(join(dir, 'SKILL.md'))) return true;
|
|
102
|
+
if (hasSkillMd(join(dir, 'skills'))) return true;
|
|
103
|
+
// A template pack ships the skills it will roll out into the user's project. From the
|
|
104
|
+
// catalogue's point of view those skills EXIST — `trip-planner` and `presentation-storyteller`
|
|
105
|
+
// are installable answers to a task — so hiding them makes the advisor deny a real capability.
|
|
106
|
+
if (hasSkillMd(join(dir, 'templates', '.claude', 'skills'))) return true;
|
|
107
|
+
return hasSkillMd(dir);
|
|
108
|
+
} catch {
|
|
109
|
+
return false;
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
|
|
80
113
|
export function discoverSkillPackDirs(cwd: string): { pack: string; dir: string }[] {
|
|
81
114
|
const seen = new Set<string>();
|
|
82
115
|
const out: { pack: string; dir: string }[] = [];
|
|
@@ -98,6 +131,48 @@ export function discoverSkillPackDirs(cwd: string): { pack: string; dir: string
|
|
|
98
131
|
return out;
|
|
99
132
|
}
|
|
100
133
|
|
|
134
|
+
/**
|
|
135
|
+
* Every directory that CARRIES skills a user can invoke — the catalogue question.
|
|
136
|
+
*
|
|
137
|
+
* A third enumerator on purpose, by the same ADR-001 reasoning that split signature verification
|
|
138
|
+
* from pack discovery. `discoverSkillPackDirs` answers "which SKILL PACKS are here?" and the
|
|
139
|
+
* `skills-` prefix is the right answer to THAT — AM-1 pins it, and widening it would let a plugin
|
|
140
|
+
* be counted as a pack. This function asks something else: "what can the assistant actually offer
|
|
141
|
+
* the user?" — and there the prefix is wrong.
|
|
142
|
+
*
|
|
143
|
+
* MEASURED 2026-09-01: 41 skill names existed on disk and were absent from `dz registry` —
|
|
144
|
+
* every medical skill of `health-advisor` (31 SKILL.md), plus keysarium, p-replicator,
|
|
145
|
+
* design-thinking, trip-planner, evidence-wiki. That gap is not "fewer results": `skill-advisor`
|
|
146
|
+
* must check a name against this catalogue and treat an unlisted one as a fabrication, so an
|
|
147
|
+
* invisible skill turns a hallucination guard into a ban on naming the right answer — asked about
|
|
148
|
+
* blood tests, a live session answered "there is no medical skill in the DZ catalogue" with 31 of
|
|
149
|
+
* them on disk. An authoritative denial of existence is worse than an empty result: it closes the
|
|
150
|
+
* question.
|
|
151
|
+
*
|
|
152
|
+
* The double-count AM-1 guards against is handled where it belongs — `buildRegistry` dedupes by
|
|
153
|
+
* skill id, so a skill reachable through both its canon and a plugin is listed once.
|
|
154
|
+
*/
|
|
155
|
+
export function discoverSkillCarryingDirs(cwd: string): { pack: string; dir: string }[] {
|
|
156
|
+
const seen = new Set<string>();
|
|
157
|
+
const out: { pack: string; dir: string }[] = [];
|
|
158
|
+
for (const { pack, dir } of discoverSkillPackDirs(cwd)) {
|
|
159
|
+
if (!seen.has(pack)) { seen.add(pack); out.push({ pack, dir }); }
|
|
160
|
+
}
|
|
161
|
+
for (const base of skillPackBaseDirs(cwd)) {
|
|
162
|
+
let entries: Dirent[];
|
|
163
|
+
try { entries = readdirSync(base, { withFileTypes: true }); } catch { continue; }
|
|
164
|
+
for (const e of entries) {
|
|
165
|
+
if (seen.has(e.name)) continue;
|
|
166
|
+
const dir = join(base, e.name);
|
|
167
|
+
const isDir = e.isDirectory() || (e.isSymbolicLink() && (() => { try { return statSync(dir).isDirectory(); } catch { return false; } })());
|
|
168
|
+
if (!isDir || !packCarriesSkills(dir)) continue;
|
|
169
|
+
seen.add(e.name);
|
|
170
|
+
out.push({ pack: e.name, dir });
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
return out;
|
|
174
|
+
}
|
|
175
|
+
|
|
101
176
|
/**
|
|
102
177
|
* Every pack whose SIGNATURE should be checked: the skill packs above, PLUS any directory in the same
|
|
103
178
|
* base dirs that carries a `.dz-manifest.json`, whatever it is called.
|
|
@@ -191,6 +266,13 @@ export interface RegistryEntry {
|
|
|
191
266
|
readonly hasEvals: boolean;
|
|
192
267
|
readonly lineCount: number;
|
|
193
268
|
readonly category: string;
|
|
269
|
+
/**
|
|
270
|
+
* Repo-relative path of the skill directory. Stored rather than reconstructed: a skill lives in
|
|
271
|
+
* one of three layouts (pack root, `skills/`, `templates/.claude/skills/`), so `<pack>/<id>` is
|
|
272
|
+
* a guess that silently breaks for two of them — the plugin generator built exactly that guess
|
|
273
|
+
* and produced unresolvable paths the moment the catalogue learned the other layouts.
|
|
274
|
+
*/
|
|
275
|
+
readonly path?: string;
|
|
194
276
|
}
|
|
195
277
|
|
|
196
278
|
/** The full registry. */
|
|
@@ -283,6 +365,19 @@ function categoryFromPack(pack: string): string {
|
|
|
283
365
|
if (pack.includes('book') || pack.includes('12factor')) return 'knowledge';
|
|
284
366
|
// Course/tutorial manufacturing (skills-tutorial-factory: package → Head-First edu-site course).
|
|
285
367
|
if (pack.includes('tutorial')) return 'learning';
|
|
368
|
+
// Packs that carry skills without the `skills-` prefix, catalogued since 2026-09-01. Each needs a
|
|
369
|
+
// category or it lands in `other`, and a test rightly forbids that bucket: an uncategorised skill
|
|
370
|
+
// cannot be filtered for, which is half of being findable.
|
|
371
|
+
if (pack === 'keysarium' || pack.includes('evidence-wiki')) return 'research';
|
|
372
|
+
// feature-adr ships its skills as templates, so they were invisible until the layout fix and the
|
|
373
|
+
// pack never needed a category before. The pipeline that manufactures features is meta-work.
|
|
374
|
+
if (pack === 'p-replicator' || pack.includes('loop-designer') || pack.includes('feature-adr')) return 'meta';
|
|
375
|
+
if (pack === 'design-thinking') return 'design';
|
|
376
|
+
if (pack === 'trip-planner') return 'personal';
|
|
377
|
+
// Packs whose skills live in `templates/` and were therefore never read until the layout fix.
|
|
378
|
+
// Each needs a home or it lands in `other`, which a test rightly forbids.
|
|
379
|
+
if (pack.includes('analyst-manual')) return 'product';
|
|
380
|
+
if (pack.includes('edu-site') || pack.includes('transcript-site')) return 'learning';
|
|
286
381
|
return 'other';
|
|
287
382
|
}
|
|
288
383
|
|
|
@@ -294,24 +389,45 @@ function categoryFromPack(pack: string): string {
|
|
|
294
389
|
*/
|
|
295
390
|
export function buildRegistry(cwd: string): Registry {
|
|
296
391
|
const entries: RegistryEntry[] = [];
|
|
297
|
-
const packs =
|
|
392
|
+
const packs = discoverSkillCarryingDirs(cwd);
|
|
393
|
+
|
|
394
|
+
// A pack keeps its skills in one of three shapes, and reading only the first made 41 real skills
|
|
395
|
+
// invisible (MEASURED 2026-09-01): the pack root (`skills-*` packs), a `skills/` subdirectory
|
|
396
|
+
// (health-advisor and friends), and `templates/.claude/skills/` for packs that roll their skills
|
|
397
|
+
// out into the user's project. The catalogue answers "what can I use", so all three count.
|
|
398
|
+
const SKILL_LAYOUTS = [[], ['skills'], ['templates', '.claude', 'skills']] as const;
|
|
399
|
+
const seenSkillIds = new Set<string>();
|
|
298
400
|
|
|
299
401
|
for (const { pack, dir: packDir } of packs) {
|
|
300
|
-
const
|
|
301
|
-
|
|
402
|
+
const found: { name: string; root: string }[] = [];
|
|
403
|
+
for (const layout of SKILL_LAYOUTS) {
|
|
404
|
+
const root = layout.length === 0 ? packDir : join(packDir, ...layout);
|
|
405
|
+
if (!existsSync(root)) continue;
|
|
406
|
+
try {
|
|
407
|
+
for (const e of readdirSync(root, { withFileTypes: true })) {
|
|
408
|
+
if (e.isDirectory() && existsSync(join(root, e.name, 'SKILL.md'))) found.push({ name: e.name, root });
|
|
409
|
+
}
|
|
410
|
+
} catch { /* an unreadable layout contributes nothing; the others still count */ }
|
|
411
|
+
}
|
|
302
412
|
|
|
303
|
-
for (const skill of
|
|
304
|
-
|
|
413
|
+
for (const skill of found) {
|
|
414
|
+
// One skill id can ship in several packs (frontend-design is bundled by three). The catalogue
|
|
415
|
+
// answers "is this available", not "in how many packs" — so the first sighting wins and the
|
|
416
|
+
// list stays a list of capabilities rather than of copies.
|
|
417
|
+
if (seenSkillIds.has(skill.name)) continue;
|
|
418
|
+
seenSkillIds.add(skill.name);
|
|
419
|
+
const skillMdPath = join(skill.root, skill.name, 'SKILL.md');
|
|
305
420
|
const content = readFileSync(skillMdPath, 'utf-8');
|
|
306
421
|
const { description, trustTier } = extractFrontmatter(content);
|
|
307
422
|
|
|
308
423
|
entries.push({
|
|
309
424
|
id: skill.name,
|
|
310
425
|
pack,
|
|
426
|
+
path: relative(cwd, join(skill.root, skill.name)) || join(skill.root, skill.name),
|
|
311
427
|
description,
|
|
312
428
|
trustTier,
|
|
313
|
-
hasSchema: existsSync(join(
|
|
314
|
-
hasEvals: existsSync(join(
|
|
429
|
+
hasSchema: existsSync(join(skill.root, skill.name, 'schemas', 'output.json')),
|
|
430
|
+
hasEvals: existsSync(join(skill.root, skill.name, 'evals')),
|
|
315
431
|
lineCount: content.split('\n').length,
|
|
316
432
|
category: categoryFromPack(pack),
|
|
317
433
|
});
|
|
@@ -332,9 +448,26 @@ export function buildRegistry(cwd: string): Registry {
|
|
|
332
448
|
/** Search registry by query (matches id and description, case-insensitive). */
|
|
333
449
|
export function searchRegistry(registry: Registry, query: string): readonly RegistryEntry[] {
|
|
334
450
|
const q = query.toLowerCase();
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
451
|
+
const queryStems = stems(query);
|
|
452
|
+
return registry.entries.filter((e) => {
|
|
453
|
+
const indexedText = `${e.id} ${e.description} ${e.category}`;
|
|
454
|
+
const rawMatch = e.id.toLowerCase().includes(q)
|
|
455
|
+
|| e.description.toLowerCase().includes(q)
|
|
456
|
+
|| e.category.toLowerCase().includes(q);
|
|
457
|
+
if (rawMatch) return true;
|
|
458
|
+
|
|
459
|
+
const indexedStems = new Set(stems(indexedText));
|
|
460
|
+
const equalStemMatch = queryStems.length > 0 && queryStems.every((stem) => indexedStems.has(stem));
|
|
461
|
+
if (equalStemMatch) return true;
|
|
462
|
+
|
|
463
|
+
// A singular raw query can already match its root inside a longer token (`анализ`
|
|
464
|
+
// inside `проанализируй`). Its inflected twin must inherit that existing hit or the
|
|
465
|
+
// preserved raw tier makes parity impossible. Limit this compatibility leg to one
|
|
466
|
+
// long stem; short common prefixes stay on exact stem equality.
|
|
467
|
+
const singleStem = queryStems.length === 1 ? queryStems[0] : undefined;
|
|
468
|
+
const foldedIndex = indexedText.normalize('NFC').toLowerCase().replaceAll('ё', 'е');
|
|
469
|
+
return singleStem !== undefined && singleStem.length >= 5 && foldedIndex.includes(singleStem);
|
|
470
|
+
});
|
|
338
471
|
}
|
|
339
472
|
|
|
340
473
|
/** Filter registry by category. */
|
package/src/run-records.ts
CHANGED
|
@@ -94,6 +94,11 @@ function shapeMismatch(kind: RecordKind, payload: Record<string, unknown>): stri
|
|
|
94
94
|
return null;
|
|
95
95
|
}
|
|
96
96
|
|
|
97
|
+
/** A runner id is missing when absent or blank — the same gap rule the date stamp uses. */
|
|
98
|
+
function isRunnerGap(v: unknown): boolean {
|
|
99
|
+
return v === null || v === undefined || (typeof v === 'string' && v.trim() === '');
|
|
100
|
+
}
|
|
101
|
+
|
|
97
102
|
export function decideRecordWrite(input: {
|
|
98
103
|
kind: RecordKind;
|
|
99
104
|
/** The raw `--row` / `--pair` argument, exactly as the caller passed it. */
|
|
@@ -109,6 +114,9 @@ export function decideRecordWrite(input: {
|
|
|
109
114
|
targetHasPair?: boolean;
|
|
110
115
|
/** Stamped INTO the object before serialising — never rewritten in the shell afterwards (FR-7). */
|
|
111
116
|
timestamp?: string | null;
|
|
117
|
+
/** Who ran it. Supplied by the CALLER, which lives outside the workflow sandbox and can see the
|
|
118
|
+
* host; absent stays absent (see the stamping comment below). */
|
|
119
|
+
runnerId?: string | null;
|
|
112
120
|
maxChars?: number;
|
|
113
121
|
}): RecordDecision {
|
|
114
122
|
const { kind, payloadRaw, stage } = input;
|
|
@@ -175,6 +183,21 @@ export function decideRecordWrite(input: {
|
|
|
175
183
|
if (kind === 'training-pair' && isGap(stamped['ts'])) stamped['ts'] = input.timestamp;
|
|
176
184
|
}
|
|
177
185
|
|
|
186
|
+
// WHO ran this. Stamped HERE and nowhere else, for a structural reason: the workflow lives in a
|
|
187
|
+
// sandbox with no host, no process and no clock, so it cannot name its own runner — but this
|
|
188
|
+
// command runs outside that sandbox and can. Same seam that already stamps the date.
|
|
189
|
+
//
|
|
190
|
+
// The field answers a DIFFERENT question from the zombie-preflight predicate (backlog 4a727ac6):
|
|
191
|
+
// that one asks "is this job's PARENT still alive", this one asks "which runner produced this
|
|
192
|
+
// row". Complementary, not duplicate — a future run index joins them, and neither can answer for
|
|
193
|
+
// the other. Absent identity stays ABSENT: an unknown runner is never invented as 'unknown',
|
|
194
|
+
// because a fabricated identity is worse than a missing one for anything that later joins on it.
|
|
195
|
+
// A blank supplied id is a gap too: `' '` sneaking in as a value would join later as a distinct
|
|
196
|
+
// runner made of spaces — the same class of harm as inventing 'unknown'.
|
|
197
|
+
if (kind === 'ledger' && isRunnerGap(stamped['runnerId']) && !isRunnerGap(input.runnerId)) {
|
|
198
|
+
stamped['runnerId'] = (input.runnerId as string).trim();
|
|
199
|
+
}
|
|
200
|
+
|
|
178
201
|
let line: string;
|
|
179
202
|
try {
|
|
180
203
|
line = JSON.stringify(stamped);
|
package/src/score.ts
CHANGED
|
@@ -35,6 +35,14 @@ export interface RunScorecard {
|
|
|
35
35
|
readonly disciplines: readonly DisciplineScore[];
|
|
36
36
|
/** Extracted cross-model grade, when one exists (e.g. "A−", "C"). */
|
|
37
37
|
readonly qeGrade: string | null;
|
|
38
|
+
/**
|
|
39
|
+
* F30а-3, ADDITIVE: present when the run's artifacts carry a structured mutation table. Optional
|
|
40
|
+
* on purpose — every scorecard written before this field existed stays a valid RunScorecard, and
|
|
41
|
+
* `scoreAggregateRowFrom` keeps accepting rows without it. A run with no table reports `absent`
|
|
42
|
+
* rather than omitting the field, so "the gate never ran" and "an older scorer wrote this row"
|
|
43
|
+
* stay distinguishable.
|
|
44
|
+
*/
|
|
45
|
+
readonly mutationEvidence?: MutationEvidence;
|
|
38
46
|
readonly passed: number;
|
|
39
47
|
readonly total: number;
|
|
40
48
|
readonly summary: string;
|
|
@@ -161,6 +169,119 @@ function evidenceLinePositive(text: string, re: RegExp, negationRe: RegExp = NEG
|
|
|
161
169
|
return null;
|
|
162
170
|
}
|
|
163
171
|
|
|
172
|
+
/**
|
|
173
|
+
* F30а-3 — structured mutation evidence.
|
|
174
|
+
*
|
|
175
|
+
* `discrimination` has always scored PROSE: a sentence the author writes about their own work.
|
|
176
|
+
* "None of the mutants survived" scores identically whether three mutations ran or zero did, which
|
|
177
|
+
* makes the strongest discipline in the pipeline rest on the weakest kind of evidence. The mutation
|
|
178
|
+
* gate already emits a five-valued verdict per registry entry, and only `PROVEN` is proof — so a
|
|
179
|
+
* table carrying those verdicts can be COUNTED instead of believed.
|
|
180
|
+
*
|
|
181
|
+
* The design is defined by what it REFUSES. A table that looks like evidence and proves nothing is
|
|
182
|
+
* worse than no table, because it buys the appearance of rigour: the empty one, the header-only one,
|
|
183
|
+
* the one whose every row is INCONCLUSIVE. Each of those returns a non-proving status and is NAMED
|
|
184
|
+
* in the scorecard evidence, never silently treated as corroboration.
|
|
185
|
+
*/
|
|
186
|
+
export type MutationEvidenceStatus =
|
|
187
|
+
| 'proven' // ≥1 row carries the gate's PROVEN verdict — the only status that is evidence
|
|
188
|
+
| 'present-unproven' // a real table, but zero PROVEN rows (includes the header-only table)
|
|
189
|
+
| 'malformed' // a table whose verdict column holds no recognised token — unreadable, not proof
|
|
190
|
+
| 'absent'; // no mutation table in the text at all
|
|
191
|
+
|
|
192
|
+
export interface MutationEvidence {
|
|
193
|
+
readonly status: MutationEvidenceStatus;
|
|
194
|
+
/** Rows whose verdict is exactly PROVEN. */
|
|
195
|
+
readonly proven: number;
|
|
196
|
+
/** Data rows found under the header (0 for the header-only table). */
|
|
197
|
+
readonly rows: number;
|
|
198
|
+
/** The line the verdict rests on — evidence must show its work. */
|
|
199
|
+
readonly evidence: string;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* The gate's own verdict vocabulary. Anything outside it is UNRECOGNISED, which is why a table of
|
|
204
|
+
* free-text opinions ("looked fine", "ok") is malformed rather than quietly scored as unproven:
|
|
205
|
+
* the difference between "the gate ran and found nothing" and "nobody ran the gate" is the whole
|
|
206
|
+
* point of the field.
|
|
207
|
+
*/
|
|
208
|
+
const MUTATION_VERDICTS = new Set([
|
|
209
|
+
'PROVEN',
|
|
210
|
+
'MUTATION_UNPARSEABLE',
|
|
211
|
+
'MUTATION_LOAD_FATAL',
|
|
212
|
+
'OVER_FAILING',
|
|
213
|
+
'INCONCLUSIVE',
|
|
214
|
+
'SURVIVED',
|
|
215
|
+
'NOT_PROVEN',
|
|
216
|
+
]);
|
|
217
|
+
|
|
218
|
+
const splitRow = (line: string): string[] =>
|
|
219
|
+
line
|
|
220
|
+
.trim()
|
|
221
|
+
.replace(/^\|/, '')
|
|
222
|
+
.replace(/\|$/, '')
|
|
223
|
+
.split('|')
|
|
224
|
+
.map((cell) => cell.trim());
|
|
225
|
+
|
|
226
|
+
const isSeparatorRow = (line: string): boolean => /^\|[\s:|-]+\|?$/.test(line.trim());
|
|
227
|
+
|
|
228
|
+
export function readMutationEvidence(text: string): MutationEvidence {
|
|
229
|
+
const lines = String(text).split('\n');
|
|
230
|
+
for (let i = 0; i < lines.length; i++) {
|
|
231
|
+
const line = lines[i];
|
|
232
|
+
if (line === undefined || !line.trim().startsWith('|')) continue;
|
|
233
|
+
const header = splitRow(line);
|
|
234
|
+
// A mutation table is identified by its COLUMNS, not by a heading above it: headings drift
|
|
235
|
+
// across reports and translations, column names are what the parser actually reads.
|
|
236
|
+
const verdictCol = header.findIndex((c) => /verdict/i.test(c));
|
|
237
|
+
if (verdictCol < 0 || !header.some((c) => /mutation|mutant/i.test(c))) continue;
|
|
238
|
+
|
|
239
|
+
let rows = 0;
|
|
240
|
+
let proven = 0;
|
|
241
|
+
let recognised = 0;
|
|
242
|
+
for (let j = i + 1; j < lines.length; j++) {
|
|
243
|
+
const row = lines[j];
|
|
244
|
+
if (row === undefined || !row.trim().startsWith('|')) break;
|
|
245
|
+
if (isSeparatorRow(row)) continue;
|
|
246
|
+
const cells = splitRow(row);
|
|
247
|
+
rows++;
|
|
248
|
+
const cell = (cells[verdictCol] ?? '').toUpperCase();
|
|
249
|
+
// Word-bounded: NOT_PROVEN contains PROVEN as a substring and must never count as one.
|
|
250
|
+
const token = (cell.match(/\b[A-Z_]{4,}\b/) ?? [])[0] ?? '';
|
|
251
|
+
if (!MUTATION_VERDICTS.has(token)) continue;
|
|
252
|
+
recognised++;
|
|
253
|
+
if (token === 'PROVEN') proven++;
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
if (proven > 0) {
|
|
257
|
+
return {
|
|
258
|
+
status: 'proven',
|
|
259
|
+
proven,
|
|
260
|
+
rows,
|
|
261
|
+
evidence: `mutation table: ${proven} of ${rows} row(s) PROVEN`,
|
|
262
|
+
};
|
|
263
|
+
}
|
|
264
|
+
if (rows > 0 && recognised === 0) {
|
|
265
|
+
return {
|
|
266
|
+
status: 'malformed',
|
|
267
|
+
proven: 0,
|
|
268
|
+
rows,
|
|
269
|
+
evidence: `mutation table: ${rows} row(s), no recognised gate verdict in the Verdict column — unreadable, not proof`,
|
|
270
|
+
};
|
|
271
|
+
}
|
|
272
|
+
return {
|
|
273
|
+
status: 'present-unproven',
|
|
274
|
+
proven: 0,
|
|
275
|
+
rows,
|
|
276
|
+
evidence:
|
|
277
|
+
rows === 0
|
|
278
|
+
? 'mutation table present with no data rows — proved nothing'
|
|
279
|
+
: `mutation table: 0 of ${rows} row(s) PROVEN — proved nothing`,
|
|
280
|
+
};
|
|
281
|
+
}
|
|
282
|
+
return { status: 'absent', proven: 0, rows: 0, evidence: 'no mutation table' };
|
|
283
|
+
}
|
|
284
|
+
|
|
164
285
|
// Word-bounded on BOTH sides: "upgrade B-tree" fabricated a B- (Codex QE #1). The lookahead also
|
|
165
286
|
// rejects "Grade B-tree" (letter after the dash) while keeping the real "Grade: A−" formats.
|
|
166
287
|
//
|
|
@@ -305,14 +426,26 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
|
|
|
305
426
|
// MUTANTS — "None of the mutants survived", "neither mutant escaped" — which is the proof, not
|
|
306
427
|
// its denial. The quantifiers would discard exactly the strongest lines this discipline exists
|
|
307
428
|
// to find. "No discrimination proof was performed" is still caught by the narrow list.
|
|
429
|
+
// F30а-3 makes this ADDITIVE: a structured mutation table is counted FIRST (it can be
|
|
430
|
+
// verified, prose can only be believed), and the prose path below is left exactly as it was —
|
|
431
|
+
// no run that scored `pass` yesterday loses it today. What the table changes is the OTHER
|
|
432
|
+
// direction: a table that proves nothing is named in the evidence instead of sitting silently
|
|
433
|
+
// beside a winning sentence and reading as corroboration.
|
|
434
|
+
const mutationEvidence = readMutationEvidence(allText);
|
|
308
435
|
const discr =
|
|
309
436
|
evidenceLinePositive(allText, /discrimination|§42/i) ??
|
|
310
437
|
evidenceLinePositive(allText, /mutation[s]?\s.*(prov|kill)|mutant[s]?\s.*(kill|red)|RED on the old|goes? RED|failed as expected/i);
|
|
438
|
+
const discrPasses = mutationEvidence.status === 'proven' || discr !== null;
|
|
311
439
|
add(
|
|
312
440
|
'discrimination',
|
|
313
441
|
'the property test is proven able to fail',
|
|
314
|
-
|
|
315
|
-
|
|
442
|
+
discrPasses ? 'pass' : 'absent',
|
|
443
|
+
mutationEvidence.status === 'proven'
|
|
444
|
+
? mutationEvidence.evidence
|
|
445
|
+
: (discr ??
|
|
446
|
+
(mutationEvidence.status === 'absent'
|
|
447
|
+
? 'no discrimination/§42/mutation-proof evidence in any artifact'
|
|
448
|
+
: mutationEvidence.evidence)),
|
|
316
449
|
);
|
|
317
450
|
|
|
318
451
|
// 3. Cross-model QE — an independent family reviewed it, and a grade exists.
|
|
@@ -414,7 +547,7 @@ export function scoreRun(slug: string, artifacts: RunArtifacts): RunScorecard {
|
|
|
414
547
|
(grade !== null ? ` · QE grade ${grade}` : ' · no QE grade') +
|
|
415
548
|
(worst.length > 0 ? ` · absent: ${worst.join(', ')}` : '');
|
|
416
549
|
|
|
417
|
-
return { slug, disciplines, qeGrade: grade, passed, total, summary };
|
|
550
|
+
return { slug, disciplines, qeGrade: grade, mutationEvidence, passed, total, summary };
|
|
418
551
|
}
|
|
419
552
|
|
|
420
553
|
const MARK: Record<DisciplineVerdict, string> = { pass: '✓', partial: '◐', absent: '✗' };
|
|
@@ -427,7 +560,232 @@ export function renderScorecard(card: RunScorecard): string {
|
|
|
427
560
|
out.push(` ${MARK[d.verdict]} ${d.title}`);
|
|
428
561
|
out.push(` ${d.evidence}`);
|
|
429
562
|
}
|
|
563
|
+
// F30а-3: the table's own verdict is reported INDEPENDENTLY of which path scored `discrimination`.
|
|
564
|
+
// When prose carries the pass, the discipline's evidence line shows the prose — and a table that
|
|
565
|
+
// proved nothing would sit in the report unmentioned, beside a sentence claiming the mutants died.
|
|
566
|
+
// Silence about a hollow table is the failure this feature exists to prevent, so it is stated here
|
|
567
|
+
// even when it changes no verdict. No table at all stays silent: the common case earns no noise.
|
|
568
|
+
if (card.mutationEvidence !== undefined && card.mutationEvidence.status !== 'absent') {
|
|
569
|
+
out.push('');
|
|
570
|
+
out.push(` ${card.mutationEvidence.evidence}`);
|
|
571
|
+
}
|
|
430
572
|
out.push('');
|
|
431
573
|
out.push(` ${card.summary}`);
|
|
432
574
|
return out.join('\n');
|
|
433
575
|
}
|
|
576
|
+
|
|
577
|
+
/** The append-only projection of one immutable `score-<qeHash>.json` receipt. */
|
|
578
|
+
export interface ScoreAggregateRow {
|
|
579
|
+
readonly ts: string;
|
|
580
|
+
readonly slug: string;
|
|
581
|
+
readonly qeHash: string;
|
|
582
|
+
readonly passed: number;
|
|
583
|
+
readonly total: number;
|
|
584
|
+
readonly qeGrade: string | null;
|
|
585
|
+
readonly disciplines: readonly {
|
|
586
|
+
readonly id: string;
|
|
587
|
+
readonly verdict: DisciplineVerdict;
|
|
588
|
+
}[];
|
|
589
|
+
/**
|
|
590
|
+
* F30а-3, ADDITIVE: rows PROVEN by the mutation gate, when the scorecard carried a table.
|
|
591
|
+
* `undefined` means the scorecard had no mutation field at all — which is NOT the same as `0`
|
|
592
|
+
* (a table that ran and proved nothing). Every row written before this field existed keeps
|
|
593
|
+
* parsing: the aggregate must never lose its history to a schema change.
|
|
594
|
+
*/
|
|
595
|
+
readonly mutationProven?: number;
|
|
596
|
+
}
|
|
597
|
+
|
|
598
|
+
export interface ScoreReceiptInput {
|
|
599
|
+
readonly content: string;
|
|
600
|
+
readonly qeHash: string;
|
|
601
|
+
/** Supplied by the impure caller. Receipt projection never reads a clock. */
|
|
602
|
+
readonly ts: string;
|
|
603
|
+
}
|
|
604
|
+
|
|
605
|
+
function isDisciplineVerdict(value: unknown): value is DisciplineVerdict {
|
|
606
|
+
return value === 'pass' || value === 'partial' || value === 'absent';
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
function scoreAggregateRowFrom(value: unknown): ScoreAggregateRow | null {
|
|
610
|
+
if (typeof value !== 'object' || value === null || Array.isArray(value)) return null;
|
|
611
|
+
const record = value as Record<string, unknown>;
|
|
612
|
+
const disciplinesValue = record['disciplines'];
|
|
613
|
+
if (!Array.isArray(disciplinesValue)) return null;
|
|
614
|
+
const disciplines: { id: string; verdict: DisciplineVerdict }[] = [];
|
|
615
|
+
const disciplineIds = new Set<string>();
|
|
616
|
+
for (const value of disciplinesValue) {
|
|
617
|
+
if (typeof value !== 'object' || value === null || Array.isArray(value)) return null;
|
|
618
|
+
const discipline = value as Record<string, unknown>;
|
|
619
|
+
if (typeof discipline['id'] !== 'string' || discipline['id'].trim() === '') return null;
|
|
620
|
+
if (!isDisciplineVerdict(discipline['verdict'])) return null;
|
|
621
|
+
if (disciplineIds.has(discipline['id'])) return null;
|
|
622
|
+
disciplineIds.add(discipline['id']);
|
|
623
|
+
disciplines.push({ id: discipline['id'], verdict: discipline['verdict'] });
|
|
624
|
+
}
|
|
625
|
+
const passed = record['passed'];
|
|
626
|
+
const total = record['total'];
|
|
627
|
+
const qeGrade = record['qeGrade'];
|
|
628
|
+
if (typeof record['ts'] !== 'string' || record['ts'].trim() === '') return null;
|
|
629
|
+
if (typeof record['slug'] !== 'string' || record['slug'].trim() === '') return null;
|
|
630
|
+
if (typeof record['qeHash'] !== 'string' || record['qeHash'].trim() === '') return null;
|
|
631
|
+
if (typeof passed !== 'number' || !Number.isSafeInteger(passed) || passed < 0) return null;
|
|
632
|
+
if (typeof total !== 'number' || !Number.isSafeInteger(total) || total < 0) return null;
|
|
633
|
+
if (qeGrade !== null && (typeof qeGrade !== 'string' || qeGrade.trim() === '')) return null;
|
|
634
|
+
if (total !== disciplines.length) return null;
|
|
635
|
+
if (passed !== disciplines.filter((discipline) => discipline.verdict === 'pass').length) return null;
|
|
636
|
+
return {
|
|
637
|
+
ts: record['ts'],
|
|
638
|
+
slug: record['slug'],
|
|
639
|
+
qeHash: record['qeHash'],
|
|
640
|
+
passed,
|
|
641
|
+
total,
|
|
642
|
+
qeGrade,
|
|
643
|
+
disciplines,
|
|
644
|
+
};
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
/**
|
|
648
|
+
* Parse one score receipt into the deliberately small aggregate schema.
|
|
649
|
+
*
|
|
650
|
+
* The caller owns I/O and supplies `ts`; this function is deterministic for the same input. A
|
|
651
|
+
* syntactically valid but internally inconsistent scorecard is unreadable evidence, not a row the
|
|
652
|
+
* aggregate should silently bless.
|
|
653
|
+
*/
|
|
654
|
+
export function scoreReceiptToAggregateRow(input: ScoreReceiptInput): ScoreAggregateRow {
|
|
655
|
+
let parsed: unknown;
|
|
656
|
+
try {
|
|
657
|
+
parsed = JSON.parse(input.content) as unknown;
|
|
658
|
+
} catch {
|
|
659
|
+
throw new Error('score receipt is not valid JSON');
|
|
660
|
+
}
|
|
661
|
+
if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) {
|
|
662
|
+
throw new Error('score receipt is not a RunScorecard object');
|
|
663
|
+
}
|
|
664
|
+
const receipt = parsed as Record<string, unknown>;
|
|
665
|
+
const row = scoreAggregateRowFrom({ ...receipt, ts: input.ts, qeHash: input.qeHash });
|
|
666
|
+
if (row === null) throw new Error('score receipt has an invalid RunScorecard shape');
|
|
667
|
+
// Projected here rather than inside scoreAggregateRowFrom so that a MALFORMED mutation field can
|
|
668
|
+
// never invalidate an otherwise-good row: the field is additive, so its absence — or its
|
|
669
|
+
// unreadability — costs the row nothing but the field itself.
|
|
670
|
+
const ev = receipt['mutationEvidence'];
|
|
671
|
+
if (typeof ev === 'object' && ev !== null && !Array.isArray(ev)) {
|
|
672
|
+
const proven = (ev as Record<string, unknown>)['proven'];
|
|
673
|
+
if (typeof proven === 'number' && Number.isFinite(proven) && proven >= 0) {
|
|
674
|
+
return { ...row, mutationProven: proven };
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
return row;
|
|
678
|
+
}
|
|
679
|
+
|
|
680
|
+
/** Read only valid aggregate rows; event-chain verification separately names malformed lines. */
|
|
681
|
+
export function readScoreAggregateRows(text: string): ScoreAggregateRow[] {
|
|
682
|
+
const rows: ScoreAggregateRow[] = [];
|
|
683
|
+
for (const line of String(text).split('\n')) {
|
|
684
|
+
if (line.trim() === '') continue;
|
|
685
|
+
try {
|
|
686
|
+
const row = scoreAggregateRowFrom(JSON.parse(line) as unknown);
|
|
687
|
+
if (row !== null) rows.push(row);
|
|
688
|
+
} catch {
|
|
689
|
+
/* A torn line is chain evidence, not a row. verifyEventChainText names it. */
|
|
690
|
+
}
|
|
691
|
+
}
|
|
692
|
+
return rows;
|
|
693
|
+
}
|
|
694
|
+
|
|
695
|
+
/** Keep the first occurrence of each `(slug, qeHash)` pair not already present in the aggregate. */
|
|
696
|
+
export function dedupeScoreAggregateRows(
|
|
697
|
+
candidates: readonly ScoreAggregateRow[],
|
|
698
|
+
existing: readonly ScoreAggregateRow[],
|
|
699
|
+
): ScoreAggregateRow[] {
|
|
700
|
+
const seen = new Set(existing.map((row) => `${row.slug}\u0000${row.qeHash}`));
|
|
701
|
+
const fresh: ScoreAggregateRow[] = [];
|
|
702
|
+
for (const row of candidates) {
|
|
703
|
+
const key = `${row.slug}\u0000${row.qeHash}`;
|
|
704
|
+
if (seen.has(key)) continue;
|
|
705
|
+
seen.add(key);
|
|
706
|
+
fresh.push(row);
|
|
707
|
+
}
|
|
708
|
+
return fresh;
|
|
709
|
+
}
|
|
710
|
+
|
|
711
|
+
export type ScoreAggregateVerdict = 'REPORTED' | 'INSUFFICIENT_DATA';
|
|
712
|
+
|
|
713
|
+
export interface ScoreDisciplineAggregate {
|
|
714
|
+
readonly id: string;
|
|
715
|
+
readonly pass: number;
|
|
716
|
+
readonly partial: number;
|
|
717
|
+
readonly absent: number;
|
|
718
|
+
}
|
|
719
|
+
|
|
720
|
+
export interface ScoreGradeAggregate {
|
|
721
|
+
readonly grade: string | null;
|
|
722
|
+
readonly count: number;
|
|
723
|
+
}
|
|
724
|
+
|
|
725
|
+
export interface ScoreAggregateReport {
|
|
726
|
+
readonly verdict: ScoreAggregateVerdict;
|
|
727
|
+
readonly receipts: number;
|
|
728
|
+
readonly appended: number;
|
|
729
|
+
readonly unreadable: number;
|
|
730
|
+
/** Repo-relative receipt paths: unreadable evidence is always named, never only counted. */
|
|
731
|
+
readonly unreadableReceipts: readonly string[];
|
|
732
|
+
readonly disciplines: readonly ScoreDisciplineAggregate[];
|
|
733
|
+
readonly grades: readonly ScoreGradeAggregate[];
|
|
734
|
+
}
|
|
735
|
+
|
|
736
|
+
/** Fold the aggregate into an advisory report. No readable rows is a third state, never success. */
|
|
737
|
+
export function buildScoreAggregateReport(
|
|
738
|
+
rows: readonly ScoreAggregateRow[],
|
|
739
|
+
unreadableReceipts: readonly string[],
|
|
740
|
+
appended: number,
|
|
741
|
+
): ScoreAggregateReport {
|
|
742
|
+
const disciplineCounts = new Map<string, { pass: number; partial: number; absent: number }>();
|
|
743
|
+
const gradeCounts = new Map<string | null, number>();
|
|
744
|
+
for (const row of rows) {
|
|
745
|
+
gradeCounts.set(row.qeGrade, (gradeCounts.get(row.qeGrade) ?? 0) + 1);
|
|
746
|
+
for (const discipline of row.disciplines) {
|
|
747
|
+
const counts = disciplineCounts.get(discipline.id) ?? { pass: 0, partial: 0, absent: 0 };
|
|
748
|
+
counts[discipline.verdict] += 1;
|
|
749
|
+
disciplineCounts.set(discipline.id, counts);
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
const disciplines = [...disciplineCounts.entries()]
|
|
753
|
+
.sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0)
|
|
754
|
+
.map(([id, counts]) => ({ id, ...counts }));
|
|
755
|
+
const grades = [...gradeCounts.entries()]
|
|
756
|
+
.sort(([a], [b]) => (a === null ? 1 : b === null ? -1 : a < b ? -1 : a > b ? 1 : 0))
|
|
757
|
+
.map(([grade, count]) => ({ grade, count }));
|
|
758
|
+
return {
|
|
759
|
+
verdict: rows.length === 0 ? 'INSUFFICIENT_DATA' : 'REPORTED',
|
|
760
|
+
receipts: rows.length + unreadableReceipts.length,
|
|
761
|
+
appended,
|
|
762
|
+
unreadable: unreadableReceipts.length,
|
|
763
|
+
unreadableReceipts: [...unreadableReceipts],
|
|
764
|
+
disciplines,
|
|
765
|
+
grades,
|
|
766
|
+
};
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
export function renderScoreAggregateReport(report: ScoreAggregateReport): string {
|
|
770
|
+
const out = [
|
|
771
|
+
'dz score --all — chained score-receipt aggregate (descriptive-only, never a gate)',
|
|
772
|
+
`VERDICT: ${report.verdict}`,
|
|
773
|
+
`receipts: ${report.receipts}`,
|
|
774
|
+
`appended: ${report.appended}`,
|
|
775
|
+
`unreadable: ${report.unreadable}`,
|
|
776
|
+
];
|
|
777
|
+
if (report.unreadableReceipts.length > 0) {
|
|
778
|
+
out.push(`unreadable receipts: ${report.unreadableReceipts.join(', ')}`);
|
|
779
|
+
}
|
|
780
|
+
if (report.verdict === 'INSUFFICIENT_DATA') {
|
|
781
|
+
out.push('No readable score receipts; no discipline ratio or percentage is reported.');
|
|
782
|
+
return out.join('\n');
|
|
783
|
+
}
|
|
784
|
+
out.push('disciplines:');
|
|
785
|
+
for (const discipline of report.disciplines) {
|
|
786
|
+
out.push(` ${discipline.id}: pass ${discipline.pass} · partial ${discipline.partial} · absent ${discipline.absent}`);
|
|
787
|
+
}
|
|
788
|
+
out.push('QE grades:');
|
|
789
|
+
for (const grade of report.grades) out.push(` ${grade.grade ?? '(no grade)'}: ${grade.count}`);
|
|
790
|
+
return out.join('\n');
|
|
791
|
+
}
|