webrecipe 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +253 -0
- package/dist/benchmark/amortization.js +254 -0
- package/dist/benchmark/fixtures.js +26 -0
- package/dist/benchmark/oracles.js +129 -0
- package/dist/benchmark/plans.js +436 -0
- package/dist/fixtures/cloaking.js +37 -0
- package/dist/fixtures/coalesce.js +52 -0
- package/dist/fixtures/data.js +23 -0
- package/dist/fixtures/harness.js +34 -0
- package/dist/fixtures/ignoring.js +27 -0
- package/dist/fixtures/limiting.js +38 -0
- package/dist/fixtures/paging.js +72 -0
- package/dist/fixtures/refusing.js +57 -0
- package/dist/fixtures/shifted.js +32 -0
- package/dist/fixtures/spa.js +71 -0
- package/dist/fixtures/ssr.js +46 -0
- package/dist/fixtures/volatile.js +40 -0
- package/dist/fixtures/xhr.js +120 -0
- package/dist/src/analyzer/classify.js +16 -0
- package/dist/src/analyzer/score.js +52 -0
- package/dist/src/authoring/candidates.js +168 -0
- package/dist/src/authoring/contract.js +31 -0
- package/dist/src/authoring/fields.js +86 -0
- package/dist/src/authoring/learn.js +51 -0
- package/dist/src/authoring/plans.js +93 -0
- package/dist/src/authoring/snapshot.js +22 -0
- package/dist/src/authoring/teach.js +136 -0
- package/dist/src/benchmark/discovery.js +355 -0
- package/dist/src/benchmark/golden.js +95 -0
- package/dist/src/benchmark/grade.js +146 -0
- package/dist/src/benchmark/ground-truth.js +35 -0
- package/dist/src/benchmark/health.js +96 -0
- package/dist/src/benchmark/labels.js +49 -0
- package/dist/src/benchmark/oracle.js +55 -0
- package/dist/src/benchmark/report.js +191 -0
- package/dist/src/benchmark/runner.js +201 -0
- package/dist/src/benchmark/screen.js +144 -0
- package/dist/src/benchmark/selector-score.js +86 -0
- package/dist/src/benchmark/verification-cases.js +138 -0
- package/dist/src/benchmark/verification-matrix.js +97 -0
- package/dist/src/browser/navigate.js +22 -0
- package/dist/src/browser/pool.js +31 -0
- package/dist/src/browser/session.js +44 -0
- package/dist/src/cli.js +559 -0
- package/dist/src/compiler/derive.js +144 -0
- package/dist/src/compiler/heuristic.js +398 -0
- package/dist/src/compiler/html.js +117 -0
- package/dist/src/compiler/types.js +12 -0
- package/dist/src/compiler/verify.js +29 -0
- package/dist/src/executor/extract.js +179 -0
- package/dist/src/executor/format.js +55 -0
- package/dist/src/executor/index.js +147 -0
- package/dist/src/executor/strategies/browser.js +60 -0
- package/dist/src/executor/strategies/http-html.js +42 -0
- package/dist/src/executor/strategies/http-json.js +71 -0
- package/dist/src/executor/strategies/warm-browser.js +57 -0
- package/dist/src/executor/tokens.js +11 -0
- package/dist/src/healing/index.js +111 -0
- package/dist/src/local.js +157 -0
- package/dist/src/mcp.js +130 -0
- package/dist/src/measurement.js +44 -0
- package/dist/src/net/politeness.js +141 -0
- package/dist/src/net/robots.js +56 -0
- package/dist/src/read.js +83 -0
- package/dist/src/recipes/fingerprint.js +41 -0
- package/dist/src/recipes/paths.js +14 -0
- package/dist/src/recipes/registry.js +81 -0
- package/dist/src/recipes/schema.js +38 -0
- package/dist/src/recipes/template.js +33 -0
- package/dist/src/recorder/body.js +59 -0
- package/dist/src/recorder/index.js +151 -0
- package/dist/src/recorder/types.js +1 -0
- package/dist/src/sites.js +45 -0
- package/dist/src/tasks.js +37 -0
- package/dist/src/types.js +32 -0
- package/dist/src/usage.js +69 -0
- package/dist/src/validator/index.js +28 -0
- package/dist/src/verification/lexical-consistency.js +88 -0
- package/dist/src/verification/pagination-honored.js +110 -0
- package/dist/src/verification/probes.js +98 -0
- package/dist/src/verification/query-honored.js +134 -0
- package/dist/src/wiring.js +33 -0
- package/package.json +56 -0
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
import { stat } from 'node:fs/promises';
|
|
2
|
+
import { join } from 'node:path';
|
|
3
|
+
import { RecipeRegistry } from '../recipes/registry.js';
|
|
4
|
+
import { HttpHtmlStrategy } from '../executor/strategies/http-html.js';
|
|
5
|
+
import { HttpJsonStrategy } from '../executor/strategies/http-json.js';
|
|
6
|
+
import { validate } from '../validator/index.js';
|
|
7
|
+
/**
|
|
8
|
+
* Whether the recipes already on disk still work.
|
|
9
|
+
*
|
|
10
|
+
* `bench run` cannot answer this. It pairs every task with a fresh browser
|
|
11
|
+
* baseline, which doubles what the site is asked for, and it grades against
|
|
12
|
+
* goldens captured days earlier, which reads a site editing its content as a
|
|
13
|
+
* recipe that broke. Durability is a narrower question — does the stored
|
|
14
|
+
* recipe still fetch and validate — and it is asked by replaying the recipe
|
|
15
|
+
* and nothing else.
|
|
16
|
+
*/
|
|
17
|
+
/** The statuses `src/executor/index.ts` treats as a refusal, for the same reason. */
|
|
18
|
+
const BLOCK_STATUSES = new Set([401, 403, 429, 503]);
|
|
19
|
+
export async function checkRecipes(opts) {
|
|
20
|
+
const registry = new RecipeRegistry(opts.recipeDir);
|
|
21
|
+
const html = new HttpHtmlStrategy(opts.net, opts.sites);
|
|
22
|
+
const json = new HttpJsonStrategy(opts.net, opts.sites);
|
|
23
|
+
const wanted = new Map();
|
|
24
|
+
for (const task of opts.tasks) {
|
|
25
|
+
const key = `${task.site}\u0000${task.intent}`;
|
|
26
|
+
const run = wanted.get(key);
|
|
27
|
+
if (run)
|
|
28
|
+
run.push(task);
|
|
29
|
+
else
|
|
30
|
+
wanted.set(key, [task]);
|
|
31
|
+
}
|
|
32
|
+
const health = [];
|
|
33
|
+
for (const [key, all] of wanted) {
|
|
34
|
+
const [site, intent] = key.split('\u0000');
|
|
35
|
+
const recipe = await registry.load(site, intent);
|
|
36
|
+
// A browser recipe has no HTTP path to outlive, so there is nothing here
|
|
37
|
+
// to measure.
|
|
38
|
+
if (recipe === null || recipe.strategy.type === 'browser')
|
|
39
|
+
continue;
|
|
40
|
+
const strategy = recipe.strategy.type === 'http-json' ? json : html;
|
|
41
|
+
const statuses = [];
|
|
42
|
+
const reasons = [];
|
|
43
|
+
let valid = 0;
|
|
44
|
+
const samples = all.slice(0, opts.samples);
|
|
45
|
+
for (const task of samples) {
|
|
46
|
+
try {
|
|
47
|
+
const result = await strategy.execute(recipe, task);
|
|
48
|
+
if (result.status !== undefined)
|
|
49
|
+
statuses.push(result.status);
|
|
50
|
+
const outcome = validate(recipe, {
|
|
51
|
+
status: result.status ?? recipe.validation.status,
|
|
52
|
+
payload: result.payload,
|
|
53
|
+
items: result.items,
|
|
54
|
+
});
|
|
55
|
+
if (outcome.valid)
|
|
56
|
+
valid += 1;
|
|
57
|
+
else
|
|
58
|
+
reasons.push(`${task.id}: ${outcome.reasons.join('; ')}`);
|
|
59
|
+
}
|
|
60
|
+
catch (err) {
|
|
61
|
+
reasons.push(`${task.id}: ${err instanceof Error ? err.message : String(err)}`);
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
// A refusal outranks a mismatch: a site that answered 403 has stopped
|
|
65
|
+
// serving this client, and whatever the recipe would have parsed is moot.
|
|
66
|
+
const verdict = statuses.some((s) => BLOCK_STATUSES.has(s))
|
|
67
|
+
? 'blocked'
|
|
68
|
+
: valid === samples.length ? 'alive' : 'broken';
|
|
69
|
+
health.push({
|
|
70
|
+
site,
|
|
71
|
+
intent,
|
|
72
|
+
compiledAt: (await stat(join(opts.recipeDir, site, `${intent}.yaml`))).mtime,
|
|
73
|
+
samples: samples.length,
|
|
74
|
+
valid,
|
|
75
|
+
statuses: [...new Set(statuses)],
|
|
76
|
+
verdict,
|
|
77
|
+
reasons,
|
|
78
|
+
});
|
|
79
|
+
}
|
|
80
|
+
return health;
|
|
81
|
+
}
|
|
82
|
+
const age = (from, now) => {
|
|
83
|
+
const hours = Math.round((now.getTime() - from.getTime()) / 3_600_000);
|
|
84
|
+
return hours < 48 ? `${hours}h` : `${Math.round(hours / 24)}d`;
|
|
85
|
+
};
|
|
86
|
+
export function formatHealth(health, now = new Date()) {
|
|
87
|
+
const lines = health.map((h) => ` ${h.verdict.padEnd(8)} ${`${h.site} ${h.intent}`.padEnd(34)} ` +
|
|
88
|
+
`${h.valid}/${h.samples} valid age ${age(h.compiledAt, now).padStart(4)}` +
|
|
89
|
+
`${h.statuses.length > 0 ? ` status ${h.statuses.join(',')}` : ''}`);
|
|
90
|
+
const count = (v) => health.filter((h) => h.verdict === v).length;
|
|
91
|
+
return [
|
|
92
|
+
...lines,
|
|
93
|
+
'',
|
|
94
|
+
` alive ${count('alive')} blocked ${count('blocked')} broken ${count('broken')} of ${health.length}`,
|
|
95
|
+
].join('\n');
|
|
96
|
+
}
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { elementsOf, isEl, loadPage, normalize } from '../authoring/candidates.js';
|
|
3
|
+
/**
|
|
4
|
+
* What a human read off the rendered page, and nothing else.
|
|
5
|
+
*
|
|
6
|
+
* There is deliberately nowhere in this schema to put a selector. The whole
|
|
7
|
+
* harness rests on the label being content while the candidate is structure; a
|
|
8
|
+
* label derived from a selector would score that selector against itself, which
|
|
9
|
+
* is the circle the compile-time equivalence check is already stuck in.
|
|
10
|
+
*/
|
|
11
|
+
export const LabelSchema = z.object({
|
|
12
|
+
snapshot: z.string().min(1),
|
|
13
|
+
url: z.string().url(),
|
|
14
|
+
capturedAt: z.string().min(1),
|
|
15
|
+
note: z.string().default(''),
|
|
16
|
+
/** Which of the two a human used to write the items down. */
|
|
17
|
+
identifier: z.enum(['text', 'href']),
|
|
18
|
+
items: z.array(z.string().min(1)).min(1),
|
|
19
|
+
});
|
|
20
|
+
/**
|
|
21
|
+
* The page's text in document order, so a human can mark the item boundaries
|
|
22
|
+
* without a selector having proposed them first.
|
|
23
|
+
*/
|
|
24
|
+
export function visibleRuns(html) {
|
|
25
|
+
const $ = loadPage(html);
|
|
26
|
+
const runs = [];
|
|
27
|
+
const walk = (node) => {
|
|
28
|
+
for (const child of $(node).contents().toArray()) {
|
|
29
|
+
if (child.type === 'text') {
|
|
30
|
+
const text = normalize($(child).text());
|
|
31
|
+
if (text !== '')
|
|
32
|
+
runs.push(text);
|
|
33
|
+
}
|
|
34
|
+
else if (isEl(child)) {
|
|
35
|
+
// An image speaks through its alt, and on a page of covers that is the
|
|
36
|
+
// only thing the item says.
|
|
37
|
+
const alt = normalize(child.tagName.toLowerCase() === 'img' ? $(child).attr('alt') ?? '' : '');
|
|
38
|
+
if (alt !== '')
|
|
39
|
+
runs.push(alt);
|
|
40
|
+
walk(child);
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
};
|
|
44
|
+
const root = elementsOf($, 'body')[0] ?? elementsOf($, 'html')[0];
|
|
45
|
+
if (root === undefined)
|
|
46
|
+
return [];
|
|
47
|
+
walk(root);
|
|
48
|
+
return runs;
|
|
49
|
+
}
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
/**
|
|
3
|
+
* `property-based` is reserved in the spec but deliberately unimplemented, so
|
|
4
|
+
* it is absent here: a task naming it fails at load rather than being quietly
|
|
5
|
+
* graded as something else.
|
|
6
|
+
*/
|
|
7
|
+
export const OracleModeSchema = z.enum(['golden', 'paired-live']);
|
|
8
|
+
export const VolatilitySchema = z.enum(['low', 'medium', 'high']);
|
|
9
|
+
// .strict() throughout: a misspelled key (e.g. `field` for `fields`) must fail
|
|
10
|
+
// at load, the same reasoning that kept `property-based` out of the mode enum
|
|
11
|
+
// rather than letting it fall through to a default silently.
|
|
12
|
+
export const OracleSpecSchema = z.object({
|
|
13
|
+
mode: OracleModeSchema.optional(),
|
|
14
|
+
volatility: VolatilitySchema.optional(),
|
|
15
|
+
compare: z.object({
|
|
16
|
+
entities: z.literal('set').optional(),
|
|
17
|
+
ordering: z.enum(['ignore', 'strict']).optional(),
|
|
18
|
+
/** Restricts comparison to the fields that carry meaning. */
|
|
19
|
+
fields: z.array(z.string()).optional(),
|
|
20
|
+
}).strict().optional(),
|
|
21
|
+
/**
|
|
22
|
+
* Opt-in. Declared only where a site is known to be able to ignore its own
|
|
23
|
+
* search input, as arbeitnow.com does. Absent, query semantics is reported
|
|
24
|
+
* as unjudged rather than as a pass.
|
|
25
|
+
*/
|
|
26
|
+
semantics: z.object({
|
|
27
|
+
input: z.string(),
|
|
28
|
+
fields: z.array(z.string()),
|
|
29
|
+
match: z.literal('contains-token'),
|
|
30
|
+
minShare: z.number().min(0).max(1),
|
|
31
|
+
}).strict().optional(),
|
|
32
|
+
}).strict();
|
|
33
|
+
/** What every run before this contract effectively used. */
|
|
34
|
+
export const DEFAULT_ORACLE = {
|
|
35
|
+
mode: 'golden',
|
|
36
|
+
volatility: 'low',
|
|
37
|
+
compare: { entities: 'set', ordering: 'strict' },
|
|
38
|
+
};
|
|
39
|
+
/**
|
|
40
|
+
* Merges field by field, so an override may name only `mode` and inherit the
|
|
41
|
+
* rest. Volatility is not uniformly a property of a site: the same host serves
|
|
42
|
+
* a search that rotates in minutes and a detail page that does not.
|
|
43
|
+
*/
|
|
44
|
+
export function resolveOracle(site, task) {
|
|
45
|
+
return {
|
|
46
|
+
mode: task?.mode ?? site?.mode ?? DEFAULT_ORACLE.mode,
|
|
47
|
+
volatility: task?.volatility ?? site?.volatility ?? DEFAULT_ORACLE.volatility,
|
|
48
|
+
compare: {
|
|
49
|
+
entities: 'set',
|
|
50
|
+
ordering: task?.compare?.ordering ?? site?.compare?.ordering ?? DEFAULT_ORACLE.compare.ordering,
|
|
51
|
+
fields: task?.compare?.fields ?? site?.compare?.fields,
|
|
52
|
+
},
|
|
53
|
+
semantics: task?.semantics ?? site?.semantics,
|
|
54
|
+
};
|
|
55
|
+
}
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
import { isEquivalent } from './grade.js';
|
|
2
|
+
const LEVELS = {
|
|
3
|
+
'http-html': 'L0',
|
|
4
|
+
'http-json': 'L1',
|
|
5
|
+
'warm-browser': 'L2',
|
|
6
|
+
browser: 'L3',
|
|
7
|
+
};
|
|
8
|
+
export const levelOf = (strategy) => LEVELS[strategy];
|
|
9
|
+
function median(values) {
|
|
10
|
+
if (values.length === 0)
|
|
11
|
+
return 0;
|
|
12
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
13
|
+
const mid = Math.floor(sorted.length / 2);
|
|
14
|
+
return sorted.length % 2 === 1 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2;
|
|
15
|
+
}
|
|
16
|
+
const share = (rows, predicate) => rows.length === 0 ? 0 : rows.filter(predicate).length / rows.length;
|
|
17
|
+
export function summarize(runs, baseline) {
|
|
18
|
+
const sets = ['controlled', 'wild'];
|
|
19
|
+
return sets.map((set) => {
|
|
20
|
+
const mine = runs.filter((r) => r.set === set);
|
|
21
|
+
const theirs = baseline.filter((r) => r.set === set);
|
|
22
|
+
const medianLatencyMs = median(mine.map((r) => r.meta.latencyMs));
|
|
23
|
+
const baselineMedianLatencyMs = median(theirs.map((r) => r.meta.latencyMs));
|
|
24
|
+
const at = (level) => share(mine, (r) => levelOf(r.meta.strategy) === level);
|
|
25
|
+
const sites = [...new Set(mine.map((r) => r.site))];
|
|
26
|
+
const siteSpeedups = sites.map((site) => {
|
|
27
|
+
const ours = median(mine.filter((r) => r.site === site).map((r) => r.meta.latencyMs));
|
|
28
|
+
const theirsHere = median(theirs.filter((r) => r.site === site).map((r) => r.meta.latencyMs));
|
|
29
|
+
return ours === 0 ? 0 : theirsHere / ours;
|
|
30
|
+
});
|
|
31
|
+
return {
|
|
32
|
+
set,
|
|
33
|
+
tasks: mine.length,
|
|
34
|
+
browserAvoidance: share(mine, (r) => r.meta.browserLaunches === 0 && !r.threw),
|
|
35
|
+
medianSiteSpeedup: median(siteSpeedups),
|
|
36
|
+
levels: { L0: at('L0'), L1: at('L1'), L2: at('L2'), L3: at('L3') },
|
|
37
|
+
browserFree: share(mine, (r) => ['L0', 'L1'].includes(levelOf(r.meta.strategy))),
|
|
38
|
+
fullBrowserAvoidance: share(mine, (r) => r.meta.browserLaunches === 0 && !r.threw),
|
|
39
|
+
blockedTasks: mine.filter((r) => r.blocked).length,
|
|
40
|
+
medianElapsedMs: mine.length > 0 && mine.every((r) => r.meta.elapsedMs !== undefined)
|
|
41
|
+
? median(mine.map((r) => r.meta.elapsedMs)) : null,
|
|
42
|
+
baselineMedianElapsedMs: theirs.length > 0 && theirs.every((r) => r.meta.elapsedMs !== undefined)
|
|
43
|
+
? median(theirs.map((r) => r.meta.elapsedMs)) : null,
|
|
44
|
+
medianLatencyMs,
|
|
45
|
+
baselineMedianLatencyMs,
|
|
46
|
+
speedup: medianLatencyMs === 0 ? 0 : baselineMedianLatencyMs / medianLatencyMs,
|
|
47
|
+
successRate: share(mine, (r) => r.success),
|
|
48
|
+
schemaRate: share(mine, (r) => r.schema !== false),
|
|
49
|
+
// This path predates the oracle contract: it never marks a baseline
|
|
50
|
+
// unusable, and it never declares a query-semantics rule.
|
|
51
|
+
usableTasks: mine.length,
|
|
52
|
+
excludedTasks: 0,
|
|
53
|
+
excludedBaselineTasks: 0,
|
|
54
|
+
excludedNoGoldenTasks: 0,
|
|
55
|
+
entityRate: share(mine, (r) => r.schema !== false),
|
|
56
|
+
querySemanticsRate: null,
|
|
57
|
+
querySemanticsJudged: 0,
|
|
58
|
+
orderingRate: (() => {
|
|
59
|
+
const judged = mine.filter((r) => r.ordering !== null && r.ordering !== undefined);
|
|
60
|
+
return judged.length === 0 ? null : share(judged, (r) => r.ordering === true);
|
|
61
|
+
})(),
|
|
62
|
+
medianPolitenessWaitMs: median(mine.map((r) => r.meta.politenessWaitMs)),
|
|
63
|
+
medianBytes: median(mine.map((r) => r.meta.bytesDownloaded)),
|
|
64
|
+
baselineMedianBytes: median(theirs.map((r) => r.meta.bytesDownloaded)),
|
|
65
|
+
medianRequests: median(mine.map((r) => r.meta.networkRequests)),
|
|
66
|
+
baselineMedianRequests: median(theirs.map((r) => r.meta.networkRequests)),
|
|
67
|
+
medianAgentTokens: median(mine.map((r) => r.meta.llmTokens)),
|
|
68
|
+
baselineMedianAgentTokens: median(theirs.map((r) => r.meta.llmTokens)),
|
|
69
|
+
};
|
|
70
|
+
});
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* Summarises pairs rather than two independent lists.
|
|
74
|
+
*
|
|
75
|
+
* A task whose baseline was unusable leaves the equivalence denominator — the
|
|
76
|
+
* engine cannot be blamed for a browser failure, and counting it as a pass
|
|
77
|
+
* would be worse — but it never leaves the report.
|
|
78
|
+
*/
|
|
79
|
+
export function summarizePairs(pairs) {
|
|
80
|
+
const sets = ['controlled', 'wild'];
|
|
81
|
+
return sets.map((set) => {
|
|
82
|
+
const mine = pairs.filter((p) => p.task.set === set);
|
|
83
|
+
const usable = mine.filter((p) => p.grade.usable);
|
|
84
|
+
const engineRuns = mine.map((p) => p.engine);
|
|
85
|
+
const baselineRuns = mine.map((p) => p.baseline);
|
|
86
|
+
const base = summarize(engineRuns, baselineRuns).find((s) => s.set === set);
|
|
87
|
+
const judgedOrdering = usable.filter((p) => p.grade.ordering !== null);
|
|
88
|
+
const judgedQuery = usable.filter((p) => p.grade.query !== null);
|
|
89
|
+
return {
|
|
90
|
+
...base,
|
|
91
|
+
tasks: mine.length,
|
|
92
|
+
usableTasks: usable.length,
|
|
93
|
+
excludedTasks: mine.length - usable.length,
|
|
94
|
+
// Two different reasons look identical as "not usable": a task with no
|
|
95
|
+
// stored golden had a perfectly healthy browser run, and blaming that on
|
|
96
|
+
// baseline validity is the exact confusion this split exists to end.
|
|
97
|
+
excludedBaselineTasks: mine.filter((p) => p.grade.excludeReason === 'baseline').length,
|
|
98
|
+
excludedNoGoldenTasks: mine.filter((p) => p.grade.excludeReason === 'no-golden').length,
|
|
99
|
+
entityRate: share(usable, (p) => p.grade.entities === true),
|
|
100
|
+
// Denominator is the tasks an oracle actually asked about, so an
|
|
101
|
+
// undeclared rule cannot inflate the rate with unexamined passes.
|
|
102
|
+
querySemanticsJudged: judgedQuery.length,
|
|
103
|
+
querySemanticsRate: judgedQuery.length === 0 ? null : share(judgedQuery, (p) => p.grade.query === true),
|
|
104
|
+
successRate: share(usable, (p) => isEquivalent(p.grade)),
|
|
105
|
+
orderingRate: judgedOrdering.length === 0 ? null : share(judgedOrdering, (p) => p.grade.ordering === true),
|
|
106
|
+
};
|
|
107
|
+
});
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* The tasks that were judged and did not pass, in the same terms the CLI
|
|
111
|
+
* prints — extracted so the CLI cannot hand-copy `isEquivalent` and drift, the
|
|
112
|
+
* way its old failure filter did.
|
|
113
|
+
*/
|
|
114
|
+
export function failedPairs(pairs) {
|
|
115
|
+
return pairs
|
|
116
|
+
.filter((p) => p.grade.usable && !isEquivalent(p.grade))
|
|
117
|
+
.map((p) => ({
|
|
118
|
+
taskId: p.task.id,
|
|
119
|
+
// `grade.reasons` holds only the oracle's verdict text; an engine that
|
|
120
|
+
// timed out or fell back leaves that reason on the run itself, and a
|
|
121
|
+
// printed failure line is useless without it.
|
|
122
|
+
reasons: [...new Set([...p.grade.reasons, ...p.engine.reasons])],
|
|
123
|
+
}));
|
|
124
|
+
}
|
|
125
|
+
/** Tasks left out of the denominator for one specific cause, so the two never print under one heading. */
|
|
126
|
+
export function excludedBy(pairs, cause) {
|
|
127
|
+
return pairs
|
|
128
|
+
.filter((p) => p.grade.excludeReason === cause)
|
|
129
|
+
.map((p) => ({ taskId: p.task.id, reason: p.grade.validityReason }));
|
|
130
|
+
}
|
|
131
|
+
const pct = (n) => `${(n * 100).toFixed(0)}%`;
|
|
132
|
+
/**
|
|
133
|
+
* One cost axis, engine against baseline. Tokens is the first axis where the
|
|
134
|
+
* engine can come out worse, so the direction is read off the numbers rather
|
|
135
|
+
* than assumed to be a saving.
|
|
136
|
+
*/
|
|
137
|
+
const ratio = (engine, baseline) => {
|
|
138
|
+
if (engine === 0 || baseline === 0)
|
|
139
|
+
return '—';
|
|
140
|
+
return baseline >= engine ? `${(baseline / engine).toFixed(1)}x less` : `${(engine / baseline).toFixed(1)}x more`;
|
|
141
|
+
};
|
|
142
|
+
export function formatReport(summaries, robots, opts = {}) {
|
|
143
|
+
const { correctness = true } = opts;
|
|
144
|
+
const lines = [];
|
|
145
|
+
for (const s of summaries) {
|
|
146
|
+
if (s.tasks === 0)
|
|
147
|
+
continue;
|
|
148
|
+
lines.push(s.set === 'controlled' ? 'Controlled' : 'Wild');
|
|
149
|
+
lines.push(` Tasks: ${s.tasks}`);
|
|
150
|
+
lines.push(` Browser-free: ${pct(s.browserFree)} (L0 + L1)`);
|
|
151
|
+
lines.push(` Full-browser avoidance: ${pct(s.fullBrowserAvoidance)} (observed zero launches)`);
|
|
152
|
+
if (s.medianElapsedMs !== null) {
|
|
153
|
+
lines.push(` Actual elapsed: ${s.medianElapsedMs}ms median (baseline ${s.baselineMedianElapsedMs ?? 'unavailable'}ms)`);
|
|
154
|
+
if (s.medianElapsedMs > 0 && s.baselineMedianElapsedMs !== null) {
|
|
155
|
+
lines.push(` Actual speedup: ${(s.baselineMedianElapsedMs / s.medianElapsedMs).toFixed(1)}x`);
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
lines.push(` Levels: L0 ${pct(s.levels.L0)} L1 ${pct(s.levels.L1)} L2 ${pct(s.levels.L2)} L3 ${pct(s.levels.L3)}`);
|
|
159
|
+
lines.push(` Blocked: ${s.blockedTasks} (site refused the recipe; ran on the browser)`);
|
|
160
|
+
lines.push(` Median latency: ${s.medianLatencyMs}ms (baseline ${s.baselineMedianLatencyMs}ms; legacy attempt timing)`);
|
|
161
|
+
lines.push(` Median speedup: ${s.speedup.toFixed(1)}x (legacy timing)`);
|
|
162
|
+
lines.push(` Median site speedup: ${s.medianSiteSpeedup.toFixed(1)}x`);
|
|
163
|
+
lines.push(` Politeness wait: ${s.medianPolitenessWaitMs}ms median (excluded from latency above)`);
|
|
164
|
+
// Latency is the noisiest of the three cost axes and was the only one
|
|
165
|
+
// reported. Bytes and requests were instrumented from the first commit and
|
|
166
|
+
// discarded on every run until held-out 3.
|
|
167
|
+
const kb = (b) => `${(b / 1024).toFixed(0)}KB`;
|
|
168
|
+
lines.push(` Data fetched: ${kb(s.medianBytes)} median (baseline ${kb(s.baselineMedianBytes)}) ${ratio(s.medianBytes, s.baselineMedianBytes)}`);
|
|
169
|
+
lines.push(` Requests: ${s.medianRequests} median (baseline ${s.baselineMedianRequests})`);
|
|
170
|
+
lines.push(` Agent reads: ${s.medianAgentTokens} tokens median (baseline ${s.baselineMedianAgentTokens}) ${ratio(s.medianAgentTokens, s.baselineMedianAgentTokens)}`);
|
|
171
|
+
// A raw browser-only run (`bench run --baseline`) never grades an engine
|
|
172
|
+
// against anything, so it has none of these figures to report — printing
|
|
173
|
+
// them anyway would be four false labels, not one.
|
|
174
|
+
if (correctness) {
|
|
175
|
+
lines.push(' Correctness');
|
|
176
|
+
lines.push(` Baseline validity: ${s.tasks - s.excludedBaselineTasks}/${s.tasks} usable (${s.excludedBaselineTasks} excluded)`);
|
|
177
|
+
lines.push(` No stored golden: ${s.excludedNoGoldenTasks} excluded`);
|
|
178
|
+
lines.push(` Entity equivalence: ${pct(s.entityRate)}`);
|
|
179
|
+
lines.push(` Query semantics: ${s.querySemanticsRate === null ? 'not declared' : `${pct(s.querySemanticsRate)} of ${s.querySemanticsJudged} judgeable`}`);
|
|
180
|
+
lines.push(` Ordering: ${s.orderingRate === null ? 'not judgeable' : `${pct(s.orderingRate)} where meaningful`}`);
|
|
181
|
+
}
|
|
182
|
+
lines.push('');
|
|
183
|
+
}
|
|
184
|
+
if (Object.keys(robots).length > 0) {
|
|
185
|
+
lines.push('robots.txt status of wild sites');
|
|
186
|
+
for (const [site, status] of Object.entries(robots))
|
|
187
|
+
lines.push(` ${site}: ${status}`);
|
|
188
|
+
lines.push('');
|
|
189
|
+
}
|
|
190
|
+
return lines.join('\n');
|
|
191
|
+
}
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
import { ExecutionFailure } from '../measurement.js';
|
|
2
|
+
import { mkdir, readFile, writeFile } from 'node:fs/promises';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
import { z } from 'zod';
|
|
5
|
+
import { gradeGolden } from './golden.js';
|
|
6
|
+
import { gradePair, planAging, querySemantics } from './grade.js';
|
|
7
|
+
import { OracleSpecSchema, resolveOracle } from './oracle.js';
|
|
8
|
+
const BenchTaskSchema = z.object({
|
|
9
|
+
id: z.string(),
|
|
10
|
+
set: z.enum(['controlled', 'wild']),
|
|
11
|
+
site: z.string(),
|
|
12
|
+
intent: z.enum(['search', 'list', 'detail']),
|
|
13
|
+
input: z.record(z.union([z.string(), z.number()])),
|
|
14
|
+
volatile: z.array(z.string()).optional(),
|
|
15
|
+
/** A search that legitimately returns nothing; an empty golden is correct here. */
|
|
16
|
+
expectEmpty: z.boolean().optional(),
|
|
17
|
+
oracle: OracleSpecSchema.optional(),
|
|
18
|
+
});
|
|
19
|
+
export async function loadTasks(path) {
|
|
20
|
+
const raw = JSON.parse(await readFile(path, 'utf8'));
|
|
21
|
+
return z.array(BenchTaskSchema).parse(raw);
|
|
22
|
+
}
|
|
23
|
+
function goldenPath(dir, taskId) {
|
|
24
|
+
return join(dir, `${taskId}.json`);
|
|
25
|
+
}
|
|
26
|
+
function failedRun(task, error) {
|
|
27
|
+
const reason = message(error);
|
|
28
|
+
return {
|
|
29
|
+
taskId: task.id,
|
|
30
|
+
site: task.site,
|
|
31
|
+
set: task.set,
|
|
32
|
+
meta: error instanceof ExecutionFailure ? error.meta : {
|
|
33
|
+
strategy: 'browser', latencyMs: 0, browserLaunches: 0,
|
|
34
|
+
pageNavigations: 0, networkRequests: 0, bytesDownloaded: 0, llmTokens: 0, politenessWaitMs: 0,
|
|
35
|
+
},
|
|
36
|
+
success: false,
|
|
37
|
+
reasons: [reason],
|
|
38
|
+
items: [],
|
|
39
|
+
threw: true,
|
|
40
|
+
blocked: false,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
const message = (err) => (err instanceof Error ? err.message : String(err));
|
|
44
|
+
/** Explicit, never implicit: a golden is only written when this is called. */
|
|
45
|
+
export async function captureGoldens(tasks, deps) {
|
|
46
|
+
await mkdir(deps.goldenDir, { recursive: true });
|
|
47
|
+
const goldens = [];
|
|
48
|
+
for (const task of tasks) {
|
|
49
|
+
const result = await deps.browser.execute({}, task);
|
|
50
|
+
const golden = { taskId: task.id, capturedAt: new Date().toISOString(), items: result.items };
|
|
51
|
+
await writeFile(goldenPath(deps.goldenDir, task.id), JSON.stringify(golden, null, 2), 'utf8');
|
|
52
|
+
goldens.push(golden);
|
|
53
|
+
}
|
|
54
|
+
return goldens;
|
|
55
|
+
}
|
|
56
|
+
export async function loadGoldens(dir, tasks) {
|
|
57
|
+
const goldens = [];
|
|
58
|
+
for (const task of tasks) {
|
|
59
|
+
try {
|
|
60
|
+
goldens.push(JSON.parse(await readFile(goldenPath(dir, task.id), 'utf8')));
|
|
61
|
+
}
|
|
62
|
+
catch {
|
|
63
|
+
// A missing golden means this task has never been captured; skip it.
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
return goldens;
|
|
67
|
+
}
|
|
68
|
+
export async function runBaseline(tasks, goldens, deps) {
|
|
69
|
+
const byId = new Map(goldens.map((g) => [g.taskId, g]));
|
|
70
|
+
const runs = [];
|
|
71
|
+
for (const task of tasks) {
|
|
72
|
+
try {
|
|
73
|
+
const result = await deps.browser.execute({}, task);
|
|
74
|
+
const golden = byId.get(task.id);
|
|
75
|
+
const grade = golden
|
|
76
|
+
? gradeGolden(golden, result.items, task.volatile)
|
|
77
|
+
: { schema: true, semantic: true, ordering: null, reasons: [] };
|
|
78
|
+
runs.push({
|
|
79
|
+
taskId: task.id, site: task.site, set: task.set, meta: result.meta,
|
|
80
|
+
success: grade.schema && grade.semantic, schema: grade.schema, ordering: grade.ordering,
|
|
81
|
+
reasons: grade.reasons,
|
|
82
|
+
items: result.items, threw: false, blocked: false,
|
|
83
|
+
});
|
|
84
|
+
}
|
|
85
|
+
catch (err) {
|
|
86
|
+
runs.push(failedRun(task, err));
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return runs;
|
|
90
|
+
}
|
|
91
|
+
async function runBenchmark(tasks, goldens, deps) {
|
|
92
|
+
const byId = new Map(goldens.map((g) => [g.taskId, g]));
|
|
93
|
+
const runs = [];
|
|
94
|
+
for (const task of tasks) {
|
|
95
|
+
try {
|
|
96
|
+
const outcome = await deps.executor.run(task);
|
|
97
|
+
const golden = byId.get(task.id);
|
|
98
|
+
const grade = golden
|
|
99
|
+
? gradeGolden(golden, outcome.items, task.volatile)
|
|
100
|
+
: { schema: true, semantic: true, ordering: null, reasons: [] };
|
|
101
|
+
runs.push({
|
|
102
|
+
taskId: task.id, site: task.site, set: task.set, meta: outcome.meta,
|
|
103
|
+
success: grade.schema && grade.semantic, schema: grade.schema, ordering: grade.ordering,
|
|
104
|
+
reasons: [...outcome.reasons, ...grade.reasons],
|
|
105
|
+
items: outcome.items, threw: false, blocked: outcome.blocked,
|
|
106
|
+
});
|
|
107
|
+
}
|
|
108
|
+
catch (err) {
|
|
109
|
+
runs.push(failedRun(task, err));
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return runs;
|
|
113
|
+
}
|
|
114
|
+
/** An engine that threw produced no items to grade; scoring it against an
|
|
115
|
+
* empty reference (an `expectEmpty` baseline, an empty golden) would read as a
|
|
116
|
+
* pass, which is worse than the failure it stands in for. Decided before
|
|
117
|
+
* either mode's layer two runs, so the two never diverge on it. */
|
|
118
|
+
function engineException(engine) {
|
|
119
|
+
return {
|
|
120
|
+
entities: false, query: null, ordering: null,
|
|
121
|
+
reasons: engine.reasons.length > 0 ? engine.reasons : ['engine threw'],
|
|
122
|
+
};
|
|
123
|
+
}
|
|
124
|
+
/**
|
|
125
|
+
* Grades by mode, because the two modes ask different questions.
|
|
126
|
+
*
|
|
127
|
+
* `golden` asks whether the engine matches a stored capture. A live browser
|
|
128
|
+
* that times out says nothing about that, so it must not gate it — routing
|
|
129
|
+
* every mode through `gradePair` would have excluded a task whose engine
|
|
130
|
+
* matched its golden exactly, on the strength of an unrelated timeout. The
|
|
131
|
+
* baseline is still run, because the speedup figure needs it.
|
|
132
|
+
*
|
|
133
|
+
* `paired-live` asks whether the engine matches the browser beside it, so there
|
|
134
|
+
* the baseline's validity is exactly the right gate.
|
|
135
|
+
*/
|
|
136
|
+
export function gradeByOracle(oracle, stored, baseline, engine, task) {
|
|
137
|
+
if (oracle.mode === 'paired-live') {
|
|
138
|
+
const grade = gradePair(baseline.items, engine.items, oracle, {
|
|
139
|
+
threw: baseline.threw,
|
|
140
|
+
expectEmpty: task.expectEmpty === true,
|
|
141
|
+
input: task.input,
|
|
142
|
+
});
|
|
143
|
+
return grade.usable && engine.threw ? { ...grade, ...engineException(engine) } : grade;
|
|
144
|
+
}
|
|
145
|
+
if (stored === undefined) {
|
|
146
|
+
return {
|
|
147
|
+
usable: false, validityReason: 'no stored golden for this task', excludeReason: 'no-golden',
|
|
148
|
+
entities: null, query: null, ordering: null,
|
|
149
|
+
reasons: ['no stored golden for this task'],
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
if (engine.threw) {
|
|
153
|
+
return { usable: true, validityReason: null, excludeReason: null, ...engineException(engine) };
|
|
154
|
+
}
|
|
155
|
+
const graded = gradeGolden(stored, engine.items, task.volatile, oracle.compare.fields);
|
|
156
|
+
const query = querySemantics(engine.items, task.input, oracle);
|
|
157
|
+
return {
|
|
158
|
+
usable: true,
|
|
159
|
+
validityReason: null,
|
|
160
|
+
excludeReason: null,
|
|
161
|
+
entities: graded.schema && graded.semantic,
|
|
162
|
+
query: query.match,
|
|
163
|
+
ordering: oracle.compare.ordering === 'ignore' ? null : graded.ordering,
|
|
164
|
+
reasons: [...graded.reasons, ...query.reasons],
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
/**
|
|
168
|
+
* Runs both sides of each task before moving to the next, alternating which
|
|
169
|
+
* goes first.
|
|
170
|
+
*
|
|
171
|
+
* Running the whole baseline and then the whole engine put every engine
|
|
172
|
+
* measurement later in time than its baseline, which on a job board cost 25
|
|
173
|
+
* points of apparent correctness. Pairing removes that; alternating removes the
|
|
174
|
+
* residual bias of one side always going first. Comparison happens once both
|
|
175
|
+
* halves are in hand, so order within a pair does not decide which is the
|
|
176
|
+
* reference.
|
|
177
|
+
*/
|
|
178
|
+
export async function runPairs(tasks, goldens, deps, oracles) {
|
|
179
|
+
const byId = new Map(goldens.map((g) => [g.taskId, g]));
|
|
180
|
+
const pairs = [];
|
|
181
|
+
for (const [index, task] of tasks.entries()) {
|
|
182
|
+
const oracle = resolveOracle(OracleSpecSchema.optional().parse(oracles[task.site]), task.oracle);
|
|
183
|
+
const engineFirst = index % 2 === 1;
|
|
184
|
+
let baseline;
|
|
185
|
+
let engine;
|
|
186
|
+
if (engineFirst) {
|
|
187
|
+
engine = (await runBenchmark([task], goldens, deps))[0];
|
|
188
|
+
baseline = (await runBaseline([task], goldens, deps))[0];
|
|
189
|
+
}
|
|
190
|
+
else {
|
|
191
|
+
baseline = (await runBaseline([task], goldens, deps))[0];
|
|
192
|
+
engine = (await runBenchmark([task], goldens, deps))[0];
|
|
193
|
+
}
|
|
194
|
+
const grade = gradeByOracle(oracle, byId.get(task.id), baseline, engine, task);
|
|
195
|
+
const aging = oracle.mode === 'paired-live'
|
|
196
|
+
? planAging(byId.get(task.id)?.items, baseline.items, oracle)
|
|
197
|
+
: null;
|
|
198
|
+
pairs.push({ task, oracle, baseline, engine, grade, aging });
|
|
199
|
+
}
|
|
200
|
+
return pairs;
|
|
201
|
+
}
|