webrecipe 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +253 -0
  3. package/dist/benchmark/amortization.js +254 -0
  4. package/dist/benchmark/fixtures.js +26 -0
  5. package/dist/benchmark/oracles.js +129 -0
  6. package/dist/benchmark/plans.js +436 -0
  7. package/dist/fixtures/cloaking.js +37 -0
  8. package/dist/fixtures/coalesce.js +52 -0
  9. package/dist/fixtures/data.js +23 -0
  10. package/dist/fixtures/harness.js +34 -0
  11. package/dist/fixtures/ignoring.js +27 -0
  12. package/dist/fixtures/limiting.js +38 -0
  13. package/dist/fixtures/paging.js +72 -0
  14. package/dist/fixtures/refusing.js +57 -0
  15. package/dist/fixtures/shifted.js +32 -0
  16. package/dist/fixtures/spa.js +71 -0
  17. package/dist/fixtures/ssr.js +46 -0
  18. package/dist/fixtures/volatile.js +40 -0
  19. package/dist/fixtures/xhr.js +120 -0
  20. package/dist/src/analyzer/classify.js +16 -0
  21. package/dist/src/analyzer/score.js +52 -0
  22. package/dist/src/authoring/candidates.js +168 -0
  23. package/dist/src/authoring/contract.js +31 -0
  24. package/dist/src/authoring/fields.js +86 -0
  25. package/dist/src/authoring/learn.js +51 -0
  26. package/dist/src/authoring/plans.js +93 -0
  27. package/dist/src/authoring/snapshot.js +22 -0
  28. package/dist/src/authoring/teach.js +136 -0
  29. package/dist/src/benchmark/discovery.js +355 -0
  30. package/dist/src/benchmark/golden.js +95 -0
  31. package/dist/src/benchmark/grade.js +146 -0
  32. package/dist/src/benchmark/ground-truth.js +35 -0
  33. package/dist/src/benchmark/health.js +96 -0
  34. package/dist/src/benchmark/labels.js +49 -0
  35. package/dist/src/benchmark/oracle.js +55 -0
  36. package/dist/src/benchmark/report.js +191 -0
  37. package/dist/src/benchmark/runner.js +201 -0
  38. package/dist/src/benchmark/screen.js +144 -0
  39. package/dist/src/benchmark/selector-score.js +86 -0
  40. package/dist/src/benchmark/verification-cases.js +138 -0
  41. package/dist/src/benchmark/verification-matrix.js +97 -0
  42. package/dist/src/browser/navigate.js +22 -0
  43. package/dist/src/browser/pool.js +31 -0
  44. package/dist/src/browser/session.js +44 -0
  45. package/dist/src/cli.js +559 -0
  46. package/dist/src/compiler/derive.js +144 -0
  47. package/dist/src/compiler/heuristic.js +398 -0
  48. package/dist/src/compiler/html.js +117 -0
  49. package/dist/src/compiler/types.js +12 -0
  50. package/dist/src/compiler/verify.js +29 -0
  51. package/dist/src/executor/extract.js +179 -0
  52. package/dist/src/executor/format.js +55 -0
  53. package/dist/src/executor/index.js +147 -0
  54. package/dist/src/executor/strategies/browser.js +60 -0
  55. package/dist/src/executor/strategies/http-html.js +42 -0
  56. package/dist/src/executor/strategies/http-json.js +71 -0
  57. package/dist/src/executor/strategies/warm-browser.js +57 -0
  58. package/dist/src/executor/tokens.js +11 -0
  59. package/dist/src/healing/index.js +111 -0
  60. package/dist/src/local.js +157 -0
  61. package/dist/src/mcp.js +130 -0
  62. package/dist/src/measurement.js +44 -0
  63. package/dist/src/net/politeness.js +141 -0
  64. package/dist/src/net/robots.js +56 -0
  65. package/dist/src/read.js +83 -0
  66. package/dist/src/recipes/fingerprint.js +41 -0
  67. package/dist/src/recipes/paths.js +14 -0
  68. package/dist/src/recipes/registry.js +81 -0
  69. package/dist/src/recipes/schema.js +38 -0
  70. package/dist/src/recipes/template.js +33 -0
  71. package/dist/src/recorder/body.js +59 -0
  72. package/dist/src/recorder/index.js +151 -0
  73. package/dist/src/recorder/types.js +1 -0
  74. package/dist/src/sites.js +45 -0
  75. package/dist/src/tasks.js +37 -0
  76. package/dist/src/types.js +32 -0
  77. package/dist/src/usage.js +69 -0
  78. package/dist/src/validator/index.js +28 -0
  79. package/dist/src/verification/lexical-consistency.js +88 -0
  80. package/dist/src/verification/pagination-honored.js +110 -0
  81. package/dist/src/verification/probes.js +98 -0
  82. package/dist/src/verification/query-honored.js +134 -0
  83. package/dist/src/wiring.js +33 -0
  84. package/package.json +56 -0
@@ -0,0 +1,69 @@
1
+ import { createHash } from 'node:crypto';
2
+ import { siteName, intentName } from './local.js';
3
+ import { appendFile, mkdir, readdir, readFile } from 'node:fs/promises';
4
+ import { join } from 'node:path';
5
+ /** Local only. No HTML, result contents, cookies or headers. */
6
+ export async function appendUsage(root, event) {
7
+ const dir = join(root, 'logs');
8
+ await mkdir(dir, { recursive: true, mode: 0o700 });
9
+ await appendFile(join(dir, `${event.at.slice(0, 10)}.jsonl`), JSON.stringify(event) + '\n', { mode: 0o600 });
10
+ }
11
+ export async function readUsage(root, days) {
12
+ const dir = join(root, 'logs');
13
+ const files = await readdir(dir).catch((error) => { if (error.code === 'ENOENT')
14
+ return []; throw error; });
15
+ const since = Date.now() - days * 86_400_000;
16
+ const events = [];
17
+ let malformed = 0;
18
+ for (const file of files.filter(f => /^\d{4}-\d{2}-\d{2}\.jsonl$/.test(f)).sort()) {
19
+ if (file.slice(0, 10) < new Date(since).toISOString().slice(0, 10))
20
+ continue;
21
+ for (const line of (await readFile(join(dir, file), 'utf8')).split('\n').filter(Boolean)) {
22
+ try {
23
+ const event = JSON.parse(line);
24
+ if (event.version !== 1 || !['start', 'finish', 'feedback'].includes(event.event) || typeof event.id !== 'string' || !Number.isFinite(Date.parse(event.at)))
25
+ throw new Error('invalid record');
26
+ if (Date.parse(event.at) >= since)
27
+ events.push(event);
28
+ }
29
+ catch {
30
+ malformed++;
31
+ }
32
+ }
33
+ }
34
+ return { events, malformed };
35
+ }
36
+ export function summarizeUsage(events) {
37
+ const starts = events.filter(e => e.event === 'start');
38
+ const finishes = events.filter(e => e.event === 'finish');
39
+ const runs = finishes.filter(e => e.command === 'fetch');
40
+ const completed = new Set(finishes.map(e => e.id));
41
+ const times = runs.map(e => Number(e.wallMs)).filter(Number.isFinite).sort((a, b) => a - b);
42
+ const feedback = new Map(events.filter(e => e.event === 'feedback').map(e => [e.id, e.verdict]));
43
+ return {
44
+ started: starts.length, finished: finishes.length,
45
+ incomplete: starts.filter(e => !completed.has(e.id)).map(e => e.id),
46
+ measuredCommands: finishes.filter(e => e.meta !== undefined).length,
47
+ totalRequests: finishes.reduce((sum, e) => sum + Number(e.meta?.networkRequests ?? 0), 0),
48
+ totalPolitenessWaitMs: finishes.reduce((sum, e) => sum + Number(e.meta?.politenessWaitMs ?? 0), 0),
49
+ fallbackRuns: runs.filter(e => e.fellBack === true).length,
50
+ recipeChanges: finishes.filter(e => e.recipeChanged === true).length,
51
+ runs: runs.length, succeeded: runs.filter(e => e.ok === true).length,
52
+ failed: runs.filter(e => e.ok === false).length,
53
+ medianRunWallMs: times.length ? (times[Math.floor((times.length - 1) / 2)] + times[Math.floor(times.length / 2)]) / 2 : null,
54
+ browserFreeRuns: runs.filter(e => e.ok === true && e.meta?.browserLaunches === 0).length,
55
+ feedback: { correct: [...feedback.values()].filter(v => v === 'correct').length, wrong: [...feedback.values()].filter(v => v === 'wrong').length },
56
+ recent: finishes.slice(-10).map(({ id, command, site, intent, ok, wallMs, error }) => ({ id, command, site, intent, ok, wallMs, error })),
57
+ };
58
+ }
59
+ export async function recipeDigest(root, site, intent) {
60
+ const path = join(root, 'recipes', siteName(site), `${intentName(intent)}.yaml`);
61
+ try {
62
+ return createHash('sha256').update(await readFile(path)).digest('hex');
63
+ }
64
+ catch (error) {
65
+ if (error.code === 'ENOENT')
66
+ return null;
67
+ throw error;
68
+ }
69
+ }
@@ -0,0 +1,28 @@
1
+ import { computeFingerprint } from '../recipes/fingerprint.js';
2
+ export function validate(recipe, input) {
3
+ const reasons = [];
4
+ if (input.status !== recipe.validation.status) {
5
+ reasons.push(`status ${input.status} !== expected ${recipe.validation.status}`);
6
+ }
7
+ if (input.items.length < recipe.validation.minItems) {
8
+ reasons.push(`item count ${input.items.length} < minItems ${recipe.validation.minItems}`);
9
+ }
10
+ // A real listing mixes in rows the selector was never meant to describe — a
11
+ // sponsored insert among fifty jobs. Failing on any empty value made one such
12
+ // row invalidate the whole recipe. What this check is for is a dead selector,
13
+ // so it fires when the field is missing from most of the result.
14
+ for (const field of recipe.validation.required) {
15
+ const missing = input.items.filter((item) => {
16
+ const value = item[field];
17
+ return value === null || value === undefined || value === '';
18
+ });
19
+ if (missing.length > input.items.length / 2) {
20
+ reasons.push(`required field "${field}" empty in ${missing.length}/${input.items.length} items`);
21
+ }
22
+ }
23
+ const actual = computeFingerprint(recipe.fingerprint.endpoint, input.payload);
24
+ if (actual !== recipe.fingerprint.hash) {
25
+ reasons.push(`fingerprint drift: expected ${recipe.fingerprint.hash}, got ${actual}`);
26
+ }
27
+ return { valid: reasons.length === 0, reasons };
28
+ }
@@ -0,0 +1,88 @@
1
+ import { tokensOf, visibleText } from './query-honored.js';
2
+ /**
3
+ * Whether the rows a search returns visibly carry the words that were searched
4
+ * for, on the taught input and on one that was never taught.
5
+ *
6
+ * This exists because `query_honored` passes a site that filters on the query
7
+ * and then attributes each hit to the wrong record: the input genuinely moves
8
+ * the answer, which is all those probes can see. What such a site cannot do is
9
+ * show the query's own words in what it returns.
10
+ *
11
+ * Named for the evidence and no wider. Passing says the extracted fields match
12
+ * the query lexically and kept doing so for a value never taught. It does not
13
+ * say the results are relevant, that the entities are the right ones, or that a
14
+ * site matching on synonyms, stems or fields we cannot see is wrong — those
15
+ * come back untested, because there is nothing here that could tell them from a
16
+ * site that is broken.
17
+ */
18
+ export const METHOD = 'active-probe/lexical-consistency-v1';
19
+ /**
20
+ * Provisional. Chosen before any evidence and deliberately not fitted: tuning
21
+ * it against six fixtures would produce a number that describes the fixtures.
22
+ * It moves only on held-out or real-site evidence.
23
+ */
24
+ export const LEXICAL_MATCH_THRESHOLD = 0.8;
25
+ /** Share of rows carrying any token of `query` in a field the plan extracts. */
26
+ export function lexicalShare(items, query) {
27
+ const tokens = tokensOf(query);
28
+ if (tokens.length === 0 || items.length === 0)
29
+ return null;
30
+ const hits = items.filter((item) => {
31
+ const text = visibleText(item);
32
+ return tokens.some((token) => text.includes(token));
33
+ }).length;
34
+ return hits / items.length;
35
+ }
36
+ export function judgeLexicalConsistency(session) {
37
+ const at = (role) => session.observations.find((o) => o.role === role);
38
+ const taught = at('taught');
39
+ if (!taught || taught.items === null) {
40
+ return { status: 'not_tested', signals: {}, shares: {}, reason: `taught probe unavailable: ${taught?.unavailable ?? 'not issued'}` };
41
+ }
42
+ // Stage one asks whether this check applies at all. A site that matches on a
43
+ // description, a tag or a synonym is answering correctly and would look
44
+ // identical to one that answers with the wrong records, so the honest report
45
+ // is that we cannot read it.
46
+ const taughtShare = lexicalShare(taught.items, session.taughtValue);
47
+ if (taughtShare === null) {
48
+ return { status: 'not_tested', signals: {}, shares: {}, reason: 'the taught query has no usable token, or returned nothing' };
49
+ }
50
+ const signals = { applicable: taughtShare >= LEXICAL_MATCH_THRESHOLD };
51
+ const shares = { taught: taughtShare };
52
+ if (!signals.applicable) {
53
+ return {
54
+ status: 'not_tested', signals, shares,
55
+ reason: `only ${(taughtShare * 100).toFixed(0)}% of the taught result carries the query in a field this plan extracts, so this site does not match on text we can read`,
56
+ };
57
+ }
58
+ const contrast = at('contrast');
59
+ if (!contrast || contrast.items === null) {
60
+ return { status: 'not_tested', signals, shares, reason: `contrast probe unavailable: ${contrast?.unavailable ?? 'no contrast token could be built'}` };
61
+ }
62
+ const contrastShare = lexicalShare(contrast.items, contrast.value);
63
+ if (contrastShare === null) {
64
+ return { status: 'not_tested', signals, shares, reason: `the contrast query "${contrast.value}" returned nothing to read` };
65
+ }
66
+ shares.contrast = contrastShare;
67
+ signals.consistent = contrastShare >= LEXICAL_MATCH_THRESHOLD;
68
+ if (!signals.consistent) {
69
+ return {
70
+ status: 'not_tested', signals, shares,
71
+ reason: `the taught query is visible in its result but "${contrast.value}" is in only ${(contrastShare * 100).toFixed(0)}% of its own, which one site matching on two different fields would also look like`,
72
+ };
73
+ }
74
+ return { status: 'passed', signals, shares, reason: '' };
75
+ }
76
+ /** Reads the same probes `query_honored` already issued; costs no new request. */
77
+ export function verifyLexicalConsistency(session) {
78
+ const verdict = judgeLexicalConsistency(session);
79
+ return {
80
+ status: verdict.status,
81
+ at: new Date().toISOString(),
82
+ method: METHOD,
83
+ parameter: session.parameter,
84
+ shares: verdict.shares,
85
+ signals: verdict.signals,
86
+ ...(verdict.reason === '' ? {} : { reason: verdict.reason }),
87
+ };
88
+ }
@@ -0,0 +1,110 @@
1
+ import { asRecords, overlapShare, sameSet } from './probes.js';
2
+ /**
3
+ * Whether changing the page input actually selects a different window, and
4
+ * whether the recipe selects the same window the page itself does.
5
+ *
6
+ * Narrower than pagination being correct. Nothing here shows that every record
7
+ * can be reached, that none is served twice, that ordering holds across the
8
+ * boundary, or that the last page is where the site says it is. Those are
9
+ * separate questions and would need separate evidence.
10
+ *
11
+ * Required exactly when the plan templates `page`, which is an authoring act
12
+ * rather than a property of the site: templating it makes `run --page N` a
13
+ * supported operation, and a task is asked to be right about what it supports.
14
+ */
15
+ export const METHOD = 'active-probe/pagination-v1';
16
+ /**
17
+ * A neighbour that probably exists, rather than the one after.
18
+ *
19
+ * Probing page + 1 from a task taught on the last page would come back empty
20
+ * and give up on a site that paginates perfectly well. What has to be shown is
21
+ * that changing the input selects a different window, and the page before does
22
+ * that as well as the page after.
23
+ */
24
+ export function alternatePage(taught) {
25
+ return taught <= 1 ? 2 : taught - 1;
26
+ }
27
+ export function judgePagination(session) {
28
+ const at = (role) => session.observations.find((o) => o.role === role);
29
+ const taught = at('taught');
30
+ const repeat = at('repeat');
31
+ const alternate = at('alternate');
32
+ const signals = {};
33
+ const shares = {};
34
+ for (const [role, probe] of [['taught', taught], ['repeat', repeat]]) {
35
+ if (!probe)
36
+ return { status: 'not_tested', signals, shares, reason: `no ${role} probe was issued` };
37
+ if (probe.items === null)
38
+ return { status: 'not_tested', signals, shares, reason: `${role} probe unavailable: ${probe.unavailable ?? 'unknown'}` };
39
+ }
40
+ if (taught.items.length === 0) {
41
+ return { status: 'not_tested', signals, shares, reason: 'the taught page returned nothing to compare against' };
42
+ }
43
+ // Volatility first: on a site whose answer changes between two identical
44
+ // requests, comparing one page against another proves nothing either way.
45
+ signals.stable = sameSet(taught.items, repeat.items);
46
+ if (!signals.stable) {
47
+ return { status: 'not_tested', signals, shares, reason: 'two identical requests returned different results; one page cannot be compared against another here' };
48
+ }
49
+ if (!alternate)
50
+ return { status: 'not_tested', signals, shares, reason: 'no alternate page probe was issued' };
51
+ if (alternate.items === null) {
52
+ return { status: 'not_tested', signals, shares, reason: `alternate page probe unavailable: ${alternate.unavailable ?? 'unknown'}` };
53
+ }
54
+ // An empty neighbour is what a real last page looks like. Never a failure.
55
+ if (alternate.items.length === 0) {
56
+ return { status: 'not_tested', signals, shares, reason: `page ${session.alternatePage} came back empty, which a last page also does` };
57
+ }
58
+ shares.overlap = overlapShare(taught.items, alternate.items);
59
+ signals.moved = !sameSet(taught.items, alternate.items);
60
+ if (!signals.moved) {
61
+ // The recipe returned the taught page again. That is a broken page control
62
+ // only if the page itself moves; if the browser sees the same rows too,
63
+ // this is a site that clamps or ignores the parameter for everyone, and
64
+ // from outside that is indistinguishable from a site with one page.
65
+ if (session.browserTaught === null || session.browserAlternate === null) {
66
+ return { status: 'not_tested', signals, shares, reason: `page ${session.alternatePage} returned the taught page again, and no browser reference was available to say whether the page itself moves` };
67
+ }
68
+ if (sameSet(session.browserTaught, session.browserAlternate)) {
69
+ return { status: 'not_tested', signals, shares, reason: `page ${session.alternatePage} returned the taught page again in the browser as well, so this site either clamps the page or has only one` };
70
+ }
71
+ return { status: 'failed', signals, shares, reason: `page ${session.alternatePage} returned the taught page again while the browser moved to a different one, so the recipe's page control does nothing` };
72
+ }
73
+ if (session.browserAlternate === null) {
74
+ return { status: 'not_tested', signals, shares, reason: 'the browser reference observation could not be taken' };
75
+ }
76
+ signals.agreed = sameSet(alternate.items, session.browserAlternate);
77
+ if (!signals.agreed) {
78
+ return { status: 'not_tested', signals, shares, reason: `on page ${session.alternatePage}, an input never taught, the recipe and the browser saw different items` };
79
+ }
80
+ return { status: 'passed', signals, shares, reason: '' };
81
+ }
82
+ export async function runPaginationProbes(probes, baseline, parameter = 'page') {
83
+ const taughtPage = Number(baseline.input[parameter] ?? 1);
84
+ const other = alternatePage(taughtPage);
85
+ const alternate = await probes.http('alternate', String(other), { ...baseline.input, [parameter]: other });
86
+ // The baseline was labelled with whichever input the first check varied, so
87
+ // relabel it here; evidence naming a query where it means a page is evidence
88
+ // nobody can read.
89
+ const atTaughtPage = (o) => ({ ...o, value: String(taughtPage) });
90
+ const observations = [atTaughtPage(baseline.taught), atTaughtPage(baseline.repeat), alternate];
91
+ // Both branches of the verdict need it, and neither an empty neighbour nor a
92
+ // declined probe does, so the launch waits until it can change the answer.
93
+ const worthBrowsing = alternate.items !== null && alternate.items.length > 0
94
+ && baseline.taught.items !== null && baseline.taught.items.length > 0;
95
+ const browserAlternate = worthBrowsing ? await probes.browser({ ...baseline.input, [parameter]: other }) : null;
96
+ return { parameter, taughtPage, alternatePage: other, observations, browserTaught: baseline.browserTaught, browserAlternate };
97
+ }
98
+ export function verifyPaginationHonored(session) {
99
+ const verdict = judgePagination(session);
100
+ return {
101
+ status: verdict.status,
102
+ at: new Date().toISOString(),
103
+ method: METHOD,
104
+ parameter: session.parameter,
105
+ probes: asRecords(session.observations),
106
+ signals: verdict.signals,
107
+ ...(verdict.shares.overlap === undefined ? {} : { shares: verdict.shares }),
108
+ ...(verdict.reason === '' ? {} : { reason: verdict.reason }),
109
+ };
110
+ }
@@ -0,0 +1,98 @@
1
+ import { HttpHtmlStrategy } from '../executor/strategies/http-html.js';
2
+ import { HttpJsonStrategy } from '../executor/strategies/http-json.js';
3
+ import { browserItemsOf } from '../compiler/verify.js';
4
+ import { record } from '../recorder/index.js';
5
+ /** The baseline pair, plus two probes each for the query checks and one for pagination. */
6
+ export const HTTP_PROBE_BUDGET = 5;
7
+ /** Statuses that mean the site declined, which is never a semantic verdict. */
8
+ const BLOCK_STATUSES = new Set([401, 403, 429, 503]);
9
+ /**
10
+ * Drives the strategies directly rather than through `Executor`, which falls
11
+ * back to a browser and re-learns on failure: a probe routed through it would
12
+ * measure whatever managed to answer instead of the recipe being judged.
13
+ * Writes nothing to the registry or the plan.
14
+ */
15
+ export class Probes {
16
+ ctx;
17
+ spent = 0;
18
+ constructor(ctx) {
19
+ this.ctx = ctx;
20
+ }
21
+ get strategy() {
22
+ return this.ctx.recipe.output.type === 'json'
23
+ ? new HttpJsonStrategy(this.ctx.net, this.ctx.sites)
24
+ : new HttpHtmlStrategy(this.ctx.net, this.ctx.sites);
25
+ }
26
+ /** One request, never retried by us; the budget is what stops a storm. */
27
+ async http(role, value, input) {
28
+ if (this.spent >= HTTP_PROBE_BUDGET)
29
+ return { role, value, items: null, unavailable: 'probe budget spent' };
30
+ this.spent += 1;
31
+ try {
32
+ const result = await this.strategy.execute(this.ctx.recipe, { id: `probe-${role}`, site: this.ctx.site, intent: this.ctx.intent, input });
33
+ const status = result.status;
34
+ if (status !== undefined && BLOCK_STATUSES.has(status))
35
+ return { role, value, items: null, unavailable: `site declined with ${status}` };
36
+ if (status !== undefined && status !== this.ctx.recipe.validation.status)
37
+ return { role, value, items: null, unavailable: `unexpected status ${status}` };
38
+ return { role, value, items: result.items };
39
+ }
40
+ catch (error) {
41
+ return { role, value, items: null, unavailable: error instanceof Error ? error.message : String(error) };
42
+ }
43
+ }
44
+ /**
45
+ * A second reading of the same page, not a statement of what is true. The
46
+ * browser can be wrong in its own way, which is why the benchmark's oracle
47
+ * has to stay a separate thing.
48
+ */
49
+ async browser(input) {
50
+ const task = { id: 'probe-reference', site: this.ctx.site, intent: this.ctx.intent, input };
51
+ let trace;
52
+ try {
53
+ trace = await record(this.ctx.browserPlan, task, this.ctx.sites);
54
+ return browserItemsOf(trace, this.ctx.browserPlan);
55
+ }
56
+ catch {
57
+ // A browser that could not run leaves a check untested, never failed.
58
+ return null;
59
+ }
60
+ finally {
61
+ await trace?.dispose?.();
62
+ }
63
+ }
64
+ }
65
+ /**
66
+ * The taught input, asked twice.
67
+ *
68
+ * Twice because a site whose answer changes between two identical requests
69
+ * cannot be judged by comparing sets, in either direction, and every check
70
+ * below rests on that comparison.
71
+ */
72
+ export async function runBaseline(probes, input, parameter, browserTaught) {
73
+ const value = String(input[parameter] ?? '');
74
+ return {
75
+ input,
76
+ taught: await probes.http('taught', value, input),
77
+ repeat: await probes.http('repeat', value, input),
78
+ browserTaught,
79
+ };
80
+ }
81
+ const identity = (item) => item.url !== undefined && item.url !== null && item.url !== ''
82
+ ? `url:${String(item.url)}`
83
+ : JSON.stringify(Object.entries(item).sort(([a], [b]) => a.localeCompare(b)));
84
+ export const setOf = (items) => new Set(items.map(identity));
85
+ export const sameSet = (a, b) => {
86
+ const [x, y] = [setOf(a), setOf(b)];
87
+ return x.size === y.size && [...x].every((k) => y.has(k));
88
+ };
89
+ export const overlapShare = (a, b) => {
90
+ const [x, y] = [setOf(a), setOf(b)];
91
+ if (y.size === 0)
92
+ return 0;
93
+ return [...y].filter((k) => x.has(k)).length / y.size;
94
+ };
95
+ export const asRecords = (observations) => observations.map(({ role, value, items, unavailable }) => ({
96
+ role, value, items: items === null ? null : items.length,
97
+ ...(unavailable === undefined ? {} : { unavailable }),
98
+ }));
@@ -0,0 +1,134 @@
1
+ import { randomUUID } from 'node:crypto';
2
+ import { asRecords, sameSet } from './probes.js';
3
+ export { HTTP_PROBE_BUDGET } from './probes.js';
4
+ /**
5
+ * Whether the learned request actually honours the input we called its query.
6
+ *
7
+ * Deliberately narrower than "the results mean what the query asked for". What
8
+ * these probes can establish is that changing the input changes the answer, that
9
+ * an input nothing could match is turned away, and that a value never taught
10
+ * produces the same answer down the HTTP path as down the browser path. A site
11
+ * that honours its query but answers with the wrong records passes all of that,
12
+ * so the check is named for what it proves.
13
+ *
14
+ * The browser side is a reference observation, not ground truth: it is a second
15
+ * extraction of the same page and can be wrong in its own way. It is what makes
16
+ * the benchmark's oracle, which judges this system from outside, a separate
17
+ * thing that must stay separate.
18
+ */
19
+ export const METHOD = 'active-probe/v1';
20
+ /** A nonce result this much smaller than the taught one counts as turned away. */
21
+ const REJECTION_RATIO = 0.5;
22
+ const WORD = /^[a-z][a-z-]{2,}$/;
23
+ export function tokensOf(text) {
24
+ return text.toLowerCase().split(/[^a-z0-9-]+/i).filter((t) => WORD.test(t));
25
+ }
26
+ /** Every value a plan extracts, which is all the text a check may reason about. */
27
+ export const visibleText = (item) => Object.values(item).map((v) => String(v ?? '')).join(' ').toLowerCase();
28
+ /**
29
+ * A token the taught result carries in only a couple of its rows.
30
+ *
31
+ * Drawn from the site's own answer rather than a word list: a fixed vocabulary
32
+ * fails honest sites it happens not to match and passes dishonest ones it
33
+ * happens to match everywhere. Rare on purpose — a token in most rows would
34
+ * come back with the same set the taught query did, and a working site would be
35
+ * read as one that ignores its input.
36
+ */
37
+ export function chooseContrastToken(items, taught) {
38
+ const excluded = new Set(tokensOf(taught));
39
+ const rows = new Map();
40
+ for (const item of items) {
41
+ const seen = new Set();
42
+ for (const value of Object.values(item))
43
+ for (const token of tokensOf(String(value ?? '')))
44
+ seen.add(token);
45
+ for (const token of seen)
46
+ rows.set(token, (rows.get(token) ?? 0) + 1);
47
+ }
48
+ const rare = [...rows]
49
+ .filter(([token, count]) => count <= 2 && count < items.length && !excluded.has(token))
50
+ // Sorted, so re-running a stored verification picks the same probe again.
51
+ .sort(([tokenA, countA], [tokenB, countB]) => countA - countB || tokenA.localeCompare(tokenB));
52
+ return rare[0]?.[0] ?? null;
53
+ }
54
+ /**
55
+ * Passing needs every signal; failing needs positive evidence that the input is
56
+ * discarded. Everything ambiguous is untested, because a wrong `failed` marks a
57
+ * working integration unverified and a wrong `passed` is the false success this
58
+ * whole model exists to prevent.
59
+ */
60
+ export function judge(observations, browserContrast) {
61
+ const at = (role) => observations.find((o) => o.role === role);
62
+ const signals = {};
63
+ const taught = at('taught');
64
+ const repeat = at('repeat');
65
+ const contrast = at('contrast');
66
+ const nonce = at('nonce');
67
+ for (const [role, probe] of [['taught', taught], ['repeat', repeat], ['nonce', nonce]]) {
68
+ if (!probe)
69
+ return { status: 'not_tested', signals, reason: `no ${role} probe was issued` };
70
+ if (probe.items === null)
71
+ return { status: 'not_tested', signals, reason: `${role} probe unavailable: ${probe.unavailable ?? 'unknown'}` };
72
+ }
73
+ if (taught.items.length === 0) {
74
+ return { status: 'not_tested', signals, reason: 'the taught query returned nothing to compare against' };
75
+ }
76
+ // Volatility first. On a site whose answer changes between two identical
77
+ // requests, every set comparison below is meaningless in both directions.
78
+ signals.stable = sameSet(taught.items, repeat.items);
79
+ if (!signals.stable) {
80
+ return { status: 'not_tested', signals, reason: 'two identical requests returned different results; set comparison cannot judge this site' };
81
+ }
82
+ if (sameSet(nonce.items, taught.items)) {
83
+ signals.nonce_rejected = false;
84
+ return { status: 'failed', signals, reason: `a query nothing could match ("${nonce.value}") returned the taught result unchanged, so the input is discarded` };
85
+ }
86
+ signals.nonce_rejected = nonce.items.length < taught.items.length * REJECTION_RATIO;
87
+ if (!signals.nonce_rejected) {
88
+ return { status: 'not_tested', signals, reason: `a query nothing could match returned ${nonce.items.length} of ${taught.items.length} items, which neither honours nor discards the input` };
89
+ }
90
+ if (!contrast)
91
+ return { status: 'not_tested', signals, reason: 'no contrast probe could be built from the taught result' };
92
+ if (contrast.items === null)
93
+ return { status: 'not_tested', signals, reason: `contrast probe unavailable: ${contrast.unavailable ?? 'unknown'}` };
94
+ signals.responsive = !sameSet(taught.items, contrast.items);
95
+ if (!signals.responsive) {
96
+ return { status: 'not_tested', signals, reason: `a different query ("${contrast.value}") returned the same items; the contrast token may be too common to separate them` };
97
+ }
98
+ if (browserContrast === null) {
99
+ return { status: 'not_tested', signals, reason: 'the browser reference observation could not be taken' };
100
+ }
101
+ signals.agreed = sameSet(contrast.items, browserContrast);
102
+ if (!signals.agreed) {
103
+ return { status: 'not_tested', signals, reason: `on an input never taught ("${contrast.value}") the recipe and the browser saw different items` };
104
+ }
105
+ return { status: 'passed', signals, reason: '' };
106
+ }
107
+ export async function runQueryProbes(probes, baseline, parameter, nonce) {
108
+ const taughtValue = String(baseline.input[parameter] ?? '');
109
+ const at = (value) => ({ ...baseline.input, [parameter]: value });
110
+ const nonceValue = (nonce ?? (() => randomUUID().replace(/-/g, '').slice(0, 12)))();
111
+ const nonceProbe = await probes.http('nonce', nonceValue, at(nonceValue));
112
+ const token = baseline.taught.items === null ? null : chooseContrastToken(baseline.taught.items, taughtValue);
113
+ const contrast = token === null ? undefined : await probes.http('contrast', token, at(token));
114
+ const observations = [baseline.taught, baseline.repeat, nonceProbe, ...(contrast ? [contrast] : [])];
115
+ // Taken only when the cheap probes have already agreed, so a site that fails
116
+ // on HTTP alone never costs a browser launch. Judging with no browser
117
+ // observation yet cannot reach `passed`, which is the point: it is asked only
118
+ // whether the probes got far enough to make the launch worth it.
119
+ const worthBrowsing = contrast !== undefined && judge(observations, null).signals.responsive === true;
120
+ const browserContrast = worthBrowsing ? await probes.browser(at(contrast.value)) : null;
121
+ return { parameter, taughtValue, observations, browserContrast };
122
+ }
123
+ export function verifyQueryHonored(session) {
124
+ const verdict = judge(session.observations, session.browserContrast);
125
+ return {
126
+ status: verdict.status,
127
+ at: new Date().toISOString(),
128
+ method: METHOD,
129
+ parameter: session.parameter,
130
+ probes: asRecords(session.observations),
131
+ signals: verdict.signals,
132
+ ...(verdict.reason === '' ? {} : { reason: verdict.reason }),
133
+ };
134
+ }
@@ -0,0 +1,33 @@
1
+ import { PolitenessLayer, DEFAULT_USER_AGENT } from './net/politeness.js';
2
+ import { RecipeRegistry } from './recipes/registry.js';
3
+ import { Executor } from './executor/index.js';
4
+ import { HttpJsonStrategy } from './executor/strategies/http-json.js';
5
+ import { HttpHtmlStrategy } from './executor/strategies/http-html.js';
6
+ import { BrowserStrategy } from './executor/strategies/browser.js';
7
+ import { WarmBrowserStrategy } from './executor/strategies/warm-browser.js';
8
+ import { SelfHealer } from './healing/index.js';
9
+ import { StaticSiteResolver, WILD_ORIGINS } from './sites.js';
10
+ export function buildEngine(opts) {
11
+ const net = new PolitenessLayer({
12
+ userAgent: DEFAULT_USER_AGENT,
13
+ minIntervalMs: opts.minIntervalMs,
14
+ });
15
+ const sites = new StaticSiteResolver({ ...WILD_ORIGINS, ...(opts.origins ?? {}) });
16
+ const registry = new RecipeRegistry(opts.recipeDir);
17
+ const browser = new BrowserStrategy(sites, opts.plans);
18
+ const warm = new WarmBrowserStrategy(sites, opts.plans);
19
+ const healer = new SelfHealer({ registry, sites, plans: opts.plans });
20
+ const executor = new Executor({
21
+ registry,
22
+ strategies: [new HttpJsonStrategy(net, sites), new HttpHtmlStrategy(net, sites), warm, browser],
23
+ onFallback: opts.heal === false ? undefined : async (event) => {
24
+ const result = await healer.heal(event);
25
+ return {
26
+ unchanged: result.healed && result.changes.length === 0,
27
+ ...(result.refused === null ? {} : { refused: result.refused }),
28
+ ...(result.meta === null ? {} : { meta: result.meta }),
29
+ };
30
+ },
31
+ });
32
+ return { executor, browser, warm, healer, registry, sites, net };
33
+ }
package/package.json ADDED
@@ -0,0 +1,56 @@
1
+ {
2
+ "name": "webrecipe",
3
+ "version": "0.1.0",
4
+ "type": "module",
5
+ "engines": {
6
+ "node": ">=22"
7
+ },
8
+ "bin": {
9
+ "webrecipe": "./dist/src/cli.js"
10
+ },
11
+ "scripts": {
12
+ "build": "tsc -p tsconfig.build.json",
13
+ "test": "vitest run",
14
+ "test:watch": "vitest",
15
+ "cli": "tsx src/cli.ts",
16
+ "prepack": "npm run build"
17
+ },
18
+ "dependencies": {
19
+ "@modelcontextprotocol/sdk": "1.30.1",
20
+ "cheerio": "^1.0.0",
21
+ "commander": "^12.1.0",
22
+ "js-tiktoken": "^1.0.21",
23
+ "playwright": "^1.63.0",
24
+ "yaml": "^2.5.1",
25
+ "zod": "^3.25.76"
26
+ },
27
+ "devDependencies": {
28
+ "@types/node": "^22.7.0",
29
+ "tsx": "^4.19.1",
30
+ "typescript": "^5.6.0",
31
+ "vitest": "^2.1.0"
32
+ },
33
+ "description": "Save how to read a public web page once, then fetch it over plain HTTP.",
34
+ "files": [
35
+ "dist/src",
36
+ "dist/benchmark",
37
+ "dist/fixtures",
38
+ "README.md",
39
+ "LICENSE"
40
+ ],
41
+ "license": "MIT",
42
+ "repository": {
43
+ "type": "git",
44
+ "url": "https://github.com/Pillsoon/webrecipe.git"
45
+ },
46
+ "keywords": [
47
+ "web",
48
+ "scraping",
49
+ "agent",
50
+ "mcp",
51
+ "cli",
52
+ "recipe",
53
+ "playwright",
54
+ "http"
55
+ ]
56
+ }