webrecipe 0.1.2 → 0.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,13 @@
1
1
  import { DATASET, PAGE_SIZE, search } from '../../fixtures/data.js';
2
- const identity = (item) => String(item.url ?? '');
2
+ const identity = (item) => {
3
+ const url = String(item.url ?? '');
4
+ try {
5
+ return new URL(url).pathname;
6
+ }
7
+ catch {
8
+ return url;
9
+ }
10
+ };
3
11
  const KNOWN = new Set(DATASET.map((r) => `/item/${r.id}`));
4
12
  const PAGES = Math.ceil(DATASET.length / PAGE_SIZE);
5
13
  /** Every match, not just page one: a different ranking is not a wrong answer. */
@@ -101,7 +101,7 @@ function compileFrom(trace, plan, source) {
101
101
  const untemplated = required.find((name) => !templated.has(name));
102
102
  if (untemplated !== undefined)
103
103
  return { refused: `required input "${untemplated}" not templated` };
104
- const items = extractHtmlItems(draft, body);
104
+ const items = extractHtmlItems(draft, body, source.url);
105
105
  if (!verifyAgainstBrowser(trace, plan, items).equivalent) {
106
106
  return { refused: equivalenceRefusal(items.length, browserItemsOf(trace, plan).length) };
107
107
  }
@@ -10,7 +10,7 @@ import { diffGolden } from '../benchmark/golden.js';
10
10
  */
11
11
  /** What the browser extracted from this visit, using the plan's own selectors. */
12
12
  export function browserItemsOf(trace, plan) {
13
- return extractBySelector(trace.finalHtml, plan.itemSelector, plan.fields);
13
+ return extractBySelector(trace.finalHtml, plan.itemSelector, plan.fields, trace.finalUrl);
14
14
  }
15
15
  export function verifyAgainstBrowser(trace, plan, recipeItems) {
16
16
  const browserItems = browserItemsOf(trace, plan);
@@ -29,10 +29,10 @@ export function extractJsonItems(recipe, payload) {
29
29
  : renderRowTemplate(spec, row).trim(),
30
30
  ])));
31
31
  }
32
- export function extractHtmlItems(recipe, html) {
32
+ export function extractHtmlItems(recipe, html, pageUrl) {
33
33
  if (recipe.output.type !== 'html')
34
34
  throw new Error('extractHtmlItems requires an html recipe');
35
- return extractBySelector(html, recipe.output.items.selector, recipe.output.items.fields);
35
+ return extractBySelector(html, recipe.output.items.selector, recipe.output.items.fields, pageUrl);
36
36
  }
37
37
  /**
38
38
  * The structural signature of an HTML response: which of the recipe's declared
@@ -90,16 +90,20 @@ export function parseHtmlFragment(html) {
90
90
  const wrapper = FRAGMENT_WRAPPERS.find(([pattern]) => pattern.test(head));
91
91
  return cheerio.load(wrapper ? wrapper[1](head) : html);
92
92
  }
93
+ /** Attributes holding a URL, reported absolute the way the browser's `a.href` reports them. */
94
+ const URL_ATTRIBUTES = new Set(['href', 'src']);
93
95
  /** One interpretation of a field spec, shared by cheerio and the browser. */
94
96
  export function planFields(fields) {
95
97
  return Object.entries(fields).map(([name, spec]) => {
96
98
  if (spec === '')
97
99
  return { name, mode: 'own-text' };
98
- if (spec.startsWith('@'))
99
- return { name, mode: 'own-attr', attribute: spec.slice(1) };
100
+ if (spec.startsWith('@')) {
101
+ const attribute = spec.slice(1);
102
+ return { name, mode: 'own-attr', attribute, url: URL_ATTRIBUTES.has(attribute) };
103
+ }
100
104
  const suffix = attributeSuffix(spec);
101
105
  return suffix
102
- ? { name, mode: 'find-attr', selector: suffix.selector, attribute: suffix.attribute }
106
+ ? { name, mode: 'find-attr', ...suffix, url: URL_ATTRIBUTES.has(suffix.attribute) }
103
107
  : { name, mode: 'find-text', selector: spec };
104
108
  });
105
109
  }
@@ -136,15 +140,39 @@ function attributeSuffix(spec) {
136
140
  return null;
137
141
  return { selector, attribute: match[1] };
138
142
  }
143
+ /** Resolves a URL attribute as the browser does; an empty or unparsable value is left alone. */
144
+ function absolute(value, base) {
145
+ if (value.trim() === '')
146
+ return value;
147
+ try {
148
+ return new URL(value, base).href;
149
+ }
150
+ catch {
151
+ return value;
152
+ }
153
+ }
154
+ /** Makes href and src fields absolute against the document's base URL, in place. */
155
+ export function resolveUrlFields(items, plans, base) {
156
+ const urlFields = plans.filter((f) => (f.mode === 'own-attr' || f.mode === 'find-attr') && f.url).map((f) => f.name);
157
+ for (const item of items) {
158
+ for (const name of urlFields) {
159
+ const value = item[name];
160
+ if (typeof value === 'string')
161
+ item[name] = absolute(value, base);
162
+ }
163
+ }
164
+ return items;
165
+ }
139
166
  /**
140
167
  * Extracts items from HTML with a bare selector and field map, independent of a
141
168
  * recipe. Used to reconstruct what the browser saw so a candidate recipe can be
142
- * checked against it.
169
+ * checked against it. Given the page's URL, href and src come back absolute;
170
+ * without it they come back as written.
143
171
  */
144
- export function extractBySelector(html, selector, fields) {
172
+ export function extractBySelector(html, selector, fields, pageUrl) {
145
173
  const $ = parseHtmlFragment(html);
146
174
  const plans = planFields(fields);
147
- return $(selector).toArray().map((element) => {
175
+ const items = $(selector).toArray().map((element) => {
148
176
  const item = $(element);
149
177
  const row = {};
150
178
  // Field candidates ask for `a`, `a@href` and `a@id` of the same item; one find serves all three.
@@ -176,4 +204,8 @@ export function extractBySelector(html, selector, fields) {
176
204
  }
177
205
  return row;
178
206
  });
207
+ if (pageUrl === undefined)
208
+ return items;
209
+ const baseHref = $('base[href]').first().attr('href');
210
+ return resolveUrlFields(items, plans, baseHref === undefined ? pageUrl : absolute(baseHref, pageUrl));
179
211
  }
@@ -1,7 +1,7 @@
1
1
  import { measureResult } from '../../measurement.js';
2
2
  import { openSession } from '../../browser/session.js';
3
3
  import { navigateAndSettle, guardPage } from '../../browser/navigate.js';
4
- import { planFields } from '../extract.js';
4
+ import { planFields, resolveUrlFields } from '../extract.js';
5
5
  import { countTokens } from '../tokens.js';
6
6
  import { emptyMeta } from '../../types.js';
7
7
  export const BROWSER_PLANS = {};
@@ -43,6 +43,8 @@ export class BrowserStrategy {
43
43
  default: return [f.name, el.querySelector(f.selector)?.textContent?.trim() ?? null];
44
44
  }
45
45
  }))), planFields(plan.fields)));
46
+ // Resolved in node rather than in the page, by the same rule the http path uses.
47
+ resolveUrlFields(items, planFields(plan.fields), await session.page.evaluate(() => document.baseURI));
46
48
  meta.pageNavigations = session.cost.pageNavigations;
47
49
  meta.networkRequests = session.cost.networkRequests;
48
50
  meta.bytesDownloaded = session.cost.bytesDownloaded;
@@ -35,7 +35,7 @@ export class HttpHtmlStrategy {
35
35
  if (res.status !== recipe.validation.status) {
36
36
  return { items: [], meta, status: res.status, payload: undefined };
37
37
  }
38
- const items = extractHtmlItems(recipe, res.body);
38
+ const items = extractHtmlItems(recipe, res.body, res.url);
39
39
  meta.llmTokens = countTokens(formatItems(items, 'tsv'));
40
40
  return { items, meta, payload: htmlSignature(recipe, items), status: res.status };
41
41
  }
@@ -1,7 +1,7 @@
1
1
  import { measureResult } from '../../measurement.js';
2
2
  import { BrowserPool } from '../../browser/pool.js';
3
3
  import { navigateAndSettle, guardPage } from '../../browser/navigate.js';
4
- import { planFields } from '../extract.js';
4
+ import { planFields, resolveUrlFields } from '../extract.js';
5
5
  import { countTokens } from '../tokens.js';
6
6
  import { BROWSER_PLANS } from './browser.js';
7
7
  import { emptyMeta } from '../../types.js';
@@ -38,6 +38,8 @@ export class WarmBrowserStrategy {
38
38
  default: return [f.name, el.querySelector(f.selector)?.textContent?.trim() ?? null];
39
39
  }
40
40
  }))), planFields(plan.fields)));
41
+ // Resolved in node rather than in the page, by the same rule the http path uses.
42
+ resolveUrlFields(items, planFields(plan.fields), await warm.page.evaluate(() => document.baseURI));
41
43
  // Zero only when the pool already had a browser. The first call starts
42
44
  // Chromium like any cold run does, and reporting none said otherwise.
43
45
  meta.browserLaunches = warm.launched ? 1 : 0;
@@ -101,7 +101,7 @@ export class PolitenessLayer {
101
101
  continue;
102
102
  }
103
103
  return {
104
- status: res.status, headers: Object.fromEntries(res.headers.entries()),
104
+ status: res.status, headers: Object.fromEntries(res.headers.entries()), url: current,
105
105
  body: buffer.toString('utf8'), bytesDownloaded: buffer.byteLength, waitedMs: 0,
106
106
  };
107
107
  }
@@ -142,6 +142,7 @@ export async function record(plan, task, sites, allowed) {
142
142
  actions,
143
143
  requests: pending,
144
144
  finalHtml,
145
+ finalUrl: session.page.url(),
145
146
  ...(session.page.viewportSize() === null ? {} : { viewport: session.page.viewportSize() }),
146
147
  cost: session.cost,
147
148
  dispose: () => bodies.dispose(),
package/package.json CHANGED
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "webrecipe",
3
- "version": "0.1.2",
3
+ "version": "0.1.3",
4
4
  "type": "module",
5
5
  "engines": {
6
6
  "node": ">=22"
7
7
  },
8
8
  "bin": {
9
- "webrecipe": "./dist/src/cli.js"
9
+ "webrecipe": "dist/src/cli.js"
10
10
  },
11
11
  "scripts": {
12
12
  "build": "tsc -p tsconfig.build.json",
@@ -41,7 +41,7 @@
41
41
  "license": "MIT",
42
42
  "repository": {
43
43
  "type": "git",
44
- "url": "https://github.com/Pillsoon/webrecipe.git"
44
+ "url": "git+https://github.com/Pillsoon/webrecipe.git"
45
45
  },
46
46
  "keywords": [
47
47
  "web",