@tangle-network/agent-knowledge 6.0.0 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/CHANGELOG.md +17 -0
  2. package/README.md +1 -1
  3. package/dist/benchmarks/index.d.ts +2 -53
  4. package/dist/benchmarks/index.js +2 -49
  5. package/dist/benchmarks-CmW6iORW.js +2718 -0
  6. package/dist/benchmarks-CmW6iORW.js.map +1 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +180 -274
  9. package/dist/cli.js.map +1 -1
  10. package/dist/ids-DRqPZ42_.js +15 -0
  11. package/dist/ids-DRqPZ42_.js.map +1 -0
  12. package/dist/index-CGBctbit.d.ts +857 -0
  13. package/dist/index-CGBctbit.d.ts.map +1 -0
  14. package/dist/index-CIW3G4s_.d.ts +680 -0
  15. package/dist/index-CIW3G4s_.d.ts.map +1 -0
  16. package/dist/index.d.ts +1671 -1868
  17. package/dist/index.d.ts.map +1 -0
  18. package/dist/index.js +5836 -6528
  19. package/dist/index.js.map +1 -1
  20. package/dist/inspect-D5iarJc2.js +1864 -0
  21. package/dist/inspect-D5iarJc2.js.map +1 -0
  22. package/dist/memory/index.d.ts +3 -8
  23. package/dist/memory/index.js +3 -81
  24. package/dist/memory-C6KPRhoU.js +4494 -0
  25. package/dist/memory-C6KPRhoU.js.map +1 -0
  26. package/dist/search-CP0QtBJZ.js +113 -0
  27. package/dist/search-CP0QtBJZ.js.map +1 -0
  28. package/dist/sources/index.d.ts +212 -205
  29. package/dist/sources/index.d.ts.map +1 -0
  30. package/dist/sources/index.js +614 -33
  31. package/dist/sources/index.js.map +1 -1
  32. package/dist/types-DcCCzreS.d.ts +175 -0
  33. package/dist/types-DcCCzreS.d.ts.map +1 -0
  34. package/dist/viz/index.d.ts +23 -22
  35. package/dist/viz/index.d.ts.map +1 -0
  36. package/dist/viz/index.js +134 -10
  37. package/dist/viz/index.js.map +1 -1
  38. package/package.json +22 -11
  39. package/dist/benchmarks/index.js.map +0 -1
  40. package/dist/chunk-46YPZHAX.js +0 -5443
  41. package/dist/chunk-46YPZHAX.js.map +0 -1
  42. package/dist/chunk-4PNXQ2NT.js +0 -147
  43. package/dist/chunk-4PNXQ2NT.js.map +0 -1
  44. package/dist/chunk-AKYJG2MR.js +0 -2183
  45. package/dist/chunk-AKYJG2MR.js.map +0 -1
  46. package/dist/chunk-DQ3PDMDP.js +0 -115
  47. package/dist/chunk-DQ3PDMDP.js.map +0 -1
  48. package/dist/chunk-MYFM6LKH.js +0 -551
  49. package/dist/chunk-MYFM6LKH.js.map +0 -1
  50. package/dist/chunk-PVCSESAF.js +0 -3153
  51. package/dist/chunk-PVCSESAF.js.map +0 -1
  52. package/dist/chunk-YMKHCTS2.js +0 -19
  53. package/dist/chunk-YMKHCTS2.js.map +0 -1
  54. package/dist/index-C--N5wQV.d.ts +0 -796
  55. package/dist/memory/index.js.map +0 -1
  56. package/dist/types-6x0OpfW6.d.ts +0 -173
  57. package/dist/types-BY-xLVw-.d.ts +0 -622
@@ -1,34 +1,615 @@
1
- import {
2
- IRS_DIMENSION_HINTS,
3
- MAX_RESPONSE_BYTES,
4
- MIN_REQUEST_GAP_MS,
5
- POLITE_USER_AGENT,
6
- __resetHttpThrottle,
7
- createCornellLiiSource,
8
- createIrsPublicationsSource,
9
- createStateSosSource,
10
- extractLinks,
11
- firstMatch,
12
- htmlToText,
13
- innerHtmlById,
14
- looksLikeBlockPage,
15
- politeFetch
16
- } from "../chunk-MYFM6LKH.js";
17
- import "../chunk-YMKHCTS2.js";
18
- export {
19
- IRS_DIMENSION_HINTS,
20
- MAX_RESPONSE_BYTES,
21
- MIN_REQUEST_GAP_MS,
22
- POLITE_USER_AGENT,
23
- __resetHttpThrottle,
24
- createCornellLiiSource,
25
- createIrsPublicationsSource,
26
- createStateSosSource,
27
- extractLinks,
28
- firstMatch,
29
- htmlToText,
30
- innerHtmlById,
31
- looksLikeBlockPage,
32
- politeFetch
33
- };
1
+ import { t as sha256 } from "../ids-DRqPZ42_.js";
2
+ import { mkdir, readFile, stat, writeFile } from "node:fs/promises";
3
+ import { dirname, join } from "node:path";
4
+ //#region src/sources/html.ts
5
+ /**
6
+ * Minimal HTML helpers used by the shipped sources.
7
+ *
8
+ * Deliberately not a full DOM parser: every authority we ship against
9
+ * (Cornell LII, IRS.gov, state SOS portals) has well-behaved server-rendered
10
+ * HTML where regex-based extraction is correct and cheap. Bringing in cheerio
11
+ * would add a 1.5MB dependency to a package whose purpose is shipping
12
+ * primitives, not parsing arbitrary web pages.
13
+ *
14
+ * If a future source needs real DOM traversal, it should depend on its own
15
+ * parser locally rather than promoting one into the package-wide deps.
16
+ *
17
+ * @stable
18
+ */
19
+ /**
20
+ * Strip HTML tags, collapse whitespace, decode common entities.
21
+ *
22
+ * Preserves paragraph and line breaks (`</p>`, `<br>`, `</li>`, `</div>`,
23
+ * `</h*>`) as `\n` so statute text retains its subsection structure.
24
+ */
25
+ function htmlToText(html) {
26
+ return html.replace(/<script[\s\S]*?<\/script>/gi, "").replace(/<style[\s\S]*?<\/style>/gi, "").replace(/<noscript[\s\S]*?<\/noscript>/gi, "").replace(/<!--([\s\S]*?)-->/g, "").replace(/<\s*br\s*\/?>/gi, "\n").replace(/<\/(p|li|div|tr|h[1-6]|blockquote|section|article)>/gi, "\n").replace(/<[^>]+>/g, "").replace(/&nbsp;/gi, " ").replace(/&amp;/gi, "&").replace(/&lt;/gi, "<").replace(/&gt;/gi, ">").replace(/&quot;/gi, "\"").replace(/&#39;/gi, "'").replace(/&sect;/gi, "§").replace(/&mdash;/gi, "—").replace(/&ndash;/gi, "–").replace(/&#(\d+);/g, (_, code) => String.fromCodePoint(Number(code))).replace(/&#x([0-9a-f]+);/gi, (_, code) => String.fromCodePoint(Number.parseInt(code, 16))).split("\n").map((line) => line.replace(/[\t  ]+/g, " ").trim()).filter((line, idx, all) => !(line === "" && all[idx - 1] === "")).join("\n").trim();
27
+ }
28
+ /** Extract the first match of a regex's first capture group, or undefined. */
29
+ function firstMatch(html, pattern) {
30
+ return pattern.exec(html)?.[1]?.trim();
31
+ }
32
+ /** Extract the inner HTML of the first matching tag with id `id`. */
33
+ function innerHtmlById(html, id) {
34
+ const escaped = id.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
35
+ return new RegExp(`<([a-z][a-z0-9]*)\\b[^>]*\\sid=["']${escaped}["'][^>]*>([\\s\\S]*?)<\\/\\1>`, "i").exec(html)?.[2];
36
+ }
37
+ /**
38
+ * Extract every (href, text) pair matching the URL regex.
39
+ * Returns absolute URLs by resolving against `baseUrl`.
40
+ */
41
+ function extractLinks(html, hrefPattern, baseUrl) {
42
+ const out = [];
43
+ for (const match of html.matchAll(/<a\b[^>]*\shref=["']([^"']+)["'][^>]*>([\s\S]*?)<\/a>/gi)) {
44
+ const href = match[1];
45
+ const inner = match[2];
46
+ if (!href || !inner) continue;
47
+ if (!hrefPattern.test(href)) continue;
48
+ const text = htmlToText(inner);
49
+ if (!text) continue;
50
+ try {
51
+ out.push({
52
+ href: new URL(href, baseUrl).toString(),
53
+ text
54
+ });
55
+ } catch {}
56
+ }
57
+ return out;
58
+ }
59
+ //#endregion
60
+ //#region src/sources/http.ts
61
+ /**
62
+ * Polite HTTP fetcher shared by remote sources.
63
+ *
64
+ * Independent sources share a per-origin throttle because rate-limited sites
65
+ * may return block pages instead of 429 responses. Responses are cached by URL
66
+ * because many publishers omit reliable ETag and Last-Modified headers. Bodies
67
+ * are checked even after a 2xx response because captcha and block pages often
68
+ * use successful status codes.
69
+ */
70
+ /** User-Agent string sent on every outbound request. */
71
+ const POLITE_USER_AGENT = "agent-knowledge (+https://github.com/tangle-network/agent-knowledge)";
72
+ /** Minimum gap between successive requests to the same origin (ms). */
73
+ const MIN_REQUEST_GAP_MS = 1e3;
74
+ /** Maximum response body we will buffer in memory (bytes). */
75
+ const MAX_RESPONSE_BYTES = 8 * 1024 * 1024;
76
+ const hostThrottle = /* @__PURE__ */ new Map();
77
+ /**
78
+ * Fetch one URL with per-host throttling, on-disk cache, and block-page
79
+ * detection. Never throws on network/HTTP failure. It returns a result with
80
+ * `verifiable: false` and `unverifiableReason` set so the caller can decide
81
+ * whether to skip, retry, or surface.
82
+ *
83
+ * Throws ONLY on `AbortError` (caller asked to stop) and on cache-write
84
+ * failures that indicate a misconfigured filesystem.
85
+ */
86
+ async function politeFetch(url, options = {}) {
87
+ const cacheTtl = options.cacheTtlMs ?? 3600 * 1e3;
88
+ const cached = options.cacheDir ? await readCache(options.cacheDir, url, cacheTtl) : void 0;
89
+ if (cached) return cached;
90
+ const host = safeHost(url);
91
+ await throttleHost(host);
92
+ const fetchedAt = (/* @__PURE__ */ new Date()).toISOString();
93
+ let response;
94
+ try {
95
+ response = await fetch(url, {
96
+ signal: options.signal,
97
+ redirect: "follow",
98
+ headers: {
99
+ "User-Agent": POLITE_USER_AGENT,
100
+ Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
101
+ "Accept-Language": "en-US,en;q=0.9",
102
+ ...options.headers ?? {}
103
+ }
104
+ });
105
+ } catch (error) {
106
+ if (error.name === "AbortError") throw error;
107
+ const result = {
108
+ url,
109
+ status: 0,
110
+ body: "",
111
+ sourceUpdatedAt: fetchedAt,
112
+ fetchedAt,
113
+ fromCache: false,
114
+ verifiable: false,
115
+ unverifiableReason: `network error: ${error.message}`
116
+ };
117
+ if (options.cacheDir) await writeCache(options.cacheDir, url, result);
118
+ return result;
119
+ }
120
+ const text = await readBoundedText(response);
121
+ const lastModified = response.headers.get("last-modified");
122
+ const dateHeader = response.headers.get("date");
123
+ const sourceUpdatedAt = parseHttpDate(lastModified) ?? parseHttpDate(dateHeader) ?? fetchedAt;
124
+ const result = {
125
+ url,
126
+ status: response.status,
127
+ body: text,
128
+ sourceUpdatedAt,
129
+ fetchedAt,
130
+ fromCache: false,
131
+ verifiable: true
132
+ };
133
+ if (response.status < 200 || response.status >= 300) {
134
+ result.verifiable = false;
135
+ result.unverifiableReason = `non-2xx status: ${response.status}`;
136
+ } else if (looksLikeBlockPage(text)) {
137
+ result.verifiable = false;
138
+ result.unverifiableReason = "block-page heuristic matched";
139
+ } else if (text.length < 200 && knownLargeAuthority(host)) {
140
+ result.verifiable = false;
141
+ result.unverifiableReason = `body shorter than expected (${text.length} chars)`;
142
+ }
143
+ if (options.cacheDir) await writeCache(options.cacheDir, url, result);
144
+ return result;
145
+ }
146
+ /** Reset the in-process throttle map. Test-only. */
147
+ function __resetHttpThrottle() {
148
+ hostThrottle.clear();
149
+ }
150
+ function safeHost(url) {
151
+ try {
152
+ return new URL(url).host;
153
+ } catch {
154
+ return "unknown";
155
+ }
156
+ }
157
+ async function throttleHost(host) {
158
+ const prev = hostThrottle.get(host) ?? Promise.resolve();
159
+ let release = () => {};
160
+ const next = new Promise((resolve) => {
161
+ release = resolve;
162
+ });
163
+ hostThrottle.set(host, prev.then(() => next));
164
+ await prev;
165
+ setTimeout(release, MIN_REQUEST_GAP_MS);
166
+ }
167
+ async function readBoundedText(response) {
168
+ if (!response.body) return "";
169
+ const reader = response.body.getReader();
170
+ const chunks = [];
171
+ let total = 0;
172
+ while (true) {
173
+ const { done, value } = await reader.read();
174
+ if (done) break;
175
+ if (!value) continue;
176
+ total += value.length;
177
+ if (total > 8388608) {
178
+ await reader.cancel();
179
+ break;
180
+ }
181
+ chunks.push(value);
182
+ }
183
+ const merged = new Uint8Array(Math.min(total, MAX_RESPONSE_BYTES));
184
+ let offset = 0;
185
+ for (const chunk of chunks) {
186
+ const take = Math.min(chunk.length, merged.length - offset);
187
+ if (take <= 0) break;
188
+ merged.set(chunk.subarray(0, take), offset);
189
+ offset += take;
190
+ }
191
+ return new TextDecoder("utf-8", { fatal: false }).decode(merged);
192
+ }
193
+ function parseHttpDate(value) {
194
+ if (!value) return void 0;
195
+ const ms = Date.parse(value);
196
+ return Number.isFinite(ms) ? new Date(ms).toISOString() : void 0;
197
+ }
198
+ /** Cheap heuristic that catches CAPTCHA, WAF block pages, and "Just a moment" interstitials. */
199
+ function looksLikeBlockPage(body) {
200
+ if (!body) return false;
201
+ const lower = body.toLowerCase();
202
+ for (const marker of [
203
+ "verify you are human",
204
+ "please enable javascript and cookies",
205
+ "just a moment",
206
+ "access denied",
207
+ "request unsuccessful",
208
+ "cf-error-details",
209
+ "captcha",
210
+ "incapsula",
211
+ "pardon our interruption"
212
+ ]) if (lower.includes(marker)) return true;
213
+ return false;
214
+ }
215
+ function knownLargeAuthority(host) {
216
+ return host.endsWith("law.cornell.edu") || host.endsWith("irs.gov") || host.endsWith("sos.ca.gov") || host.endsWith("sos.state.tx.us") || host.endsWith("sos.state.us");
217
+ }
218
+ function cachePath(cacheDir, url) {
219
+ const key = sha256(url);
220
+ return join(cacheDir, "http", `${key.slice(0, 2)}`, `${key}.json`);
221
+ }
222
+ async function readCache(cacheDir, url, ttlMs) {
223
+ const path = cachePath(cacheDir, url);
224
+ try {
225
+ const info = await stat(path);
226
+ if (Date.now() - info.mtimeMs > ttlMs) return void 0;
227
+ const raw = await readFile(path, "utf8");
228
+ return {
229
+ ...JSON.parse(raw),
230
+ fromCache: true
231
+ };
232
+ } catch {
233
+ return;
234
+ }
235
+ }
236
+ async function writeCache(cacheDir, url, value) {
237
+ const path = cachePath(cacheDir, url);
238
+ await mkdir(dirname(path), { recursive: true });
239
+ await writeFile(path, JSON.stringify(value), "utf8");
240
+ }
241
+ //#endregion
242
+ //#region src/sources/cornell-lii.ts
243
+ /**
244
+ * Cornell Legal Information Institute (LII) source.
245
+ *
246
+ * Pulls federal US Code sections and Wex encyclopedia entries — the two
247
+ * Cornell LII surfaces an agent typically grounds against. The Wex
248
+ * "non-compete" page is the canonical test case for the Ryan-LLC v. FTC
249
+ * vacatur drift the continuous-ingestion story is designed to catch.
250
+ *
251
+ * @stable
252
+ */
253
+ const BASE_URL$1 = "https://www.law.cornell.edu";
254
+ /**
255
+ * Build a Cornell LII source for the listed selectors.
256
+ *
257
+ * Example: track DTSA + non-compete:
258
+ * ```
259
+ * createCornellLiiSource({
260
+ * selectors: [
261
+ * { kind: 'uscode', path: '18/1836' },
262
+ * { kind: 'wex', path: 'non-compete', dimensionHints: ['jurisdictional_accuracy'] },
263
+ * ],
264
+ * })
265
+ * ```
266
+ */
267
+ function createCornellLiiSource(options) {
268
+ const id = options.id ?? "cornell-lii";
269
+ return {
270
+ id,
271
+ name: "Cornell Legal Information Institute",
272
+ description: "Federal US Code sections (uscode/text/...) and Wex legal encyclopedia entries from law.cornell.edu.",
273
+ async fetch(opts) {
274
+ const limit = opts.limit ?? options.selectors.length;
275
+ const selectors = options.selectors.slice(0, limit);
276
+ const out = [];
277
+ for (const selector of selectors) out.push(await fetchOne(id, selector, opts));
278
+ return out;
279
+ }
280
+ };
281
+ }
282
+ async function fetchOne(sourceId, selector, opts) {
283
+ const path = selector.path.replace(/^\/+/, "");
284
+ const url = selector.kind === "uscode" ? `${BASE_URL$1}/uscode/text/${path}` : `${BASE_URL$1}/wex/${path}`;
285
+ const response = await politeFetch(url, {
286
+ signal: opts.signal,
287
+ cacheDir: opts.cacheDir
288
+ });
289
+ const fragmentId = `${selector.kind}:${selector.path}`;
290
+ const dimensionHints = selector.dimensionHints ?? defaultDimensionHints(selector);
291
+ if (!response.verifiable) return {
292
+ id: fragmentId,
293
+ title: `Cornell LII ${selector.kind} ${selector.path}`,
294
+ body: "",
295
+ bodyHash: sha256(""),
296
+ provenance: {
297
+ url,
298
+ sourceUpdatedAt: response.sourceUpdatedAt,
299
+ fetchedAt: response.fetchedAt,
300
+ jurisdiction: "US-FED",
301
+ verifiable: false,
302
+ unverifiableReason: response.unverifiableReason
303
+ },
304
+ dimensionHints,
305
+ metadata: {
306
+ sourceId,
307
+ status: response.status,
308
+ fromCache: response.fromCache
309
+ }
310
+ };
311
+ const html = response.body;
312
+ const title = extractTitle$1(html, selector);
313
+ const body = extractBody(html, selector);
314
+ const effective = extractEffectiveDate(html) ?? response.sourceUpdatedAt;
315
+ const verifiable = body.length > 50;
316
+ return {
317
+ id: fragmentId,
318
+ title,
319
+ body,
320
+ bodyHash: sha256(body),
321
+ provenance: {
322
+ url,
323
+ sourceUpdatedAt: effective,
324
+ fetchedAt: response.fetchedAt,
325
+ jurisdiction: "US-FED",
326
+ verifiable,
327
+ unverifiableReason: verifiable ? void 0 : "extracted body too short"
328
+ },
329
+ dimensionHints,
330
+ metadata: {
331
+ sourceId,
332
+ status: response.status,
333
+ fromCache: response.fromCache
334
+ }
335
+ };
336
+ }
337
+ function extractTitle$1(html, selector) {
338
+ const h1 = /<h1[^>]*\bid=["']page_title["'][^>]*>([\s\S]*?)<\/h1>/i.exec(html)?.[1];
339
+ if (h1) return htmlToText(h1);
340
+ const t = /<title>([\s\S]*?)<\/title>/i.exec(html)?.[1];
341
+ if (t) return htmlToText(t).split(" | ")[0] ?? `Cornell LII ${selector.path}`;
342
+ return `Cornell LII ${selector.kind} ${selector.path}`;
343
+ }
344
+ function extractBody(html, selector) {
345
+ if (selector.kind === "uscode") {
346
+ const text = /<text>([\s\S]*?)<\/text>/i.exec(html)?.[1];
347
+ if (text) return htmlToText(text);
348
+ const tab = innerHtmlById(html, "tab_default_1");
349
+ if (tab) return htmlToText(tab);
350
+ }
351
+ const mainContent = innerHtmlById(html, "main-content");
352
+ if (mainContent) return htmlToText(mainContent.replace(/<h1[\s\S]*?<\/h1>/i, ""));
353
+ const extracted = innerHtmlById(html, "extracted-content");
354
+ if (extracted) return htmlToText(extracted.replace(/<h1[\s\S]*?<\/h1>/i, ""));
355
+ return htmlToText(html);
356
+ }
357
+ function extractEffectiveDate(html) {
358
+ const amend = /Amendments[\s\S]{0,200}?(\d{4})/i.exec(html)?.[1];
359
+ if (amend) {
360
+ const y = Number.parseInt(amend, 10);
361
+ if (Number.isFinite(y) && y > 1900 && y <= (/* @__PURE__ */ new Date()).getUTCFullYear() + 1) return new Date(Date.UTC(y, 11, 31)).toISOString();
362
+ }
363
+ }
364
+ function defaultDimensionHints(selector) {
365
+ if (selector.kind === "uscode") return ["jurisdictional_accuracy", "citation_hygiene"];
366
+ return ["citation_hygiene"];
367
+ }
368
+ //#endregion
369
+ //#region src/sources/irs-publications.ts
370
+ /**
371
+ * IRS publications source.
372
+ *
373
+ * Two surfaces:
374
+ *
375
+ * 1. The publications index at https://www.irs.gov/publications enumerates
376
+ * every active publication with its revision year — a single fragment
377
+ * with the full table lets change detection notice when a publication
378
+ * year flips (e.g. Pub 15 (2025) → Pub 15 (2026)).
379
+ *
380
+ * 2. Individual publication landing pages at /publications/p<N>[<suffix>]
381
+ * return one fragment per publication with summary text. Callers list
382
+ * the publications they need tracked via `selectors`.
383
+ *
384
+ * Revenue procedures are fetched under their numbered URLs; the IRS does
385
+ * not maintain a stable HTML index of rev-procs, so the caller passes the
386
+ * specific rev-proc paths they care about.
387
+ *
388
+ * @stable
389
+ */
390
+ const BASE_URL = "https://www.irs.gov";
391
+ const INDEX_URL = `${BASE_URL}/publications`;
392
+ /** Default eval dimensions for IRS-sourced fragments. */
393
+ const IRS_DIMENSION_HINTS = [
394
+ "tax_compliance",
395
+ "regulatory_currency",
396
+ "citation_hygiene"
397
+ ];
398
+ function createIrsPublicationsSource(options = {}) {
399
+ const id = options.id ?? "irs-publications";
400
+ const includeIndex = options.includeIndex ?? true;
401
+ return {
402
+ id,
403
+ name: "IRS Publications",
404
+ description: "Internal Revenue Service publications index and individual publication landing pages from irs.gov.",
405
+ async fetch(opts) {
406
+ const out = [];
407
+ const limit = opts.limit ?? Number.POSITIVE_INFINITY;
408
+ if (includeIndex && out.length < limit) out.push(await fetchIndex(id, opts));
409
+ for (const slug of options.publications ?? []) {
410
+ if (out.length >= limit) break;
411
+ out.push(await fetchPublication(id, slug, opts));
412
+ }
413
+ for (const path of options.revenueProcedures ?? []) {
414
+ if (out.length >= limit) break;
415
+ out.push(await fetchRevenueProcedure(id, path, opts));
416
+ }
417
+ return out;
418
+ }
419
+ };
420
+ }
421
+ async function fetchIndex(sourceId, opts) {
422
+ const response = await politeFetch(INDEX_URL, {
423
+ signal: opts.signal,
424
+ cacheDir: opts.cacheDir
425
+ });
426
+ const body = (response.body.match(/<table[\s\S]*?<\/table>/gi) ?? []).map((t) => htmlToText(t)).filter((t) => /Publication\s*\d+/i.test(t)).join("\n\n").slice(0, 2e5);
427
+ const verifiable = response.verifiable && body.length > 200;
428
+ return {
429
+ id: "index",
430
+ title: "IRS Publications Index",
431
+ body,
432
+ bodyHash: sha256(body),
433
+ provenance: {
434
+ url: INDEX_URL,
435
+ sourceUpdatedAt: response.sourceUpdatedAt,
436
+ fetchedAt: response.fetchedAt,
437
+ jurisdiction: "US-FED",
438
+ verifiable,
439
+ unverifiableReason: response.unverifiableReason ?? (verifiable ? void 0 : "no publication rows extracted")
440
+ },
441
+ dimensionHints: IRS_DIMENSION_HINTS,
442
+ metadata: {
443
+ sourceId,
444
+ status: response.status,
445
+ fromCache: response.fromCache,
446
+ kind: "index"
447
+ }
448
+ };
449
+ }
450
+ async function fetchPublication(sourceId, slug, opts) {
451
+ const url = `${BASE_URL}/publications/${slug.replace(/^\/+/, "")}`;
452
+ const response = await politeFetch(url, {
453
+ signal: opts.signal,
454
+ cacheDir: opts.cacheDir
455
+ });
456
+ const title = extractTitle(response.body, `IRS Publication ${slug}`);
457
+ const body = extractMainContent(response.body);
458
+ const verifiable = response.verifiable && body.length > 200;
459
+ return {
460
+ id: `publication:${slug}`,
461
+ title,
462
+ body,
463
+ bodyHash: sha256(body),
464
+ provenance: {
465
+ url,
466
+ sourceUpdatedAt: extractRevisionDate(response.body) ?? response.sourceUpdatedAt,
467
+ fetchedAt: response.fetchedAt,
468
+ jurisdiction: "US-FED",
469
+ verifiable,
470
+ unverifiableReason: response.unverifiableReason ?? (verifiable ? void 0 : "no publication body extracted")
471
+ },
472
+ dimensionHints: IRS_DIMENSION_HINTS,
473
+ metadata: {
474
+ sourceId,
475
+ status: response.status,
476
+ fromCache: response.fromCache,
477
+ kind: "publication",
478
+ slug
479
+ }
480
+ };
481
+ }
482
+ async function fetchRevenueProcedure(sourceId, path, opts) {
483
+ const url = `${BASE_URL}${path.startsWith("/") ? path : `/${path}`}`;
484
+ const response = await politeFetch(url, {
485
+ signal: opts.signal,
486
+ cacheDir: opts.cacheDir
487
+ });
488
+ const body = extractMainContent(response.body);
489
+ const verifiable = response.verifiable && body.length > 200;
490
+ return {
491
+ id: `rev-proc:${path}`,
492
+ title: extractTitle(response.body, `IRS Revenue Procedure ${path}`),
493
+ body,
494
+ bodyHash: sha256(body),
495
+ provenance: {
496
+ url,
497
+ sourceUpdatedAt: response.sourceUpdatedAt,
498
+ fetchedAt: response.fetchedAt,
499
+ jurisdiction: "US-FED",
500
+ verifiable,
501
+ unverifiableReason: response.unverifiableReason ?? (verifiable ? void 0 : "no revenue-procedure body extracted")
502
+ },
503
+ dimensionHints: [...IRS_DIMENSION_HINTS, "procedural_currency"],
504
+ metadata: {
505
+ sourceId,
506
+ status: response.status,
507
+ fromCache: response.fromCache,
508
+ kind: "rev-proc",
509
+ path
510
+ }
511
+ };
512
+ }
513
+ function extractTitle(html, fallback) {
514
+ const og = /<meta\s+property=["']og:title["']\s+content=["']([^"']+)["']/i.exec(html)?.[1];
515
+ if (og) return decodeHtml(og);
516
+ const title = /<title>([\s\S]*?)<\/title>/i.exec(html)?.[1];
517
+ if (title) return htmlToText(title).split(" | ")[0] ?? fallback;
518
+ return fallback;
519
+ }
520
+ function extractMainContent(html) {
521
+ const main = /<main\b[\s\S]*?<\/main>/i.exec(html)?.[0];
522
+ if (main) return htmlToText(main.replace(/<nav[\s\S]*?<\/nav>/gi, "").replace(/<header[\s\S]*?<\/header>/gi, "").replace(/<footer[\s\S]*?<\/footer>/gi, "")).slice(0, 2e5);
523
+ const body = /<body\b[\s\S]*?<\/body>/i.exec(html)?.[0];
524
+ return body ? htmlToText(body).slice(0, 2e5) : htmlToText(html).slice(0, 2e5);
525
+ }
526
+ function extractRevisionDate(html) {
527
+ const m = /Publication\s+\S+\s*\((\d{4})\)/i.exec(html);
528
+ if (m?.[1]) {
529
+ const year = Number.parseInt(m[1], 10);
530
+ if (Number.isFinite(year) && year >= 2e3 && year <= (/* @__PURE__ */ new Date()).getUTCFullYear() + 1) return new Date(Date.UTC(year, 0, 1)).toISOString();
531
+ }
532
+ }
533
+ function decodeHtml(value) {
534
+ return htmlToText(value);
535
+ }
536
+ //#endregion
537
+ //#region src/sources/state-sos.ts
538
+ function createStateSosSource(config) {
539
+ const id = config.id ?? `state-sos:${config.state.toLowerCase()}`;
540
+ return {
541
+ id,
542
+ name: config.name ?? `${config.state} Secretary of State`,
543
+ description: `${config.state} Secretary of State filings and formation guidance pages.`,
544
+ async fetch(opts) {
545
+ const limit = opts.limit ?? config.entities.length;
546
+ const entities = config.entities.slice(0, limit);
547
+ const out = [];
548
+ for (const entity of entities) out.push(await fetchEntity(id, config, entity, opts));
549
+ return out;
550
+ }
551
+ };
552
+ }
553
+ async function fetchEntity(sourceId, config, entity, opts) {
554
+ const url = joinUrl(config.baseUrl, entity.path);
555
+ const response = await politeFetch(url, {
556
+ signal: opts.signal,
557
+ cacheDir: opts.cacheDir
558
+ });
559
+ const body = response.verifiable ? extractBySelector(response.body, entity.selector) : "";
560
+ const verifiable = response.verifiable && body.length > 100;
561
+ return {
562
+ id: entity.id,
563
+ title: entity.title,
564
+ body,
565
+ bodyHash: sha256(body),
566
+ provenance: {
567
+ url,
568
+ sourceUpdatedAt: response.sourceUpdatedAt,
569
+ fetchedAt: response.fetchedAt,
570
+ jurisdiction: `US-${config.state.toUpperCase()}`,
571
+ verifiable,
572
+ unverifiableReason: response.unverifiableReason ?? (verifiable ? void 0 : "extracted body too short")
573
+ },
574
+ dimensionHints: entity.dimensionHints ?? [
575
+ "jurisdictional_accuracy",
576
+ "corporate_formation",
577
+ "citation_hygiene"
578
+ ],
579
+ metadata: {
580
+ sourceId,
581
+ status: response.status,
582
+ fromCache: response.fromCache,
583
+ state: config.state
584
+ }
585
+ };
586
+ }
587
+ function extractBySelector(html, selector) {
588
+ if (selector.kind === "whole") {
589
+ const main = /<main\b[\s\S]*?<\/main>/i.exec(html)?.[0];
590
+ return htmlToText(main ?? html).slice(0, 2e5);
591
+ }
592
+ if (selector.kind === "regex") {
593
+ const m = selector.value.exec(html)?.[0];
594
+ return m ? htmlToText(m).slice(0, 2e5) : "";
595
+ }
596
+ if (selector.kind === "id") {
597
+ const escaped = selector.value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
598
+ const inner = new RegExp(`<([a-z][a-z0-9]*)\\b[^>]*\\sid=["']${escaped}["'][^>]*>([\\s\\S]*?)<\\/\\1>`, "i").exec(html)?.[2];
599
+ return inner ? htmlToText(inner).slice(0, 2e5) : "";
600
+ }
601
+ const escaped = selector.value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
602
+ const inner = new RegExp(`<([a-z][a-z0-9]*)\\b[^>]*\\sclass=["'][^"']*\\b${escaped}\\b[^"']*["'][^>]*>([\\s\\S]*?)<\\/\\1>`, "i").exec(html)?.[2];
603
+ return inner ? htmlToText(inner).slice(0, 2e5) : "";
604
+ }
605
+ function joinUrl(base, path) {
606
+ try {
607
+ return new URL(path, base.endsWith("/") ? base : `${base}/`).toString();
608
+ } catch {
609
+ return `${base.replace(/\/+$/, "")}/${path.replace(/^\/+/, "")}`;
610
+ }
611
+ }
612
+ //#endregion
613
+ export { IRS_DIMENSION_HINTS, MAX_RESPONSE_BYTES, MIN_REQUEST_GAP_MS, POLITE_USER_AGENT, __resetHttpThrottle, createCornellLiiSource, createIrsPublicationsSource, createStateSosSource, extractLinks, firstMatch, htmlToText, innerHtmlById, looksLikeBlockPage, politeFetch };
614
+
34
615
  //# sourceMappingURL=index.js.map