@tangle-network/agent-knowledge 5.0.4 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/README.md +2 -2
- package/dist/benchmarks/index.d.ts +2 -53
- package/dist/benchmarks/index.js +2 -49
- package/dist/benchmarks-CmW6iORW.js +2718 -0
- package/dist/benchmarks-CmW6iORW.js.map +1 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +180 -274
- package/dist/cli.js.map +1 -1
- package/dist/ids-DRqPZ42_.js +15 -0
- package/dist/ids-DRqPZ42_.js.map +1 -0
- package/dist/index-CGBctbit.d.ts +857 -0
- package/dist/index-CGBctbit.d.ts.map +1 -0
- package/dist/index-CIW3G4s_.d.ts +680 -0
- package/dist/index-CIW3G4s_.d.ts.map +1 -0
- package/dist/index.d.ts +1671 -1876
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +5836 -6530
- package/dist/index.js.map +1 -1
- package/dist/inspect-D5iarJc2.js +1864 -0
- package/dist/inspect-D5iarJc2.js.map +1 -0
- package/dist/memory/index.d.ts +3 -8
- package/dist/memory/index.js +3 -81
- package/dist/memory-C6KPRhoU.js +4494 -0
- package/dist/memory-C6KPRhoU.js.map +1 -0
- package/dist/search-CP0QtBJZ.js +113 -0
- package/dist/search-CP0QtBJZ.js.map +1 -0
- package/dist/sources/index.d.ts +212 -205
- package/dist/sources/index.d.ts.map +1 -0
- package/dist/sources/index.js +614 -33
- package/dist/sources/index.js.map +1 -1
- package/dist/types-DcCCzreS.d.ts +175 -0
- package/dist/types-DcCCzreS.d.ts.map +1 -0
- package/dist/viz/index.d.ts +23 -22
- package/dist/viz/index.d.ts.map +1 -0
- package/dist/viz/index.js +134 -10
- package/dist/viz/index.js.map +1 -1
- package/docs/results/investment-thesis.md +1 -1
- package/docs/results/research-driving.md +1 -1
- package/docs/{two-agent-research-ab.md → verified-research-ab.md} +2 -2
- package/package.json +22 -11
- package/dist/benchmarks/index.js.map +0 -1
- package/dist/chunk-4PNXQ2NT.js +0 -147
- package/dist/chunk-4PNXQ2NT.js.map +0 -1
- package/dist/chunk-AKYJG2MR.js +0 -2183
- package/dist/chunk-AKYJG2MR.js.map +0 -1
- package/dist/chunk-DQ3PDMDP.js +0 -115
- package/dist/chunk-DQ3PDMDP.js.map +0 -1
- package/dist/chunk-EYIA5PLQ.js +0 -3153
- package/dist/chunk-EYIA5PLQ.js.map +0 -1
- package/dist/chunk-LMR53POQ.js +0 -5437
- package/dist/chunk-LMR53POQ.js.map +0 -1
- package/dist/chunk-MYFM6LKH.js +0 -551
- package/dist/chunk-MYFM6LKH.js.map +0 -1
- package/dist/chunk-YMKHCTS2.js +0 -19
- package/dist/chunk-YMKHCTS2.js.map +0 -1
- package/dist/index-Cf7txrYP.d.ts +0 -790
- package/dist/memory/index.js.map +0 -1
- package/dist/types-6x0OpfW6.d.ts +0 -173
- package/dist/types-BY-xLVw-.d.ts +0 -622
package/dist/sources/index.js
CHANGED
|
@@ -1,34 +1,615 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
1
|
+
import { t as sha256 } from "../ids-DRqPZ42_.js";
|
|
2
|
+
import { mkdir, readFile, stat, writeFile } from "node:fs/promises";
|
|
3
|
+
import { dirname, join } from "node:path";
|
|
4
|
+
//#region src/sources/html.ts
|
|
5
|
+
/**
|
|
6
|
+
* Minimal HTML helpers used by the shipped sources.
|
|
7
|
+
*
|
|
8
|
+
* Deliberately not a full DOM parser: every authority we ship against
|
|
9
|
+
* (Cornell LII, IRS.gov, state SOS portals) has well-behaved server-rendered
|
|
10
|
+
* HTML where regex-based extraction is correct and cheap. Bringing in cheerio
|
|
11
|
+
* would add a 1.5MB dependency to a package whose purpose is shipping
|
|
12
|
+
* primitives, not parsing arbitrary web pages.
|
|
13
|
+
*
|
|
14
|
+
* If a future source needs real DOM traversal, it should depend on its own
|
|
15
|
+
* parser locally rather than promoting one into the package-wide deps.
|
|
16
|
+
*
|
|
17
|
+
* @stable
|
|
18
|
+
*/
|
|
19
|
+
/**
|
|
20
|
+
* Strip HTML tags, collapse whitespace, decode common entities.
|
|
21
|
+
*
|
|
22
|
+
* Preserves paragraph and line breaks (`</p>`, `<br>`, `</li>`, `</div>`,
|
|
23
|
+
* `</h*>`) as `\n` so statute text retains its subsection structure.
|
|
24
|
+
*/
|
|
25
|
+
function htmlToText(html) {
|
|
26
|
+
return html.replace(/<script[\s\S]*?<\/script>/gi, "").replace(/<style[\s\S]*?<\/style>/gi, "").replace(/<noscript[\s\S]*?<\/noscript>/gi, "").replace(/<!--([\s\S]*?)-->/g, "").replace(/<\s*br\s*\/?>/gi, "\n").replace(/<\/(p|li|div|tr|h[1-6]|blockquote|section|article)>/gi, "\n").replace(/<[^>]+>/g, "").replace(/ /gi, " ").replace(/&/gi, "&").replace(/</gi, "<").replace(/>/gi, ">").replace(/"/gi, "\"").replace(/'/gi, "'").replace(/§/gi, "§").replace(/—/gi, "—").replace(/–/gi, "–").replace(/&#(\d+);/g, (_, code) => String.fromCodePoint(Number(code))).replace(/&#x([0-9a-f]+);/gi, (_, code) => String.fromCodePoint(Number.parseInt(code, 16))).split("\n").map((line) => line.replace(/[\t ]+/g, " ").trim()).filter((line, idx, all) => !(line === "" && all[idx - 1] === "")).join("\n").trim();
|
|
27
|
+
}
|
|
28
|
+
/** Extract the first match of a regex's first capture group, or undefined. */
|
|
29
|
+
function firstMatch(html, pattern) {
|
|
30
|
+
return pattern.exec(html)?.[1]?.trim();
|
|
31
|
+
}
|
|
32
|
+
/** Extract the inner HTML of the first matching tag with id `id`. */
|
|
33
|
+
function innerHtmlById(html, id) {
|
|
34
|
+
const escaped = id.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
35
|
+
return new RegExp(`<([a-z][a-z0-9]*)\\b[^>]*\\sid=["']${escaped}["'][^>]*>([\\s\\S]*?)<\\/\\1>`, "i").exec(html)?.[2];
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Extract every (href, text) pair matching the URL regex.
|
|
39
|
+
* Returns absolute URLs by resolving against `baseUrl`.
|
|
40
|
+
*/
|
|
41
|
+
function extractLinks(html, hrefPattern, baseUrl) {
|
|
42
|
+
const out = [];
|
|
43
|
+
for (const match of html.matchAll(/<a\b[^>]*\shref=["']([^"']+)["'][^>]*>([\s\S]*?)<\/a>/gi)) {
|
|
44
|
+
const href = match[1];
|
|
45
|
+
const inner = match[2];
|
|
46
|
+
if (!href || !inner) continue;
|
|
47
|
+
if (!hrefPattern.test(href)) continue;
|
|
48
|
+
const text = htmlToText(inner);
|
|
49
|
+
if (!text) continue;
|
|
50
|
+
try {
|
|
51
|
+
out.push({
|
|
52
|
+
href: new URL(href, baseUrl).toString(),
|
|
53
|
+
text
|
|
54
|
+
});
|
|
55
|
+
} catch {}
|
|
56
|
+
}
|
|
57
|
+
return out;
|
|
58
|
+
}
|
|
59
|
+
//#endregion
|
|
60
|
+
//#region src/sources/http.ts
|
|
61
|
+
/**
|
|
62
|
+
* Polite HTTP fetcher shared by remote sources.
|
|
63
|
+
*
|
|
64
|
+
* Independent sources share a per-origin throttle because rate-limited sites
|
|
65
|
+
* may return block pages instead of 429 responses. Responses are cached by URL
|
|
66
|
+
* because many publishers omit reliable ETag and Last-Modified headers. Bodies
|
|
67
|
+
* are checked even after a 2xx response because captcha and block pages often
|
|
68
|
+
* use successful status codes.
|
|
69
|
+
*/
|
|
70
|
+
/** User-Agent string sent on every outbound request. */
|
|
71
|
+
const POLITE_USER_AGENT = "agent-knowledge (+https://github.com/tangle-network/agent-knowledge)";
|
|
72
|
+
/** Minimum gap between successive requests to the same origin (ms). */
|
|
73
|
+
const MIN_REQUEST_GAP_MS = 1e3;
|
|
74
|
+
/** Maximum response body we will buffer in memory (bytes). */
|
|
75
|
+
const MAX_RESPONSE_BYTES = 8 * 1024 * 1024;
|
|
76
|
+
const hostThrottle = /* @__PURE__ */ new Map();
|
|
77
|
+
/**
|
|
78
|
+
* Fetch one URL with per-host throttling, on-disk cache, and block-page
|
|
79
|
+
* detection. Never throws on network/HTTP failure. It returns a result with
|
|
80
|
+
* `verifiable: false` and `unverifiableReason` set so the caller can decide
|
|
81
|
+
* whether to skip, retry, or surface.
|
|
82
|
+
*
|
|
83
|
+
* Throws ONLY on `AbortError` (caller asked to stop) and on cache-write
|
|
84
|
+
* failures that indicate a misconfigured filesystem.
|
|
85
|
+
*/
|
|
86
|
+
async function politeFetch(url, options = {}) {
|
|
87
|
+
const cacheTtl = options.cacheTtlMs ?? 3600 * 1e3;
|
|
88
|
+
const cached = options.cacheDir ? await readCache(options.cacheDir, url, cacheTtl) : void 0;
|
|
89
|
+
if (cached) return cached;
|
|
90
|
+
const host = safeHost(url);
|
|
91
|
+
await throttleHost(host);
|
|
92
|
+
const fetchedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
93
|
+
let response;
|
|
94
|
+
try {
|
|
95
|
+
response = await fetch(url, {
|
|
96
|
+
signal: options.signal,
|
|
97
|
+
redirect: "follow",
|
|
98
|
+
headers: {
|
|
99
|
+
"User-Agent": POLITE_USER_AGENT,
|
|
100
|
+
Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|
101
|
+
"Accept-Language": "en-US,en;q=0.9",
|
|
102
|
+
...options.headers ?? {}
|
|
103
|
+
}
|
|
104
|
+
});
|
|
105
|
+
} catch (error) {
|
|
106
|
+
if (error.name === "AbortError") throw error;
|
|
107
|
+
const result = {
|
|
108
|
+
url,
|
|
109
|
+
status: 0,
|
|
110
|
+
body: "",
|
|
111
|
+
sourceUpdatedAt: fetchedAt,
|
|
112
|
+
fetchedAt,
|
|
113
|
+
fromCache: false,
|
|
114
|
+
verifiable: false,
|
|
115
|
+
unverifiableReason: `network error: ${error.message}`
|
|
116
|
+
};
|
|
117
|
+
if (options.cacheDir) await writeCache(options.cacheDir, url, result);
|
|
118
|
+
return result;
|
|
119
|
+
}
|
|
120
|
+
const text = await readBoundedText(response);
|
|
121
|
+
const lastModified = response.headers.get("last-modified");
|
|
122
|
+
const dateHeader = response.headers.get("date");
|
|
123
|
+
const sourceUpdatedAt = parseHttpDate(lastModified) ?? parseHttpDate(dateHeader) ?? fetchedAt;
|
|
124
|
+
const result = {
|
|
125
|
+
url,
|
|
126
|
+
status: response.status,
|
|
127
|
+
body: text,
|
|
128
|
+
sourceUpdatedAt,
|
|
129
|
+
fetchedAt,
|
|
130
|
+
fromCache: false,
|
|
131
|
+
verifiable: true
|
|
132
|
+
};
|
|
133
|
+
if (response.status < 200 || response.status >= 300) {
|
|
134
|
+
result.verifiable = false;
|
|
135
|
+
result.unverifiableReason = `non-2xx status: ${response.status}`;
|
|
136
|
+
} else if (looksLikeBlockPage(text)) {
|
|
137
|
+
result.verifiable = false;
|
|
138
|
+
result.unverifiableReason = "block-page heuristic matched";
|
|
139
|
+
} else if (text.length < 200 && knownLargeAuthority(host)) {
|
|
140
|
+
result.verifiable = false;
|
|
141
|
+
result.unverifiableReason = `body shorter than expected (${text.length} chars)`;
|
|
142
|
+
}
|
|
143
|
+
if (options.cacheDir) await writeCache(options.cacheDir, url, result);
|
|
144
|
+
return result;
|
|
145
|
+
}
|
|
146
|
+
/** Reset the in-process throttle map. Test-only. */
|
|
147
|
+
function __resetHttpThrottle() {
|
|
148
|
+
hostThrottle.clear();
|
|
149
|
+
}
|
|
150
|
+
function safeHost(url) {
|
|
151
|
+
try {
|
|
152
|
+
return new URL(url).host;
|
|
153
|
+
} catch {
|
|
154
|
+
return "unknown";
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
async function throttleHost(host) {
|
|
158
|
+
const prev = hostThrottle.get(host) ?? Promise.resolve();
|
|
159
|
+
let release = () => {};
|
|
160
|
+
const next = new Promise((resolve) => {
|
|
161
|
+
release = resolve;
|
|
162
|
+
});
|
|
163
|
+
hostThrottle.set(host, prev.then(() => next));
|
|
164
|
+
await prev;
|
|
165
|
+
setTimeout(release, MIN_REQUEST_GAP_MS);
|
|
166
|
+
}
|
|
167
|
+
async function readBoundedText(response) {
|
|
168
|
+
if (!response.body) return "";
|
|
169
|
+
const reader = response.body.getReader();
|
|
170
|
+
const chunks = [];
|
|
171
|
+
let total = 0;
|
|
172
|
+
while (true) {
|
|
173
|
+
const { done, value } = await reader.read();
|
|
174
|
+
if (done) break;
|
|
175
|
+
if (!value) continue;
|
|
176
|
+
total += value.length;
|
|
177
|
+
if (total > 8388608) {
|
|
178
|
+
await reader.cancel();
|
|
179
|
+
break;
|
|
180
|
+
}
|
|
181
|
+
chunks.push(value);
|
|
182
|
+
}
|
|
183
|
+
const merged = new Uint8Array(Math.min(total, MAX_RESPONSE_BYTES));
|
|
184
|
+
let offset = 0;
|
|
185
|
+
for (const chunk of chunks) {
|
|
186
|
+
const take = Math.min(chunk.length, merged.length - offset);
|
|
187
|
+
if (take <= 0) break;
|
|
188
|
+
merged.set(chunk.subarray(0, take), offset);
|
|
189
|
+
offset += take;
|
|
190
|
+
}
|
|
191
|
+
return new TextDecoder("utf-8", { fatal: false }).decode(merged);
|
|
192
|
+
}
|
|
193
|
+
function parseHttpDate(value) {
|
|
194
|
+
if (!value) return void 0;
|
|
195
|
+
const ms = Date.parse(value);
|
|
196
|
+
return Number.isFinite(ms) ? new Date(ms).toISOString() : void 0;
|
|
197
|
+
}
|
|
198
|
+
/** Cheap heuristic that catches CAPTCHA, WAF block pages, and "Just a moment" interstitials. */
|
|
199
|
+
function looksLikeBlockPage(body) {
|
|
200
|
+
if (!body) return false;
|
|
201
|
+
const lower = body.toLowerCase();
|
|
202
|
+
for (const marker of [
|
|
203
|
+
"verify you are human",
|
|
204
|
+
"please enable javascript and cookies",
|
|
205
|
+
"just a moment",
|
|
206
|
+
"access denied",
|
|
207
|
+
"request unsuccessful",
|
|
208
|
+
"cf-error-details",
|
|
209
|
+
"captcha",
|
|
210
|
+
"incapsula",
|
|
211
|
+
"pardon our interruption"
|
|
212
|
+
]) if (lower.includes(marker)) return true;
|
|
213
|
+
return false;
|
|
214
|
+
}
|
|
215
|
+
function knownLargeAuthority(host) {
|
|
216
|
+
return host.endsWith("law.cornell.edu") || host.endsWith("irs.gov") || host.endsWith("sos.ca.gov") || host.endsWith("sos.state.tx.us") || host.endsWith("sos.state.us");
|
|
217
|
+
}
|
|
218
|
+
function cachePath(cacheDir, url) {
|
|
219
|
+
const key = sha256(url);
|
|
220
|
+
return join(cacheDir, "http", `${key.slice(0, 2)}`, `${key}.json`);
|
|
221
|
+
}
|
|
222
|
+
async function readCache(cacheDir, url, ttlMs) {
|
|
223
|
+
const path = cachePath(cacheDir, url);
|
|
224
|
+
try {
|
|
225
|
+
const info = await stat(path);
|
|
226
|
+
if (Date.now() - info.mtimeMs > ttlMs) return void 0;
|
|
227
|
+
const raw = await readFile(path, "utf8");
|
|
228
|
+
return {
|
|
229
|
+
...JSON.parse(raw),
|
|
230
|
+
fromCache: true
|
|
231
|
+
};
|
|
232
|
+
} catch {
|
|
233
|
+
return;
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
async function writeCache(cacheDir, url, value) {
|
|
237
|
+
const path = cachePath(cacheDir, url);
|
|
238
|
+
await mkdir(dirname(path), { recursive: true });
|
|
239
|
+
await writeFile(path, JSON.stringify(value), "utf8");
|
|
240
|
+
}
|
|
241
|
+
//#endregion
|
|
242
|
+
//#region src/sources/cornell-lii.ts
|
|
243
|
+
/**
|
|
244
|
+
* Cornell Legal Information Institute (LII) source.
|
|
245
|
+
*
|
|
246
|
+
* Pulls federal US Code sections and Wex encyclopedia entries — the two
|
|
247
|
+
* Cornell LII surfaces an agent typically grounds against. The Wex
|
|
248
|
+
* "non-compete" page is the canonical test case for the Ryan-LLC v. FTC
|
|
249
|
+
* vacatur drift the continuous-ingestion story is designed to catch.
|
|
250
|
+
*
|
|
251
|
+
* @stable
|
|
252
|
+
*/
|
|
253
|
+
const BASE_URL$1 = "https://www.law.cornell.edu";
|
|
254
|
+
/**
|
|
255
|
+
* Build a Cornell LII source for the listed selectors.
|
|
256
|
+
*
|
|
257
|
+
* Example: track DTSA + non-compete:
|
|
258
|
+
* ```
|
|
259
|
+
* createCornellLiiSource({
|
|
260
|
+
* selectors: [
|
|
261
|
+
* { kind: 'uscode', path: '18/1836' },
|
|
262
|
+
* { kind: 'wex', path: 'non-compete', dimensionHints: ['jurisdictional_accuracy'] },
|
|
263
|
+
* ],
|
|
264
|
+
* })
|
|
265
|
+
* ```
|
|
266
|
+
*/
|
|
267
|
+
function createCornellLiiSource(options) {
|
|
268
|
+
const id = options.id ?? "cornell-lii";
|
|
269
|
+
return {
|
|
270
|
+
id,
|
|
271
|
+
name: "Cornell Legal Information Institute",
|
|
272
|
+
description: "Federal US Code sections (uscode/text/...) and Wex legal encyclopedia entries from law.cornell.edu.",
|
|
273
|
+
async fetch(opts) {
|
|
274
|
+
const limit = opts.limit ?? options.selectors.length;
|
|
275
|
+
const selectors = options.selectors.slice(0, limit);
|
|
276
|
+
const out = [];
|
|
277
|
+
for (const selector of selectors) out.push(await fetchOne(id, selector, opts));
|
|
278
|
+
return out;
|
|
279
|
+
}
|
|
280
|
+
};
|
|
281
|
+
}
|
|
282
|
+
async function fetchOne(sourceId, selector, opts) {
|
|
283
|
+
const path = selector.path.replace(/^\/+/, "");
|
|
284
|
+
const url = selector.kind === "uscode" ? `${BASE_URL$1}/uscode/text/${path}` : `${BASE_URL$1}/wex/${path}`;
|
|
285
|
+
const response = await politeFetch(url, {
|
|
286
|
+
signal: opts.signal,
|
|
287
|
+
cacheDir: opts.cacheDir
|
|
288
|
+
});
|
|
289
|
+
const fragmentId = `${selector.kind}:${selector.path}`;
|
|
290
|
+
const dimensionHints = selector.dimensionHints ?? defaultDimensionHints(selector);
|
|
291
|
+
if (!response.verifiable) return {
|
|
292
|
+
id: fragmentId,
|
|
293
|
+
title: `Cornell LII ${selector.kind} ${selector.path}`,
|
|
294
|
+
body: "",
|
|
295
|
+
bodyHash: sha256(""),
|
|
296
|
+
provenance: {
|
|
297
|
+
url,
|
|
298
|
+
sourceUpdatedAt: response.sourceUpdatedAt,
|
|
299
|
+
fetchedAt: response.fetchedAt,
|
|
300
|
+
jurisdiction: "US-FED",
|
|
301
|
+
verifiable: false,
|
|
302
|
+
unverifiableReason: response.unverifiableReason
|
|
303
|
+
},
|
|
304
|
+
dimensionHints,
|
|
305
|
+
metadata: {
|
|
306
|
+
sourceId,
|
|
307
|
+
status: response.status,
|
|
308
|
+
fromCache: response.fromCache
|
|
309
|
+
}
|
|
310
|
+
};
|
|
311
|
+
const html = response.body;
|
|
312
|
+
const title = extractTitle$1(html, selector);
|
|
313
|
+
const body = extractBody(html, selector);
|
|
314
|
+
const effective = extractEffectiveDate(html) ?? response.sourceUpdatedAt;
|
|
315
|
+
const verifiable = body.length > 50;
|
|
316
|
+
return {
|
|
317
|
+
id: fragmentId,
|
|
318
|
+
title,
|
|
319
|
+
body,
|
|
320
|
+
bodyHash: sha256(body),
|
|
321
|
+
provenance: {
|
|
322
|
+
url,
|
|
323
|
+
sourceUpdatedAt: effective,
|
|
324
|
+
fetchedAt: response.fetchedAt,
|
|
325
|
+
jurisdiction: "US-FED",
|
|
326
|
+
verifiable,
|
|
327
|
+
unverifiableReason: verifiable ? void 0 : "extracted body too short"
|
|
328
|
+
},
|
|
329
|
+
dimensionHints,
|
|
330
|
+
metadata: {
|
|
331
|
+
sourceId,
|
|
332
|
+
status: response.status,
|
|
333
|
+
fromCache: response.fromCache
|
|
334
|
+
}
|
|
335
|
+
};
|
|
336
|
+
}
|
|
337
|
+
function extractTitle$1(html, selector) {
|
|
338
|
+
const h1 = /<h1[^>]*\bid=["']page_title["'][^>]*>([\s\S]*?)<\/h1>/i.exec(html)?.[1];
|
|
339
|
+
if (h1) return htmlToText(h1);
|
|
340
|
+
const t = /<title>([\s\S]*?)<\/title>/i.exec(html)?.[1];
|
|
341
|
+
if (t) return htmlToText(t).split(" | ")[0] ?? `Cornell LII ${selector.path}`;
|
|
342
|
+
return `Cornell LII ${selector.kind} ${selector.path}`;
|
|
343
|
+
}
|
|
344
|
+
function extractBody(html, selector) {
|
|
345
|
+
if (selector.kind === "uscode") {
|
|
346
|
+
const text = /<text>([\s\S]*?)<\/text>/i.exec(html)?.[1];
|
|
347
|
+
if (text) return htmlToText(text);
|
|
348
|
+
const tab = innerHtmlById(html, "tab_default_1");
|
|
349
|
+
if (tab) return htmlToText(tab);
|
|
350
|
+
}
|
|
351
|
+
const mainContent = innerHtmlById(html, "main-content");
|
|
352
|
+
if (mainContent) return htmlToText(mainContent.replace(/<h1[\s\S]*?<\/h1>/i, ""));
|
|
353
|
+
const extracted = innerHtmlById(html, "extracted-content");
|
|
354
|
+
if (extracted) return htmlToText(extracted.replace(/<h1[\s\S]*?<\/h1>/i, ""));
|
|
355
|
+
return htmlToText(html);
|
|
356
|
+
}
|
|
357
|
+
function extractEffectiveDate(html) {
|
|
358
|
+
const amend = /Amendments[\s\S]{0,200}?(\d{4})/i.exec(html)?.[1];
|
|
359
|
+
if (amend) {
|
|
360
|
+
const y = Number.parseInt(amend, 10);
|
|
361
|
+
if (Number.isFinite(y) && y > 1900 && y <= (/* @__PURE__ */ new Date()).getUTCFullYear() + 1) return new Date(Date.UTC(y, 11, 31)).toISOString();
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
function defaultDimensionHints(selector) {
|
|
365
|
+
if (selector.kind === "uscode") return ["jurisdictional_accuracy", "citation_hygiene"];
|
|
366
|
+
return ["citation_hygiene"];
|
|
367
|
+
}
|
|
368
|
+
//#endregion
|
|
369
|
+
//#region src/sources/irs-publications.ts
|
|
370
|
+
/**
|
|
371
|
+
* IRS publications source.
|
|
372
|
+
*
|
|
373
|
+
* Two surfaces:
|
|
374
|
+
*
|
|
375
|
+
* 1. The publications index at https://www.irs.gov/publications enumerates
|
|
376
|
+
* every active publication with its revision year — a single fragment
|
|
377
|
+
* with the full table lets change detection notice when a publication
|
|
378
|
+
* year flips (e.g. Pub 15 (2025) → Pub 15 (2026)).
|
|
379
|
+
*
|
|
380
|
+
* 2. Individual publication landing pages at /publications/p<N>[<suffix>]
|
|
381
|
+
* return one fragment per publication with summary text. Callers list
|
|
382
|
+
* the publications they need tracked via `selectors`.
|
|
383
|
+
*
|
|
384
|
+
* Revenue procedures are fetched under their numbered URLs; the IRS does
|
|
385
|
+
* not maintain a stable HTML index of rev-procs, so the caller passes the
|
|
386
|
+
* specific rev-proc paths they care about.
|
|
387
|
+
*
|
|
388
|
+
* @stable
|
|
389
|
+
*/
|
|
390
|
+
const BASE_URL = "https://www.irs.gov";
|
|
391
|
+
const INDEX_URL = `${BASE_URL}/publications`;
|
|
392
|
+
/** Default eval dimensions for IRS-sourced fragments. */
|
|
393
|
+
const IRS_DIMENSION_HINTS = [
|
|
394
|
+
"tax_compliance",
|
|
395
|
+
"regulatory_currency",
|
|
396
|
+
"citation_hygiene"
|
|
397
|
+
];
|
|
398
|
+
function createIrsPublicationsSource(options = {}) {
|
|
399
|
+
const id = options.id ?? "irs-publications";
|
|
400
|
+
const includeIndex = options.includeIndex ?? true;
|
|
401
|
+
return {
|
|
402
|
+
id,
|
|
403
|
+
name: "IRS Publications",
|
|
404
|
+
description: "Internal Revenue Service publications index and individual publication landing pages from irs.gov.",
|
|
405
|
+
async fetch(opts) {
|
|
406
|
+
const out = [];
|
|
407
|
+
const limit = opts.limit ?? Number.POSITIVE_INFINITY;
|
|
408
|
+
if (includeIndex && out.length < limit) out.push(await fetchIndex(id, opts));
|
|
409
|
+
for (const slug of options.publications ?? []) {
|
|
410
|
+
if (out.length >= limit) break;
|
|
411
|
+
out.push(await fetchPublication(id, slug, opts));
|
|
412
|
+
}
|
|
413
|
+
for (const path of options.revenueProcedures ?? []) {
|
|
414
|
+
if (out.length >= limit) break;
|
|
415
|
+
out.push(await fetchRevenueProcedure(id, path, opts));
|
|
416
|
+
}
|
|
417
|
+
return out;
|
|
418
|
+
}
|
|
419
|
+
};
|
|
420
|
+
}
|
|
421
|
+
async function fetchIndex(sourceId, opts) {
|
|
422
|
+
const response = await politeFetch(INDEX_URL, {
|
|
423
|
+
signal: opts.signal,
|
|
424
|
+
cacheDir: opts.cacheDir
|
|
425
|
+
});
|
|
426
|
+
const body = (response.body.match(/<table[\s\S]*?<\/table>/gi) ?? []).map((t) => htmlToText(t)).filter((t) => /Publication\s*\d+/i.test(t)).join("\n\n").slice(0, 2e5);
|
|
427
|
+
const verifiable = response.verifiable && body.length > 200;
|
|
428
|
+
return {
|
|
429
|
+
id: "index",
|
|
430
|
+
title: "IRS Publications Index",
|
|
431
|
+
body,
|
|
432
|
+
bodyHash: sha256(body),
|
|
433
|
+
provenance: {
|
|
434
|
+
url: INDEX_URL,
|
|
435
|
+
sourceUpdatedAt: response.sourceUpdatedAt,
|
|
436
|
+
fetchedAt: response.fetchedAt,
|
|
437
|
+
jurisdiction: "US-FED",
|
|
438
|
+
verifiable,
|
|
439
|
+
unverifiableReason: response.unverifiableReason ?? (verifiable ? void 0 : "no publication rows extracted")
|
|
440
|
+
},
|
|
441
|
+
dimensionHints: IRS_DIMENSION_HINTS,
|
|
442
|
+
metadata: {
|
|
443
|
+
sourceId,
|
|
444
|
+
status: response.status,
|
|
445
|
+
fromCache: response.fromCache,
|
|
446
|
+
kind: "index"
|
|
447
|
+
}
|
|
448
|
+
};
|
|
449
|
+
}
|
|
450
|
+
async function fetchPublication(sourceId, slug, opts) {
|
|
451
|
+
const url = `${BASE_URL}/publications/${slug.replace(/^\/+/, "")}`;
|
|
452
|
+
const response = await politeFetch(url, {
|
|
453
|
+
signal: opts.signal,
|
|
454
|
+
cacheDir: opts.cacheDir
|
|
455
|
+
});
|
|
456
|
+
const title = extractTitle(response.body, `IRS Publication ${slug}`);
|
|
457
|
+
const body = extractMainContent(response.body);
|
|
458
|
+
const verifiable = response.verifiable && body.length > 200;
|
|
459
|
+
return {
|
|
460
|
+
id: `publication:${slug}`,
|
|
461
|
+
title,
|
|
462
|
+
body,
|
|
463
|
+
bodyHash: sha256(body),
|
|
464
|
+
provenance: {
|
|
465
|
+
url,
|
|
466
|
+
sourceUpdatedAt: extractRevisionDate(response.body) ?? response.sourceUpdatedAt,
|
|
467
|
+
fetchedAt: response.fetchedAt,
|
|
468
|
+
jurisdiction: "US-FED",
|
|
469
|
+
verifiable,
|
|
470
|
+
unverifiableReason: response.unverifiableReason ?? (verifiable ? void 0 : "no publication body extracted")
|
|
471
|
+
},
|
|
472
|
+
dimensionHints: IRS_DIMENSION_HINTS,
|
|
473
|
+
metadata: {
|
|
474
|
+
sourceId,
|
|
475
|
+
status: response.status,
|
|
476
|
+
fromCache: response.fromCache,
|
|
477
|
+
kind: "publication",
|
|
478
|
+
slug
|
|
479
|
+
}
|
|
480
|
+
};
|
|
481
|
+
}
|
|
482
|
+
async function fetchRevenueProcedure(sourceId, path, opts) {
|
|
483
|
+
const url = `${BASE_URL}${path.startsWith("/") ? path : `/${path}`}`;
|
|
484
|
+
const response = await politeFetch(url, {
|
|
485
|
+
signal: opts.signal,
|
|
486
|
+
cacheDir: opts.cacheDir
|
|
487
|
+
});
|
|
488
|
+
const body = extractMainContent(response.body);
|
|
489
|
+
const verifiable = response.verifiable && body.length > 200;
|
|
490
|
+
return {
|
|
491
|
+
id: `rev-proc:${path}`,
|
|
492
|
+
title: extractTitle(response.body, `IRS Revenue Procedure ${path}`),
|
|
493
|
+
body,
|
|
494
|
+
bodyHash: sha256(body),
|
|
495
|
+
provenance: {
|
|
496
|
+
url,
|
|
497
|
+
sourceUpdatedAt: response.sourceUpdatedAt,
|
|
498
|
+
fetchedAt: response.fetchedAt,
|
|
499
|
+
jurisdiction: "US-FED",
|
|
500
|
+
verifiable,
|
|
501
|
+
unverifiableReason: response.unverifiableReason ?? (verifiable ? void 0 : "no revenue-procedure body extracted")
|
|
502
|
+
},
|
|
503
|
+
dimensionHints: [...IRS_DIMENSION_HINTS, "procedural_currency"],
|
|
504
|
+
metadata: {
|
|
505
|
+
sourceId,
|
|
506
|
+
status: response.status,
|
|
507
|
+
fromCache: response.fromCache,
|
|
508
|
+
kind: "rev-proc",
|
|
509
|
+
path
|
|
510
|
+
}
|
|
511
|
+
};
|
|
512
|
+
}
|
|
513
|
+
function extractTitle(html, fallback) {
|
|
514
|
+
const og = /<meta\s+property=["']og:title["']\s+content=["']([^"']+)["']/i.exec(html)?.[1];
|
|
515
|
+
if (og) return decodeHtml(og);
|
|
516
|
+
const title = /<title>([\s\S]*?)<\/title>/i.exec(html)?.[1];
|
|
517
|
+
if (title) return htmlToText(title).split(" | ")[0] ?? fallback;
|
|
518
|
+
return fallback;
|
|
519
|
+
}
|
|
520
|
+
function extractMainContent(html) {
|
|
521
|
+
const main = /<main\b[\s\S]*?<\/main>/i.exec(html)?.[0];
|
|
522
|
+
if (main) return htmlToText(main.replace(/<nav[\s\S]*?<\/nav>/gi, "").replace(/<header[\s\S]*?<\/header>/gi, "").replace(/<footer[\s\S]*?<\/footer>/gi, "")).slice(0, 2e5);
|
|
523
|
+
const body = /<body\b[\s\S]*?<\/body>/i.exec(html)?.[0];
|
|
524
|
+
return body ? htmlToText(body).slice(0, 2e5) : htmlToText(html).slice(0, 2e5);
|
|
525
|
+
}
|
|
526
|
+
function extractRevisionDate(html) {
|
|
527
|
+
const m = /Publication\s+\S+\s*\((\d{4})\)/i.exec(html);
|
|
528
|
+
if (m?.[1]) {
|
|
529
|
+
const year = Number.parseInt(m[1], 10);
|
|
530
|
+
if (Number.isFinite(year) && year >= 2e3 && year <= (/* @__PURE__ */ new Date()).getUTCFullYear() + 1) return new Date(Date.UTC(year, 0, 1)).toISOString();
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
function decodeHtml(value) {
|
|
534
|
+
return htmlToText(value);
|
|
535
|
+
}
|
|
536
|
+
//#endregion
|
|
537
|
+
//#region src/sources/state-sos.ts
|
|
538
|
+
function createStateSosSource(config) {
|
|
539
|
+
const id = config.id ?? `state-sos:${config.state.toLowerCase()}`;
|
|
540
|
+
return {
|
|
541
|
+
id,
|
|
542
|
+
name: config.name ?? `${config.state} Secretary of State`,
|
|
543
|
+
description: `${config.state} Secretary of State filings and formation guidance pages.`,
|
|
544
|
+
async fetch(opts) {
|
|
545
|
+
const limit = opts.limit ?? config.entities.length;
|
|
546
|
+
const entities = config.entities.slice(0, limit);
|
|
547
|
+
const out = [];
|
|
548
|
+
for (const entity of entities) out.push(await fetchEntity(id, config, entity, opts));
|
|
549
|
+
return out;
|
|
550
|
+
}
|
|
551
|
+
};
|
|
552
|
+
}
|
|
553
|
+
async function fetchEntity(sourceId, config, entity, opts) {
|
|
554
|
+
const url = joinUrl(config.baseUrl, entity.path);
|
|
555
|
+
const response = await politeFetch(url, {
|
|
556
|
+
signal: opts.signal,
|
|
557
|
+
cacheDir: opts.cacheDir
|
|
558
|
+
});
|
|
559
|
+
const body = response.verifiable ? extractBySelector(response.body, entity.selector) : "";
|
|
560
|
+
const verifiable = response.verifiable && body.length > 100;
|
|
561
|
+
return {
|
|
562
|
+
id: entity.id,
|
|
563
|
+
title: entity.title,
|
|
564
|
+
body,
|
|
565
|
+
bodyHash: sha256(body),
|
|
566
|
+
provenance: {
|
|
567
|
+
url,
|
|
568
|
+
sourceUpdatedAt: response.sourceUpdatedAt,
|
|
569
|
+
fetchedAt: response.fetchedAt,
|
|
570
|
+
jurisdiction: `US-${config.state.toUpperCase()}`,
|
|
571
|
+
verifiable,
|
|
572
|
+
unverifiableReason: response.unverifiableReason ?? (verifiable ? void 0 : "extracted body too short")
|
|
573
|
+
},
|
|
574
|
+
dimensionHints: entity.dimensionHints ?? [
|
|
575
|
+
"jurisdictional_accuracy",
|
|
576
|
+
"corporate_formation",
|
|
577
|
+
"citation_hygiene"
|
|
578
|
+
],
|
|
579
|
+
metadata: {
|
|
580
|
+
sourceId,
|
|
581
|
+
status: response.status,
|
|
582
|
+
fromCache: response.fromCache,
|
|
583
|
+
state: config.state
|
|
584
|
+
}
|
|
585
|
+
};
|
|
586
|
+
}
|
|
587
|
+
function extractBySelector(html, selector) {
|
|
588
|
+
if (selector.kind === "whole") {
|
|
589
|
+
const main = /<main\b[\s\S]*?<\/main>/i.exec(html)?.[0];
|
|
590
|
+
return htmlToText(main ?? html).slice(0, 2e5);
|
|
591
|
+
}
|
|
592
|
+
if (selector.kind === "regex") {
|
|
593
|
+
const m = selector.value.exec(html)?.[0];
|
|
594
|
+
return m ? htmlToText(m).slice(0, 2e5) : "";
|
|
595
|
+
}
|
|
596
|
+
if (selector.kind === "id") {
|
|
597
|
+
const escaped = selector.value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
598
|
+
const inner = new RegExp(`<([a-z][a-z0-9]*)\\b[^>]*\\sid=["']${escaped}["'][^>]*>([\\s\\S]*?)<\\/\\1>`, "i").exec(html)?.[2];
|
|
599
|
+
return inner ? htmlToText(inner).slice(0, 2e5) : "";
|
|
600
|
+
}
|
|
601
|
+
const escaped = selector.value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
602
|
+
const inner = new RegExp(`<([a-z][a-z0-9]*)\\b[^>]*\\sclass=["'][^"']*\\b${escaped}\\b[^"']*["'][^>]*>([\\s\\S]*?)<\\/\\1>`, "i").exec(html)?.[2];
|
|
603
|
+
return inner ? htmlToText(inner).slice(0, 2e5) : "";
|
|
604
|
+
}
|
|
605
|
+
function joinUrl(base, path) {
|
|
606
|
+
try {
|
|
607
|
+
return new URL(path, base.endsWith("/") ? base : `${base}/`).toString();
|
|
608
|
+
} catch {
|
|
609
|
+
return `${base.replace(/\/+$/, "")}/${path.replace(/^\/+/, "")}`;
|
|
610
|
+
}
|
|
611
|
+
}
|
|
612
|
+
//#endregion
|
|
613
|
+
export { IRS_DIMENSION_HINTS, MAX_RESPONSE_BYTES, MIN_REQUEST_GAP_MS, POLITE_USER_AGENT, __resetHttpThrottle, createCornellLiiSource, createIrsPublicationsSource, createStateSosSource, extractLinks, firstMatch, htmlToText, innerHtmlById, looksLikeBlockPage, politeFetch };
|
|
614
|
+
|
|
34
615
|
//# sourceMappingURL=index.js.map
|