@tangle-network/agent-knowledge 17.1.10 → 18.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -30,7 +30,7 @@
30
30
  * cache; an undefined value disables caching (useful in tests). `now` is
31
31
  * injected for deterministic tests of change-detection windows.
32
32
  */
33
- interface FetchOpts {
33
+ export interface FetchOpts {
34
34
  /** Abort signal forwarded to the underlying HTTP fetcher. */
35
35
  signal?: AbortSignal;
36
36
  /** Absolute path under which the source may cache raw bytes. */
@@ -58,7 +58,7 @@ interface FetchOpts {
58
58
  * `KnowledgeFragment` so freshness/change code can pass it around without
59
59
  * also dragging the body text.
60
60
  */
61
- interface FragmentProvenance {
61
+ export interface FragmentProvenance {
62
62
  /** Canonical URL the fragment was extracted from. */
63
63
  url: string;
64
64
  /**
@@ -92,7 +92,7 @@ interface FragmentProvenance {
92
92
  * One unit of authoritative content. Stable hash on `(id, body)` lets change
93
93
  * detection reason about identity across snapshots.
94
94
  */
95
- interface KnowledgeFragment {
95
+ export interface KnowledgeFragment {
96
96
  /**
97
97
  * Stable identity within (sourceId, selector-space). Two fetches against
98
98
  * the same authority section MUST produce the same `id`. The (sourceId,
@@ -130,7 +130,7 @@ interface KnowledgeFragment {
130
130
  * application's source list — there is no global registry by design (per
131
131
  * the per-tenant isolation contract; see README).
132
132
  */
133
- interface KnowledgeSource {
133
+ export interface KnowledgeSource {
134
134
  /** Stable id — used to key freshness state. MUST NOT change once shipped. */
135
135
  id: string;
136
136
  /** Human-readable name for dashboards. */
@@ -150,7 +150,7 @@ interface KnowledgeSource {
150
150
  }
151
151
  //#endregion
152
152
  //#region src/sources/cornell-lii.d.ts
153
- interface CornellLiiSelector {
153
+ export interface CornellLiiSelector {
154
154
  /** Either 'uscode' or 'wex'. */
155
155
  kind: 'uscode' | 'wex';
156
156
  /**
@@ -164,7 +164,7 @@ interface CornellLiiSelector {
164
164
  */
165
165
  dimensionHints?: string[];
166
166
  }
167
- interface CornellLiiSourceOptions {
167
+ export interface CornellLiiSourceOptions {
168
168
  /**
169
169
  * Selectors to fetch on each `fetch()` call. The caller (a per-tenant
170
170
  * workspace config, typically) lists exactly the authorities they need
@@ -188,7 +188,7 @@ interface CornellLiiSourceOptions {
188
188
  * })
189
189
  * ```
190
190
  */
191
- declare function createCornellLiiSource(options: CornellLiiSourceOptions): KnowledgeSource;
191
+ export declare function createCornellLiiSource(options: CornellLiiSourceOptions): KnowledgeSource;
192
192
  //#endregion
193
193
  //#region src/sources/html.d.ts
194
194
  /**
@@ -211,16 +211,16 @@ declare function createCornellLiiSource(options: CornellLiiSourceOptions): Knowl
211
211
  * Preserves paragraph and line breaks (`</p>`, `<br>`, `</li>`, `</div>`,
212
212
  * `</h*>`) as `\n` so statute text retains its subsection structure.
213
213
  */
214
- declare function htmlToText(html: string): string;
214
+ export declare function htmlToText(html: string): string;
215
215
  /** Extract the first match of a regex's first capture group, or undefined. */
216
- declare function firstMatch(html: string, pattern: RegExp): string | undefined;
216
+ export declare function firstMatch(html: string, pattern: RegExp): string | undefined;
217
217
  /** Extract the inner HTML of the first matching tag with id `id`. */
218
- declare function innerHtmlById(html: string, id: string): string | undefined;
218
+ export declare function innerHtmlById(html: string, id: string): string | undefined;
219
219
  /**
220
220
  * Extract every (href, text) pair matching the URL regex.
221
221
  * Returns absolute URLs by resolving against `baseUrl`.
222
222
  */
223
- declare function extractLinks(html: string, hrefPattern: RegExp, baseUrl: string): {
223
+ export declare function extractLinks(html: string, hrefPattern: RegExp, baseUrl: string): {
224
224
  href: string;
225
225
  text: string;
226
226
  }[];
@@ -236,12 +236,12 @@ declare function extractLinks(html: string, hrefPattern: RegExp, baseUrl: string
236
236
  * use successful status codes.
237
237
  */
238
238
  /** User-Agent string sent on every outbound request. */
239
- declare const POLITE_USER_AGENT = "agent-knowledge (+https://github.com/tangle-network/agent-knowledge)";
239
+ export declare const POLITE_USER_AGENT = "agent-knowledge (+https://github.com/tangle-network/agent-knowledge)";
240
240
  /** Minimum gap between successive requests to the same origin (ms). */
241
- declare const MIN_REQUEST_GAP_MS = 1000;
241
+ export declare const MIN_REQUEST_GAP_MS = 1000;
242
242
  /** Maximum response body we will buffer in memory (bytes). */
243
- declare const MAX_RESPONSE_BYTES: number;
244
- interface PoliteFetchOptions {
243
+ export declare const MAX_RESPONSE_BYTES: number;
244
+ export interface PoliteFetchOptions {
245
245
  signal?: AbortSignal;
246
246
  cacheDir?: string;
247
247
  /**
@@ -256,7 +256,7 @@ interface PoliteFetchOptions {
256
256
  */
257
257
  headers?: Record<string, string>;
258
258
  }
259
- interface PoliteFetchResult {
259
+ export interface PoliteFetchResult {
260
260
  url: string;
261
261
  status: number;
262
262
  /** Decoded UTF-8 body. Truncated to `MAX_RESPONSE_BYTES`. */
@@ -286,14 +286,14 @@ interface PoliteFetchResult {
286
286
  * Throws ONLY on `AbortError` (caller asked to stop) and on cache-write
287
287
  * failures that indicate a misconfigured filesystem.
288
288
  */
289
- declare function politeFetch(url: string, options?: PoliteFetchOptions): Promise<PoliteFetchResult>;
289
+ export declare function politeFetch(url: string, options?: PoliteFetchOptions): Promise<PoliteFetchResult>;
290
290
  /** Reset the in-process throttle map. Test-only. */
291
- declare function __resetHttpThrottle(): void;
291
+ export declare function __resetHttpThrottle(): void;
292
292
  /** Cheap heuristic that catches CAPTCHA, WAF block pages, and "Just a moment" interstitials. */
293
- declare function looksLikeBlockPage(body: string): boolean;
293
+ export declare function looksLikeBlockPage(body: string): boolean;
294
294
  //#endregion
295
295
  //#region src/sources/irs-publications.d.ts
296
- interface IrsPublicationsSourceOptions {
296
+ export interface IrsPublicationsSourceOptions {
297
297
  /**
298
298
  * Specific publication slugs to fetch (e.g. `['p15', 'p17', 'p463']`).
299
299
  * When `includeIndex` is true (default), the publications index page is
@@ -310,8 +310,8 @@ interface IrsPublicationsSourceOptions {
310
310
  id?: string;
311
311
  }
312
312
  /** Default eval dimensions for IRS-sourced fragments. */
313
- declare const IRS_DIMENSION_HINTS: string[];
314
- declare function createIrsPublicationsSource(options?: IrsPublicationsSourceOptions): KnowledgeSource;
313
+ export declare const IRS_DIMENSION_HINTS: string[];
314
+ export declare function createIrsPublicationsSource(options?: IrsPublicationsSourceOptions): KnowledgeSource;
315
315
  //#endregion
316
316
  //#region src/sources/state-sos.d.ts
317
317
  /**
@@ -331,7 +331,7 @@ declare function createIrsPublicationsSource(options?: IrsPublicationsSourceOpti
331
331
  *
332
332
  * @experimental Interface will likely grow as we add more state coverage.
333
333
  */
334
- interface StateSosEntity {
334
+ export interface StateSosEntity {
335
335
  /** Stable id for this fragment within the state (e.g. 'llc-formation', 'corp-formation'). */
336
336
  id: string;
337
337
  /** Path under the configured `baseUrl` for this entity. */
@@ -359,7 +359,7 @@ interface StateSosEntity {
359
359
  /** Eval dimensions this entity feeds. */
360
360
  dimensionHints?: string[];
361
361
  }
362
- interface StateSosSourceConfig {
362
+ export interface StateSosSourceConfig {
363
363
  /** US state postal code, e.g. 'CA', 'DE', 'TX'. */
364
364
  state: string;
365
365
  /** Base URL for the state SOS — e.g. 'https://www.sos.ca.gov'. */
@@ -371,7 +371,6 @@ interface StateSosSourceConfig {
371
371
  /** Display name; default `<state> Secretary of State`. */
372
372
  name?: string;
373
373
  }
374
- declare function createStateSosSource(config: StateSosSourceConfig): KnowledgeSource;
374
+ export declare function createStateSosSource(config: StateSosSourceConfig): KnowledgeSource;
375
375
  //#endregion
376
- export { CornellLiiSelector, CornellLiiSourceOptions, FetchOpts, FragmentProvenance, IRS_DIMENSION_HINTS, IrsPublicationsSourceOptions, KnowledgeFragment, KnowledgeSource, MAX_RESPONSE_BYTES, MIN_REQUEST_GAP_MS, POLITE_USER_AGENT, PoliteFetchOptions, PoliteFetchResult, StateSosEntity, StateSosSourceConfig, __resetHttpThrottle, createCornellLiiSource, createIrsPublicationsSource, createStateSosSource, extractLinks, firstMatch, htmlToText, innerHtmlById, looksLikeBlockPage, politeFetch };
377
376
  //# sourceMappingURL=index.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/sources/types.ts","../../src/sources/cornell-lii.ts","../../src/sources/html.ts","../../src/sources/http.ts","../../src/sources/irs-publications.ts","../../src/sources/state-sos.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;UAgCiB;;EAEf,SAAS;;EAET;;EAEA,YAAY;;;;;;EAMZ;;;;;;;;;EASA;;;;;;;UAQe;;EAEf;;;;;;;EAOA;;EAEA;;;;;;;EAOA;;;;;;;;EAQA;;EAEA;;;;;;UAOe;;;;;;EAMf;;EAEA;;EAEA;;EAEA;EACA,YAAY;;;;;;;;;;;;;EAaZ;;EAEA,WAAW;;;;;;;;;;UAWI;;EAEf;;EAEA;;EAEA;;;;;;;;;;EAUA,MAAM,MAAM,YAAY,QAAQ;;;;UCrIjB;;EAEf;;;;;EAKA;;;;;EAKA;;UAGe;;;;;;;EAOf,WAAW;;EAEX;;;;;;;;;;;;;;;iBAgBc,uBAAuB,SAAS,0BAA0B;;;;;;;;;;;;;;;;;;;;;;;iBCrC1D,WAAW;;iBA4BX,WAAW,cAAc,SAAS;;iBAKlC,cAAc,cAAc;;;;;iBAa5B,aACd,cACA,aAAa,QACb;EACG;EAAc;;;;;;;;;;;;;;cCxDN;;cAIA;;cAGA;UAII;EACf,SAAS;EACT;;;;;;EAMA;;;;;EAKA,UAAU;;UAGK;EACf;EACA;;EAEA;;;;;EAKA;EACA;;EAEA;;;;;;EAMA;EACA;;;;;;;;;;;iBAYoB,YACpB,aACA,UAAS,qBACR,QAAQ;;iBAoEK;;iBA6DA,mBAAmB;;;UChLlB;;;;;;;EAOf;;;;;EAKA;EACA;EACA;;;cAIW;iBAEG,4BACd,UAAS,+BACR;;;;;;;;;;;;;;;;;;;;UC5Bc;;EAEf;;EAEA;;;;;;;;EAQA;IACM;IAAY;;IACZ;IAAe;;IACf;IAAe,OAAO;;IACtB;;EACN;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA,UAAU;;EAEV;;EAEA;;iBAGc,qBAAqB,QAAQ,uBAAuB"}
1
+ {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/sources/types.ts","../../src/sources/cornell-lii.ts","../../src/sources/html.ts","../../src/sources/http.ts","../../src/sources/irs-publications.ts","../../src/sources/state-sos.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBAgCiB;;EAEf,SAAS;;EAET;;EAEA,YAAY;;;;;;EAMZ;;;;;;;;;EASA;;;;;;;iBAQe;;EAEf;;;;;;;EAOA;;EAEA;;;;;;;EAOA;;;;;;;;EAQA;;EAEA;;;;;;iBAOe;;;;;;EAMf;;EAEA;;EAEA;;EAEA;EACA,YAAY;;;;;;;;;;;;;EAaZ;;EAEA,WAAW;;;;;;;;;;iBAWI;;EAEf;;EAEA;;EAEA;;;;;;;;;;EAUA,MAAM,MAAM,YAAY,QAAQ;;;;iBCrIjB;;EAEf;;;;;EAKA;;;;;EAKA;;iBAGe;;;;;;;EAOf,WAAW;;EAEX;;;;;;;;;;;;;;;wBAgBc,uBAAuB,SAAS,0BAA0B;;;;;;;;;;;;;;;;;;;;;;;wBCrC1D,WAAW;;wBA4BX,WAAW,cAAc,SAAS;;wBAKlC,cAAc,cAAc;;;;;wBAa5B,aACd,cACA,aAAa,QACb;EACG;EAAc;;;;;;;;;;;;;;qBCxDN;;qBAIA;;qBAGA;iBAII;EACf,SAAS;EACT;;;;;;EAMA;;;;;EAKA,UAAU;;iBAGK;EACf;EACA;;EAEA;;;;;EAKA;EACA;;EAEA;;;;;;EAMA;EACA;;;;;;;;;;;wBAYoB,YACpB,aACA,UAAS,qBACR,QAAQ;;wBAoEK;;wBA6DA,mBAAmB;;;iBChLlB;;;;;;;EAOf;;;;;EAKA;EACA;EACA;;;qBAIW;wBAEG,4BACd,UAAS,+BACR;;;;;;;;;;;;;;;;;;;;iBC5Bc;;EAEf;;EAEA;;;;;;;;EAQA;IACM;IAAY;;IACZ;IAAe;;IACf;IAAe,OAAO;;IACtB;;EACN;;EAEA;;iBAGe;;EAEf;;EAEA;;EAEA,UAAU;;EAEV;;EAEA;;wBAGc,qBAAqB,QAAQ,uBAAuB"}
@@ -72,7 +72,7 @@ const POLITE_USER_AGENT = "agent-knowledge (+https://github.com/tangle-network/a
72
72
  /** Minimum gap between successive requests to the same origin (ms). */
73
73
  const MIN_REQUEST_GAP_MS = 1e3;
74
74
  /** Maximum response body we will buffer in memory (bytes). */
75
- const MAX_RESPONSE_BYTES = 8 * 1024 * 1024;
75
+ const MAX_RESPONSE_BYTES = 8388608;
76
76
  const hostThrottle = /* @__PURE__ */ new Map();
77
77
  /**
78
78
  * Fetch one URL with per-host throttling, on-disk cache, and block-page
@@ -84,7 +84,7 @@ const hostThrottle = /* @__PURE__ */ new Map();
84
84
  * failures that indicate a misconfigured filesystem.
85
85
  */
86
86
  async function politeFetch(url, options = {}) {
87
- const cacheTtl = options.cacheTtlMs ?? 3600 * 1e3;
87
+ const cacheTtl = options.cacheTtlMs ?? 36e5;
88
88
  const cached = options.cacheDir ? await readCache(options.cacheDir, url, cacheTtl) : void 0;
89
89
  if (cached) return cached;
90
90
  const host = safeHost(url);
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","names":["BASE_URL","extractTitle"],"sources":["../../src/sources/html.ts","../../src/sources/http.ts","../../src/sources/cornell-lii.ts","../../src/sources/irs-publications.ts","../../src/sources/state-sos.ts"],"sourcesContent":["/**\n * Minimal HTML helpers used by the shipped sources.\n *\n * Deliberately not a full DOM parser: every authority we ship against\n * (Cornell LII, IRS.gov, state SOS portals) has well-behaved server-rendered\n * HTML where regex-based extraction is correct and cheap. Bringing in cheerio\n * would add a 1.5MB dependency to a package whose purpose is shipping\n * primitives, not parsing arbitrary web pages.\n *\n * If a future source needs real DOM traversal, it should depend on its own\n * parser locally rather than promoting one into the package-wide deps.\n *\n * @stable\n */\n\n/**\n * Strip HTML tags, collapse whitespace, decode common entities.\n *\n * Preserves paragraph and line breaks (`</p>`, `<br>`, `</li>`, `</div>`,\n * `</h*>`) as `\\n` so statute text retains its subsection structure.\n */\nexport function htmlToText(html: string): string {\n return html\n .replace(/<script[\\s\\S]*?<\\/script>/gi, '')\n .replace(/<style[\\s\\S]*?<\\/style>/gi, '')\n .replace(/<noscript[\\s\\S]*?<\\/noscript>/gi, '')\n .replace(/<!--([\\s\\S]*?)-->/g, '')\n .replace(/<\\s*br\\s*\\/?>/gi, '\\n')\n .replace(/<\\/(p|li|div|tr|h[1-6]|blockquote|section|article)>/gi, '\\n')\n .replace(/<[^>]+>/g, '')\n .replace(/&nbsp;/gi, ' ')\n .replace(/&amp;/gi, '&')\n .replace(/&lt;/gi, '<')\n .replace(/&gt;/gi, '>')\n .replace(/&quot;/gi, '\"')\n .replace(/&#39;/gi, \"'\")\n .replace(/&sect;/gi, '§')\n .replace(/&mdash;/gi, '—')\n .replace(/&ndash;/gi, '–')\n .replace(/&#(\\d+);/g, (_, code) => String.fromCodePoint(Number(code)))\n .replace(/&#x([0-9a-f]+);/gi, (_, code) => String.fromCodePoint(Number.parseInt(code, 16)))\n .split('\\n')\n .map((line) => line.replace(/[\\t  ]+/g, ' ').trim())\n .filter((line, idx, all) => !(line === '' && all[idx - 1] === ''))\n .join('\\n')\n .trim()\n}\n\n/** Extract the first match of a regex's first capture group, or undefined. */\nexport function firstMatch(html: string, pattern: RegExp): string | undefined {\n return pattern.exec(html)?.[1]?.trim()\n}\n\n/** Extract the inner HTML of the first matching tag with id `id`. */\nexport function innerHtmlById(html: string, id: string): string | undefined {\n const escaped = id.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n const tagPattern = new RegExp(\n `<([a-z][a-z0-9]*)\\\\b[^>]*\\\\sid=[\"']${escaped}[\"'][^>]*>([\\\\s\\\\S]*?)<\\\\/\\\\1>`,\n 'i',\n )\n return tagPattern.exec(html)?.[2]\n}\n\n/**\n * Extract every (href, text) pair matching the URL regex.\n * Returns absolute URLs by resolving against `baseUrl`.\n */\nexport function extractLinks(\n html: string,\n hrefPattern: RegExp,\n baseUrl: string,\n): { href: string; text: string }[] {\n const out: { href: string; text: string }[] = []\n const anchor = /<a\\b[^>]*\\shref=[\"']([^\"']+)[\"'][^>]*>([\\s\\S]*?)<\\/a>/gi\n for (const match of html.matchAll(anchor)) {\n const href = match[1]\n const inner = match[2]\n if (!href || !inner) continue\n if (!hrefPattern.test(href)) continue\n const text = htmlToText(inner)\n if (!text) continue\n try {\n out.push({ href: new URL(href, baseUrl).toString(), text })\n } catch {\n /* skip malformed URL */\n }\n }\n return out\n}\n","import { mkdir, readFile, stat, writeFile } from 'node:fs/promises'\nimport { dirname, join } from 'node:path'\nimport { sha256 } from '../ids'\n\n/**\n * Polite HTTP fetcher shared by remote sources.\n *\n * Independent sources share a per-origin throttle because rate-limited sites\n * may return block pages instead of 429 responses. Responses are cached by URL\n * because many publishers omit reliable ETag and Last-Modified headers. Bodies\n * are checked even after a 2xx response because captcha and block pages often\n * use successful status codes.\n */\n\n/** User-Agent string sent on every outbound request. */\nexport const POLITE_USER_AGENT =\n 'agent-knowledge (+https://github.com/tangle-network/agent-knowledge)'\n\n/** Minimum gap between successive requests to the same origin (ms). */\nexport const MIN_REQUEST_GAP_MS = 1_000\n\n/** Maximum response body we will buffer in memory (bytes). */\nexport const MAX_RESPONSE_BYTES = 8 * 1024 * 1024\n\nconst hostThrottle = new Map<string, Promise<void>>()\n\nexport interface PoliteFetchOptions {\n signal?: AbortSignal\n cacheDir?: string\n /**\n * Cache age beyond which we re-fetch. Default 1 hour, long enough to\n * batch a cron sweep across many selectors, short enough that hourly\n * authoritative-page changes get picked up next tick.\n */\n cacheTtlMs?: number\n /**\n * Extra request headers. The fetcher always sets `User-Agent` and\n * `Accept`; callers can add `Accept-Language` etc.\n */\n headers?: Record<string, string>\n}\n\nexport interface PoliteFetchResult {\n url: string\n status: number\n /** Decoded UTF-8 body. Truncated to `MAX_RESPONSE_BYTES`. */\n body: string\n /**\n * Best-effort source-attested timestamp. Reads `Last-Modified`,\n * falling back to `Date`, falling back to fetch time. Always ISO 8601.\n */\n sourceUpdatedAt: string\n fetchedAt: string\n /** True iff the response was satisfied from disk cache. */\n fromCache: boolean\n /**\n * False on: non-2xx status, captcha/block page heuristic match, or\n * decoded body below 200 chars from a host known to serve real content\n * (Cornell, IRS, state SOS). `unverifiableReason` carries the why.\n */\n verifiable: boolean\n unverifiableReason?: string\n}\n\n/**\n * Fetch one URL with per-host throttling, on-disk cache, and block-page\n * detection. Never throws on network/HTTP failure. It returns a result with\n * `verifiable: false` and `unverifiableReason` set so the caller can decide\n * whether to skip, retry, or surface.\n *\n * Throws ONLY on `AbortError` (caller asked to stop) and on cache-write\n * failures that indicate a misconfigured filesystem.\n */\nexport async function politeFetch(\n url: string,\n options: PoliteFetchOptions = {},\n): Promise<PoliteFetchResult> {\n const cacheTtl = options.cacheTtlMs ?? 60 * 60 * 1000\n const cached = options.cacheDir ? await readCache(options.cacheDir, url, cacheTtl) : undefined\n if (cached) return cached\n\n const host = safeHost(url)\n await throttleHost(host)\n\n const fetchedAt = new Date().toISOString()\n let response: Response\n try {\n response = await fetch(url, {\n signal: options.signal,\n redirect: 'follow',\n headers: {\n 'User-Agent': POLITE_USER_AGENT,\n Accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',\n 'Accept-Language': 'en-US,en;q=0.9',\n ...(options.headers ?? {}),\n },\n })\n } catch (error) {\n if ((error as { name?: string }).name === 'AbortError') throw error\n const result: PoliteFetchResult = {\n url,\n status: 0,\n body: '',\n sourceUpdatedAt: fetchedAt,\n fetchedAt,\n fromCache: false,\n verifiable: false,\n unverifiableReason: `network error: ${(error as Error).message}`,\n }\n if (options.cacheDir) await writeCache(options.cacheDir, url, result)\n return result\n }\n\n const text = await readBoundedText(response)\n const lastModified = response.headers.get('last-modified')\n const dateHeader = response.headers.get('date')\n const sourceUpdatedAt = parseHttpDate(lastModified) ?? parseHttpDate(dateHeader) ?? fetchedAt\n\n const result: PoliteFetchResult = {\n url,\n status: response.status,\n body: text,\n sourceUpdatedAt,\n fetchedAt,\n fromCache: false,\n verifiable: true,\n }\n\n if (response.status < 200 || response.status >= 300) {\n result.verifiable = false\n result.unverifiableReason = `non-2xx status: ${response.status}`\n } else if (looksLikeBlockPage(text)) {\n result.verifiable = false\n result.unverifiableReason = 'block-page heuristic matched'\n } else if (text.length < 200 && knownLargeAuthority(host)) {\n result.verifiable = false\n result.unverifiableReason = `body shorter than expected (${text.length} chars)`\n }\n\n if (options.cacheDir) await writeCache(options.cacheDir, url, result)\n return result\n}\n\n/** Reset the in-process throttle map. Test-only. */\nexport function __resetHttpThrottle(): void {\n hostThrottle.clear()\n}\n\nfunction safeHost(url: string): string {\n try {\n return new URL(url).host\n } catch {\n return 'unknown'\n }\n}\n\nasync function throttleHost(host: string): Promise<void> {\n const prev = hostThrottle.get(host) ?? Promise.resolve()\n let release: () => void = () => {}\n const next = new Promise<void>((resolve) => {\n release = resolve\n })\n hostThrottle.set(\n host,\n prev.then(() => next),\n )\n await prev\n setTimeout(release, MIN_REQUEST_GAP_MS)\n}\n\nasync function readBoundedText(response: Response): Promise<string> {\n if (!response.body) return ''\n const reader = response.body.getReader()\n const chunks: Uint8Array[] = []\n let total = 0\n while (true) {\n const { done, value } = await reader.read()\n if (done) break\n if (!value) continue\n total += value.length\n if (total > MAX_RESPONSE_BYTES) {\n // Stop reading; release the underlying connection.\n await reader.cancel()\n break\n }\n chunks.push(value)\n }\n const merged = new Uint8Array(Math.min(total, MAX_RESPONSE_BYTES))\n let offset = 0\n for (const chunk of chunks) {\n const take = Math.min(chunk.length, merged.length - offset)\n if (take <= 0) break\n merged.set(chunk.subarray(0, take), offset)\n offset += take\n }\n return new TextDecoder('utf-8', { fatal: false }).decode(merged)\n}\n\nfunction parseHttpDate(value: string | null): string | undefined {\n if (!value) return undefined\n const ms = Date.parse(value)\n return Number.isFinite(ms) ? new Date(ms).toISOString() : undefined\n}\n\n/** Cheap heuristic that catches CAPTCHA, WAF block pages, and \"Just a moment\" interstitials. */\nexport function looksLikeBlockPage(body: string): boolean {\n if (!body) return false\n const lower = body.toLowerCase()\n const markers = [\n 'verify you are human',\n 'please enable javascript and cookies',\n 'just a moment',\n 'access denied',\n 'request unsuccessful',\n 'cf-error-details',\n 'captcha',\n 'incapsula',\n 'pardon our interruption',\n ]\n for (const marker of markers) {\n if (lower.includes(marker)) return true\n }\n return false\n}\n\nfunction knownLargeAuthority(host: string): boolean {\n return (\n host.endsWith('law.cornell.edu') ||\n host.endsWith('irs.gov') ||\n host.endsWith('sos.ca.gov') ||\n host.endsWith('sos.state.tx.us') ||\n host.endsWith('sos.state.us')\n )\n}\n\nfunction cachePath(cacheDir: string, url: string): string {\n const key = sha256(url)\n return join(cacheDir, 'http', `${key.slice(0, 2)}`, `${key}.json`)\n}\n\nasync function readCache(\n cacheDir: string,\n url: string,\n ttlMs: number,\n): Promise<PoliteFetchResult | undefined> {\n const path = cachePath(cacheDir, url)\n try {\n const info = await stat(path)\n if (Date.now() - info.mtimeMs > ttlMs) return undefined\n const raw = await readFile(path, 'utf8')\n const parsed = JSON.parse(raw) as PoliteFetchResult\n return { ...parsed, fromCache: true }\n } catch {\n return undefined\n }\n}\n\nasync function writeCache(cacheDir: string, url: string, value: PoliteFetchResult): Promise<void> {\n const path = cachePath(cacheDir, url)\n await mkdir(dirname(path), { recursive: true })\n await writeFile(path, JSON.stringify(value), 'utf8')\n}\n","import { sha256 } from '../ids'\nimport { htmlToText, innerHtmlById } from './html'\nimport { politeFetch } from './http'\nimport type { FetchOpts, KnowledgeFragment, KnowledgeSource } from './types'\n\n/**\n * Cornell Legal Information Institute (LII) source.\n *\n * Pulls federal US Code sections and Wex encyclopedia entries — the two\n * Cornell LII surfaces an agent typically grounds against. The Wex\n * \"non-compete\" page is the canonical test case for the Ryan-LLC v. FTC\n * vacatur drift the continuous-ingestion story is designed to catch.\n *\n * @stable\n */\n\nconst BASE_URL = 'https://www.law.cornell.edu'\n\nexport interface CornellLiiSelector {\n /** Either 'uscode' or 'wex'. */\n kind: 'uscode' | 'wex'\n /**\n * For `uscode`: `<title>/<section>` (e.g. `'18/1836'` for DTSA).\n * For `wex`: the slug (e.g. `'non-compete'`).\n */\n path: string\n /**\n * Optional pre-declared eval dimensions affected by this section. If\n * omitted, defaults are chosen from `kind` + path heuristics.\n */\n dimensionHints?: string[]\n}\n\nexport interface CornellLiiSourceOptions {\n /**\n * Selectors to fetch on each `fetch()` call. The caller (a per-tenant\n * workspace config, typically) lists exactly the authorities they need\n * tracked. There is no auto-discovery; that would crawl Cornell at\n * cron speed, which is what the polite-fetch contract exists to avoid.\n */\n selectors: CornellLiiSelector[]\n /** Source id override; default is `'cornell-lii'`. */\n id?: string\n}\n\n/**\n * Build a Cornell LII source for the listed selectors.\n *\n * Example: track DTSA + non-compete:\n * ```\n * createCornellLiiSource({\n * selectors: [\n * { kind: 'uscode', path: '18/1836' },\n * { kind: 'wex', path: 'non-compete', dimensionHints: ['jurisdictional_accuracy'] },\n * ],\n * })\n * ```\n */\nexport function createCornellLiiSource(options: CornellLiiSourceOptions): KnowledgeSource {\n const id = options.id ?? 'cornell-lii'\n return {\n id,\n name: 'Cornell Legal Information Institute',\n description:\n 'Federal US Code sections (uscode/text/...) and Wex legal encyclopedia entries from law.cornell.edu.',\n async fetch(opts: FetchOpts): Promise<KnowledgeFragment[]> {\n const limit = opts.limit ?? options.selectors.length\n const selectors = options.selectors.slice(0, limit)\n const out: KnowledgeFragment[] = []\n for (const selector of selectors) {\n out.push(await fetchOne(id, selector, opts))\n }\n return out\n },\n }\n}\n\nasync function fetchOne(\n sourceId: string,\n selector: CornellLiiSelector,\n opts: FetchOpts,\n): Promise<KnowledgeFragment> {\n const path = selector.path.replace(/^\\/+/, '')\n const url =\n selector.kind === 'uscode' ? `${BASE_URL}/uscode/text/${path}` : `${BASE_URL}/wex/${path}`\n\n const response = await politeFetch(url, {\n signal: opts.signal,\n cacheDir: opts.cacheDir,\n })\n\n const fragmentId = `${selector.kind}:${selector.path}`\n const dimensionHints = selector.dimensionHints ?? defaultDimensionHints(selector)\n\n if (!response.verifiable) {\n return {\n id: fragmentId,\n title: `Cornell LII ${selector.kind} ${selector.path}`,\n body: '',\n bodyHash: sha256(''),\n provenance: {\n url,\n sourceUpdatedAt: response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable: false,\n unverifiableReason: response.unverifiableReason,\n },\n dimensionHints,\n metadata: { sourceId, status: response.status, fromCache: response.fromCache },\n }\n }\n\n const html = response.body\n const title = extractTitle(html, selector)\n const body = extractBody(html, selector)\n const effective = extractEffectiveDate(html) ?? response.sourceUpdatedAt\n\n const verifiable = body.length > 50\n return {\n id: fragmentId,\n title,\n body,\n bodyHash: sha256(body),\n provenance: {\n url,\n sourceUpdatedAt: effective,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable,\n unverifiableReason: verifiable ? undefined : 'extracted body too short',\n },\n dimensionHints,\n metadata: { sourceId, status: response.status, fromCache: response.fromCache },\n }\n}\n\nfunction extractTitle(html: string, selector: CornellLiiSelector): string {\n const h1 = /<h1[^>]*\\bid=[\"']page_title[\"'][^>]*>([\\s\\S]*?)<\\/h1>/i.exec(html)?.[1]\n if (h1) return htmlToText(h1)\n const t = /<title>([\\s\\S]*?)<\\/title>/i.exec(html)?.[1]\n if (t) return htmlToText(t).split(' | ')[0] ?? `Cornell LII ${selector.path}`\n return `Cornell LII ${selector.kind} ${selector.path}`\n}\n\nfunction extractBody(html: string, selector: CornellLiiSelector): string {\n if (selector.kind === 'uscode') {\n // The statute text lives inside a <text><div class=\"text\">…</div></text>\n // block on US Code section pages. Prefer it; fall back to #tab_default_1\n // which always contains the section body.\n const text = /<text>([\\s\\S]*?)<\\/text>/i.exec(html)?.[1]\n if (text) return htmlToText(text)\n const tab = innerHtmlById(html, 'tab_default_1')\n if (tab) return htmlToText(tab)\n }\n // Wex pages wrap the encyclopedia entry under <div id=\"main-content\"> (newer\n // Drupal template) or directly inside <div id=\"extracted-content\"> (older\n // template). Try both — lazy regex matching against a nested-div container\n // returns the wrong (shorter) slice, so we anchor on the leaf containers.\n const mainContent = innerHtmlById(html, 'main-content')\n if (mainContent) {\n return htmlToText(mainContent.replace(/<h1[\\s\\S]*?<\\/h1>/i, ''))\n }\n const extracted = innerHtmlById(html, 'extracted-content')\n if (extracted) {\n return htmlToText(extracted.replace(/<h1[\\s\\S]*?<\\/h1>/i, ''))\n }\n return htmlToText(html)\n}\n\nfunction extractEffectiveDate(html: string): string | undefined {\n // Cornell LII includes \"Editorial Notes\" / \"Amendments\" blocks with\n // dates; the most reliable machine-readable signal is the last\n // amendment year embedded near the section text.\n const amend = /Amendments[\\s\\S]{0,200}?(\\d{4})/i.exec(html)?.[1]\n if (amend) {\n const y = Number.parseInt(amend, 10)\n if (Number.isFinite(y) && y > 1900 && y <= new Date().getUTCFullYear() + 1) {\n return new Date(Date.UTC(y, 11, 31)).toISOString()\n }\n }\n return undefined\n}\n\nfunction defaultDimensionHints(selector: CornellLiiSelector): string[] {\n if (selector.kind === 'uscode') return ['jurisdictional_accuracy', 'citation_hygiene']\n return ['citation_hygiene']\n}\n","import { sha256 } from '../ids'\nimport { htmlToText } from './html'\nimport { politeFetch } from './http'\nimport type { FetchOpts, KnowledgeFragment, KnowledgeSource } from './types'\n\n/**\n * IRS publications source.\n *\n * Two surfaces:\n *\n * 1. The publications index at https://www.irs.gov/publications enumerates\n * every active publication with its revision year — a single fragment\n * with the full table lets change detection notice when a publication\n * year flips (e.g. Pub 15 (2025) → Pub 15 (2026)).\n *\n * 2. Individual publication landing pages at /publications/p<N>[<suffix>]\n * return one fragment per publication with summary text. Callers list\n * the publications they need tracked via `selectors`.\n *\n * Revenue procedures are fetched under their numbered URLs; the IRS does\n * not maintain a stable HTML index of rev-procs, so the caller passes the\n * specific rev-proc paths they care about.\n *\n * @stable\n */\n\nconst BASE_URL = 'https://www.irs.gov'\nconst INDEX_URL = `${BASE_URL}/publications`\n\nexport interface IrsPublicationsSourceOptions {\n /**\n * Specific publication slugs to fetch (e.g. `['p15', 'p17', 'p463']`).\n * When `includeIndex` is true (default), the publications index page is\n * also fetched as a single fragment so change detection can notice\n * year/revision shifts across the whole catalogue.\n */\n publications?: string[]\n /**\n * Revenue procedure paths to fetch (e.g. `['/irb/2024-31_IRB']`). The\n * caller passes the exact path; this source does not auto-discover.\n */\n revenueProcedures?: string[]\n includeIndex?: boolean\n id?: string\n}\n\n/** Default eval dimensions for IRS-sourced fragments. */\nexport const IRS_DIMENSION_HINTS = ['tax_compliance', 'regulatory_currency', 'citation_hygiene']\n\nexport function createIrsPublicationsSource(\n options: IrsPublicationsSourceOptions = {},\n): KnowledgeSource {\n const id = options.id ?? 'irs-publications'\n const includeIndex = options.includeIndex ?? true\n return {\n id,\n name: 'IRS Publications',\n description:\n 'Internal Revenue Service publications index and individual publication landing pages from irs.gov.',\n async fetch(opts: FetchOpts): Promise<KnowledgeFragment[]> {\n const out: KnowledgeFragment[] = []\n const limit = opts.limit ?? Number.POSITIVE_INFINITY\n\n if (includeIndex && out.length < limit) {\n out.push(await fetchIndex(id, opts))\n }\n for (const slug of options.publications ?? []) {\n if (out.length >= limit) break\n out.push(await fetchPublication(id, slug, opts))\n }\n for (const path of options.revenueProcedures ?? []) {\n if (out.length >= limit) break\n out.push(await fetchRevenueProcedure(id, path, opts))\n }\n return out\n },\n }\n}\n\nasync function fetchIndex(sourceId: string, opts: FetchOpts): Promise<KnowledgeFragment> {\n const response = await politeFetch(INDEX_URL, { signal: opts.signal, cacheDir: opts.cacheDir })\n const tablePattern = /<table[\\s\\S]*?<\\/table>/gi\n const matches = response.body.match(tablePattern) ?? []\n // Extract the table that lists current-year publications. IRS publishes\n // one table per year on the index; the most recent table is always the\n // first that mentions a year ≥ current.\n const tables = matches.map((t) => htmlToText(t))\n const body = tables\n .filter((t) => /Publication\\s*\\d+/i.test(t))\n .join('\\n\\n')\n .slice(0, 200_000)\n\n const verifiable = response.verifiable && body.length > 200\n return {\n id: 'index',\n title: 'IRS Publications Index',\n body,\n bodyHash: sha256(body),\n provenance: {\n url: INDEX_URL,\n sourceUpdatedAt: response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable,\n unverifiableReason:\n response.unverifiableReason ?? (verifiable ? undefined : 'no publication rows extracted'),\n },\n dimensionHints: IRS_DIMENSION_HINTS,\n metadata: { sourceId, status: response.status, fromCache: response.fromCache, kind: 'index' },\n }\n}\n\nasync function fetchPublication(\n sourceId: string,\n slug: string,\n opts: FetchOpts,\n): Promise<KnowledgeFragment> {\n const url = `${BASE_URL}/publications/${slug.replace(/^\\/+/, '')}`\n const response = await politeFetch(url, { signal: opts.signal, cacheDir: opts.cacheDir })\n\n const title = extractTitle(response.body, `IRS Publication ${slug}`)\n const body = extractMainContent(response.body)\n const verifiable = response.verifiable && body.length > 200\n\n return {\n id: `publication:${slug}`,\n title,\n body,\n bodyHash: sha256(body),\n provenance: {\n url,\n sourceUpdatedAt: extractRevisionDate(response.body) ?? response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable,\n unverifiableReason:\n response.unverifiableReason ?? (verifiable ? undefined : 'no publication body extracted'),\n },\n dimensionHints: IRS_DIMENSION_HINTS,\n metadata: {\n sourceId,\n status: response.status,\n fromCache: response.fromCache,\n kind: 'publication',\n slug,\n },\n }\n}\n\nasync function fetchRevenueProcedure(\n sourceId: string,\n path: string,\n opts: FetchOpts,\n): Promise<KnowledgeFragment> {\n const url = `${BASE_URL}${path.startsWith('/') ? path : `/${path}`}`\n const response = await politeFetch(url, { signal: opts.signal, cacheDir: opts.cacheDir })\n const body = extractMainContent(response.body)\n const verifiable = response.verifiable && body.length > 200\n return {\n id: `rev-proc:${path}`,\n title: extractTitle(response.body, `IRS Revenue Procedure ${path}`),\n body,\n bodyHash: sha256(body),\n provenance: {\n url,\n sourceUpdatedAt: response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable,\n unverifiableReason:\n response.unverifiableReason ??\n (verifiable ? undefined : 'no revenue-procedure body extracted'),\n },\n dimensionHints: [...IRS_DIMENSION_HINTS, 'procedural_currency'],\n metadata: {\n sourceId,\n status: response.status,\n fromCache: response.fromCache,\n kind: 'rev-proc',\n path,\n },\n }\n}\n\nfunction extractTitle(html: string, fallback: string): string {\n const og = /<meta\\s+property=[\"']og:title[\"']\\s+content=[\"']([^\"']+)[\"']/i.exec(html)?.[1]\n if (og) return decodeHtml(og)\n const title = /<title>([\\s\\S]*?)<\\/title>/i.exec(html)?.[1]\n if (title) return htmlToText(title).split(' | ')[0] ?? fallback\n return fallback\n}\n\nfunction extractMainContent(html: string): string {\n // IRS uses Drupal — the main publication body is inside <main role=\"main\">\n // or under .field--name-body. We try main first; on miss, body.\n const main = /<main\\b[\\s\\S]*?<\\/main>/i.exec(html)?.[0]\n if (main) {\n const noNav = main\n .replace(/<nav[\\s\\S]*?<\\/nav>/gi, '')\n .replace(/<header[\\s\\S]*?<\\/header>/gi, '')\n .replace(/<footer[\\s\\S]*?<\\/footer>/gi, '')\n return htmlToText(noNav).slice(0, 200_000)\n }\n const body = /<body\\b[\\s\\S]*?<\\/body>/i.exec(html)?.[0]\n return body ? htmlToText(body).slice(0, 200_000) : htmlToText(html).slice(0, 200_000)\n}\n\nfunction extractRevisionDate(html: string): string | undefined {\n // IRS publication pages typically show \"Publication X (YYYY)\" in the title;\n // pulling the year gives a stable revision marker.\n const m = /Publication\\s+\\S+\\s*\\((\\d{4})\\)/i.exec(html)\n if (m?.[1]) {\n const year = Number.parseInt(m[1], 10)\n if (Number.isFinite(year) && year >= 2000 && year <= new Date().getUTCFullYear() + 1) {\n return new Date(Date.UTC(year, 0, 1)).toISOString()\n }\n }\n return undefined\n}\n\nfunction decodeHtml(value: string): string {\n return htmlToText(value)\n}\n","import { sha256 } from '../ids'\nimport { htmlToText } from './html'\nimport { politeFetch } from './http'\nimport type { FetchOpts, KnowledgeFragment, KnowledgeSource } from './types'\n\n/**\n * Generic Secretary-of-State source.\n *\n * Every US state SOS surfaces LLC/Corp formation requirements differently\n * (CA via static forms pages, DE via division of corporations pages, TX\n * via SOSDirect content pages). Rather than baking 50 state-specific\n * parsers into this package, the source takes a config that names the URL\n * pattern + CSS-equivalent selector + jurisdiction tag. Callers supply one\n * config per state they need tracked.\n *\n * The selector is interpreted as a substring/regex of an HTML element id\n * or class — see `StateSosSourceConfig` for the contract. This is\n * intentionally minimal; richer extraction belongs in a state-specific\n * adapter the consumer authors.\n *\n * @experimental Interface will likely grow as we add more state coverage.\n */\n\nexport interface StateSosEntity {\n /** Stable id for this fragment within the state (e.g. 'llc-formation', 'corp-formation'). */\n id: string\n /** Path under the configured `baseUrl` for this entity. */\n path: string\n /**\n * Extraction selector. Choose one:\n * - `{ kind: 'id', value: 'main-content' }` — innermost match of element with that id\n * - `{ kind: 'class', value: 'field--name-body' }` — innermost match of element with that class\n * - `{ kind: 'regex', value: /<article[\\s\\S]*?<\\/article>/i }` — raw regex\n * - `{ kind: 'whole' }` — full body, tags stripped (fallback for unstructured pages)\n */\n selector:\n | { kind: 'id'; value: string }\n | { kind: 'class'; value: string }\n | { kind: 'regex'; value: RegExp }\n | { kind: 'whole' }\n title: string\n /** Eval dimensions this entity feeds. */\n dimensionHints?: string[]\n}\n\nexport interface StateSosSourceConfig {\n /** US state postal code, e.g. 'CA', 'DE', 'TX'. */\n state: string\n /** Base URL for the state SOS — e.g. 'https://www.sos.ca.gov'. */\n baseUrl: string\n /** Entities this state exposes (LLC, Corp, etc). */\n entities: StateSosEntity[]\n /** Source id; default `state-sos:<state>`. */\n id?: string\n /** Display name; default `<state> Secretary of State`. */\n name?: string\n}\n\nexport function createStateSosSource(config: StateSosSourceConfig): KnowledgeSource {\n const id = config.id ?? `state-sos:${config.state.toLowerCase()}`\n const name = config.name ?? `${config.state} Secretary of State`\n return {\n id,\n name,\n description: `${config.state} Secretary of State filings and formation guidance pages.`,\n async fetch(opts: FetchOpts): Promise<KnowledgeFragment[]> {\n const limit = opts.limit ?? config.entities.length\n const entities = config.entities.slice(0, limit)\n const out: KnowledgeFragment[] = []\n for (const entity of entities) {\n out.push(await fetchEntity(id, config, entity, opts))\n }\n return out\n },\n }\n}\n\nasync function fetchEntity(\n sourceId: string,\n config: StateSosSourceConfig,\n entity: StateSosEntity,\n opts: FetchOpts,\n): Promise<KnowledgeFragment> {\n const url = joinUrl(config.baseUrl, entity.path)\n const response = await politeFetch(url, { signal: opts.signal, cacheDir: opts.cacheDir })\n\n const body = response.verifiable ? extractBySelector(response.body, entity.selector) : ''\n const verifiable = response.verifiable && body.length > 100\n\n return {\n id: entity.id,\n title: entity.title,\n body,\n bodyHash: sha256(body),\n provenance: {\n url,\n sourceUpdatedAt: response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: `US-${config.state.toUpperCase()}`,\n verifiable,\n unverifiableReason:\n response.unverifiableReason ?? (verifiable ? undefined : 'extracted body too short'),\n },\n dimensionHints: entity.dimensionHints ?? [\n 'jurisdictional_accuracy',\n 'corporate_formation',\n 'citation_hygiene',\n ],\n metadata: {\n sourceId,\n status: response.status,\n fromCache: response.fromCache,\n state: config.state,\n },\n }\n}\n\nfunction extractBySelector(html: string, selector: StateSosEntity['selector']): string {\n if (selector.kind === 'whole') {\n const main = /<main\\b[\\s\\S]*?<\\/main>/i.exec(html)?.[0]\n return htmlToText(main ?? html).slice(0, 200_000)\n }\n if (selector.kind === 'regex') {\n const m = selector.value.exec(html)?.[0]\n return m ? htmlToText(m).slice(0, 200_000) : ''\n }\n if (selector.kind === 'id') {\n const escaped = selector.value.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n const pattern = new RegExp(\n `<([a-z][a-z0-9]*)\\\\b[^>]*\\\\sid=[\"']${escaped}[\"'][^>]*>([\\\\s\\\\S]*?)<\\\\/\\\\1>`,\n 'i',\n )\n const inner = pattern.exec(html)?.[2]\n return inner ? htmlToText(inner).slice(0, 200_000) : ''\n }\n const escaped = selector.value.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n const pattern = new RegExp(\n `<([a-z][a-z0-9]*)\\\\b[^>]*\\\\sclass=[\"'][^\"']*\\\\b${escaped}\\\\b[^\"']*[\"'][^>]*>([\\\\s\\\\S]*?)<\\\\/\\\\1>`,\n 'i',\n )\n const inner = pattern.exec(html)?.[2]\n return inner ? htmlToText(inner).slice(0, 200_000) : ''\n}\n\nfunction joinUrl(base: string, path: string): string {\n try {\n return new URL(path, base.endsWith('/') ? base : `${base}/`).toString()\n } catch {\n return `${base.replace(/\\/+$/, '')}/${path.replace(/^\\/+/, '')}`\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;AAqBA,SAAgB,WAAW,MAAsB;CAC/C,OAAO,KACJ,QAAQ,+BAA+B,EAAE,CAAC,CAC1C,QAAQ,6BAA6B,EAAE,CAAC,CACxC,QAAQ,mCAAmC,EAAE,CAAC,CAC9C,QAAQ,sBAAsB,EAAE,CAAC,CACjC,QAAQ,mBAAmB,IAAI,CAAC,CAChC,QAAQ,yDAAyD,IAAI,CAAC,CACtE,QAAQ,YAAY,EAAE,CAAC,CACvB,QAAQ,YAAY,GAAG,CAAC,CACxB,QAAQ,WAAW,GAAG,CAAC,CACvB,QAAQ,UAAU,GAAG,CAAC,CACtB,QAAQ,UAAU,GAAG,CAAC,CACtB,QAAQ,YAAY,IAAG,CAAC,CACxB,QAAQ,WAAW,GAAG,CAAC,CACvB,QAAQ,YAAY,GAAG,CAAC,CACxB,QAAQ,aAAa,GAAG,CAAC,CACzB,QAAQ,aAAa,GAAG,CAAC,CACzB,QAAQ,cAAc,GAAG,SAAS,OAAO,cAAc,OAAO,IAAI,CAAC,CAAC,CAAC,CACrE,QAAQ,sBAAsB,GAAG,SAAS,OAAO,cAAc,OAAO,SAAS,MAAM,EAAE,CAAC,CAAC,CAAC,CAC1F,MAAM,IAAI,CAAC,CACX,KAAK,SAAS,KAAK,QAAQ,YAAY,GAAG,CAAC,CAAC,KAAK,CAAC,CAAC,CACnD,QAAQ,MAAM,KAAK,QAAQ,EAAE,SAAS,MAAM,IAAI,MAAM,OAAO,GAAG,CAAC,CACjE,KAAK,IAAI,CAAC,CACV,KAAK;AACV;;AAGA,SAAgB,WAAW,MAAc,SAAqC;CAC5E,OAAO,QAAQ,KAAK,IAAI,CAAC,GAAG,EAAE,EAAE,KAAK;AACvC;;AAGA,SAAgB,cAAc,MAAc,IAAgC;CAC1E,MAAM,UAAU,GAAG,QAAQ,uBAAuB,MAAM;CAKxD,OAAO,IAJgB,OACrB,sCAAsC,QAAQ,iCAC9C,GAEc,CAAC,CAAC,KAAK,IAAI,CAAC,GAAG;AACjC;;;;;AAMA,SAAgB,aACd,MACA,aACA,SACkC;CAClC,MAAM,MAAwC,CAAC;CAE/C,KAAK,MAAM,SAAS,KAAK,SAAS,yDAAM,GAAG;EACzC,MAAM,OAAO,MAAM;EACnB,MAAM,QAAQ,MAAM;EACpB,IAAI,CAAC,QAAQ,CAAC,OAAO;EACrB,IAAI,CAAC,YAAY,KAAK,IAAI,GAAG;EAC7B,MAAM,OAAO,WAAW,KAAK;EAC7B,IAAI,CAAC,MAAM;EACX,IAAI;GACF,IAAI,KAAK;IAAE,MAAM,IAAI,IAAI,MAAM,OAAO,CAAC,CAAC,SAAS;IAAG;GAAK,CAAC;EAC5D,QAAQ,CAER;CACF;CACA,OAAO;AACT;;;;;;;;;;;;;ACzEA,MAAa,oBACX;;AAGF,MAAa,qBAAqB;;AAGlC,MAAa,qBAAqB,IAAI,OAAO;AAE7C,MAAM,+BAAe,IAAI,IAA2B;;;;;;;;;;AAiDpD,eAAsB,YACpB,KACA,UAA8B,CAAC,GACH;CAC5B,MAAM,WAAW,QAAQ,cAAc,OAAU;CACjD,MAAM,SAAS,QAAQ,WAAW,MAAM,UAAU,QAAQ,UAAU,KAAK,QAAQ,IAAI,KAAA;CACrF,IAAI,QAAQ,OAAO;CAEnB,MAAM,OAAO,SAAS,GAAG;CACzB,MAAM,aAAa,IAAI;CAEvB,MAAM,6BAAY,IAAI,KAAK,EAAA,CAAE,YAAY;CACzC,IAAI;CACJ,IAAI;EACF,WAAW,MAAM,MAAM,KAAK;GAC1B,QAAQ,QAAQ;GAChB,UAAU;GACV,SAAS;IACP,cAAc;IACd,QAAQ;IACR,mBAAmB;IACnB,GAAI,QAAQ,WAAW,CAAC;GAC1B;EACF,CAAC;CACH,SAAS,OAAO;EACd,IAAK,MAA4B,SAAS,cAAc,MAAM;EAC9D,MAAM,SAA4B;GAChC;GACA,QAAQ;GACR,MAAM;GACN,iBAAiB;GACjB;GACA,WAAW;GACX,YAAY;GACZ,oBAAoB,kBAAmB,MAAgB;EACzD;EACA,IAAI,QAAQ,UAAU,MAAM,WAAW,QAAQ,UAAU,KAAK,MAAM;EACpE,OAAO;CACT;CAEA,MAAM,OAAO,MAAM,gBAAgB,QAAQ;CAC3C,MAAM,eAAe,SAAS,QAAQ,IAAI,eAAe;CACzD,MAAM,aAAa,SAAS,QAAQ,IAAI,MAAM;CAC9C,MAAM,kBAAkB,cAAc,YAAY,KAAK,cAAc,UAAU,KAAK;CAEpF,MAAM,SAA4B;EAChC;EACA,QAAQ,SAAS;EACjB,MAAM;EACN;EACA;EACA,WAAW;EACX,YAAY;CACd;CAEA,IAAI,SAAS,SAAS,OAAO,SAAS,UAAU,KAAK;EACnD,OAAO,aAAa;EACpB,OAAO,qBAAqB,mBAAmB,SAAS;CAC1D,OAAO,IAAI,mBAAmB,IAAI,GAAG;EACnC,OAAO,aAAa;EACpB,OAAO,qBAAqB;CAC9B,OAAO,IAAI,KAAK,SAAS,OAAO,oBAAoB,IAAI,GAAG;EACzD,OAAO,aAAa;EACpB,OAAO,qBAAqB,+BAA+B,KAAK,OAAO;CACzE;CAEA,IAAI,QAAQ,UAAU,MAAM,WAAW,QAAQ,UAAU,KAAK,MAAM;CACpE,OAAO;AACT;;AAGA,SAAgB,sBAA4B;CAC1C,aAAa,MAAM;AACrB;AAEA,SAAS,SAAS,KAAqB;CACrC,IAAI;EACF,OAAO,IAAI,IAAI,GAAG,CAAC,CAAC;CACtB,QAAQ;EACN,OAAO;CACT;AACF;AAEA,eAAe,aAAa,MAA6B;CACvD,MAAM,OAAO,aAAa,IAAI,IAAI,KAAK,QAAQ,QAAQ;CACvD,IAAI,gBAA4B,CAAC;CACjC,MAAM,OAAO,IAAI,SAAe,YAAY;EAC1C,UAAU;CACZ,CAAC;CACD,aAAa,IACX,MACA,KAAK,WAAW,IAAI,CACtB;CACA,MAAM;CACN,WAAW,SAAS,kBAAkB;AACxC;AAEA,eAAe,gBAAgB,UAAqC;CAClE,IAAI,CAAC,SAAS,MAAM,OAAO;CAC3B,MAAM,SAAS,SAAS,KAAK,UAAU;CACvC,MAAM,SAAuB,CAAC;CAC9B,IAAI,QAAQ;CACZ,OAAO,MAAM;EACX,MAAM,EAAE,MAAM,UAAU,MAAM,OAAO,KAAK;EAC1C,IAAI,MAAM;EACV,IAAI,CAAC,OAAO;EACZ,SAAS,MAAM;EACf,IAAI,QAAA,SAA4B;GAE9B,MAAM,OAAO,OAAO;GACpB;EACF;EACA,OAAO,KAAK,KAAK;CACnB;CACA,MAAM,SAAS,IAAI,WAAW,KAAK,IAAI,OAAO,kBAAkB,CAAC;CACjE,IAAI,SAAS;CACb,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,OAAO,KAAK,IAAI,MAAM,QAAQ,OAAO,SAAS,MAAM;EAC1D,IAAI,QAAQ,GAAG;EACf,OAAO,IAAI,MAAM,SAAS,GAAG,IAAI,GAAG,MAAM;EAC1C,UAAU;CACZ;CACA,OAAO,IAAI,YAAY,SAAS,EAAE,OAAO,MAAM,CAAC,CAAC,CAAC,OAAO,MAAM;AACjE;AAEA,SAAS,cAAc,OAA0C;CAC/D,IAAI,CAAC,OAAO,OAAO,KAAA;CACnB,MAAM,KAAK,KAAK,MAAM,KAAK;CAC3B,OAAO,OAAO,SAAS,EAAE,IAAI,IAAI,KAAK,EAAE,CAAC,CAAC,YAAY,IAAI,KAAA;AAC5D;;AAGA,SAAgB,mBAAmB,MAAuB;CACxD,IAAI,CAAC,MAAM,OAAO;CAClB,MAAM,QAAQ,KAAK,YAAY;CAY/B,KAAK,MAAM,UAAU;EAVnB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CAEyB,GACzB,IAAI,MAAM,SAAS,MAAM,GAAG,OAAO;CAErC,OAAO;AACT;AAEA,SAAS,oBAAoB,MAAuB;CAClD,OACE,KAAK,SAAS,iBAAiB,KAC/B,KAAK,SAAS,SAAS,KACvB,KAAK,SAAS,YAAY,KAC1B,KAAK,SAAS,iBAAiB,KAC/B,KAAK,SAAS,cAAc;AAEhC;AAEA,SAAS,UAAU,UAAkB,KAAqB;CACxD,MAAM,MAAM,OAAO,GAAG;CACtB,OAAO,KAAK,UAAU,QAAQ,GAAG,IAAI,MAAM,GAAG,CAAC,KAAK,GAAG,IAAI,MAAM;AACnE;AAEA,eAAe,UACb,UACA,KACA,OACwC;CACxC,MAAM,OAAO,UAAU,UAAU,GAAG;CACpC,IAAI;EACF,MAAM,OAAO,MAAM,KAAK,IAAI;EAC5B,IAAI,KAAK,IAAI,IAAI,KAAK,UAAU,OAAO,OAAO,KAAA;EAC9C,MAAM,MAAM,MAAM,SAAS,MAAM,MAAM;EAEvC,OAAO;GAAE,GADM,KAAK,MAAM,GACT;GAAG,WAAW;EAAK;CACtC,QAAQ;EACN;CACF;AACF;AAEA,eAAe,WAAW,UAAkB,KAAa,OAAyC;CAChG,MAAM,OAAO,UAAU,UAAU,GAAG;CACpC,MAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAC9C,MAAM,UAAU,MAAM,KAAK,UAAU,KAAK,GAAG,MAAM;AACrD;;;;;;;;;;;;;ACrPA,MAAMA,aAAW;;;;;;;;;;;;;;AA0CjB,SAAgB,uBAAuB,SAAmD;CACxF,MAAM,KAAK,QAAQ,MAAM;CACzB,OAAO;EACL;EACA,MAAM;EACN,aACE;EACF,MAAM,MAAM,MAA+C;GACzD,MAAM,QAAQ,KAAK,SAAS,QAAQ,UAAU;GAC9C,MAAM,YAAY,QAAQ,UAAU,MAAM,GAAG,KAAK;GAClD,MAAM,MAA2B,CAAC;GAClC,KAAK,MAAM,YAAY,WACrB,IAAI,KAAK,MAAM,SAAS,IAAI,UAAU,IAAI,CAAC;GAE7C,OAAO;EACT;CACF;AACF;AAEA,eAAe,SACb,UACA,UACA,MAC4B;CAC5B,MAAM,OAAO,SAAS,KAAK,QAAQ,QAAQ,EAAE;CAC7C,MAAM,MACJ,SAAS,SAAS,WAAW,GAAGA,WAAS,eAAe,SAAS,GAAGA,WAAS,OAAO;CAEtF,MAAM,WAAW,MAAM,YAAY,KAAK;EACtC,QAAQ,KAAK;EACb,UAAU,KAAK;CACjB,CAAC;CAED,MAAM,aAAa,GAAG,SAAS,KAAK,GAAG,SAAS;CAChD,MAAM,iBAAiB,SAAS,kBAAkB,sBAAsB,QAAQ;CAEhF,IAAI,CAAC,SAAS,YACZ,OAAO;EACL,IAAI;EACJ,OAAO,eAAe,SAAS,KAAK,GAAG,SAAS;EAChD,MAAM;EACN,UAAU,OAAO,EAAE;EACnB,YAAY;GACV;GACA,iBAAiB,SAAS;GAC1B,WAAW,SAAS;GACpB,cAAc;GACd,YAAY;GACZ,oBAAoB,SAAS;EAC/B;EACA;EACA,UAAU;GAAE;GAAU,QAAQ,SAAS;GAAQ,WAAW,SAAS;EAAU;CAC/E;CAGF,MAAM,OAAO,SAAS;CACtB,MAAM,QAAQC,eAAa,MAAM,QAAQ;CACzC,MAAM,OAAO,YAAY,MAAM,QAAQ;CACvC,MAAM,YAAY,qBAAqB,IAAI,KAAK,SAAS;CAEzD,MAAM,aAAa,KAAK,SAAS;CACjC,OAAO;EACL,IAAI;EACJ;EACA;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV;GACA,iBAAiB;GACjB,WAAW,SAAS;GACpB,cAAc;GACd;GACA,oBAAoB,aAAa,KAAA,IAAY;EAC/C;EACA;EACA,UAAU;GAAE;GAAU,QAAQ,SAAS;GAAQ,WAAW,SAAS;EAAU;CAC/E;AACF;AAEA,SAASA,eAAa,MAAc,UAAsC;CACxE,MAAM,KAAK,yDAAyD,KAAK,IAAI,CAAC,GAAG;CACjF,IAAI,IAAI,OAAO,WAAW,EAAE;CAC5B,MAAM,IAAI,8BAA8B,KAAK,IAAI,CAAC,GAAG;CACrD,IAAI,GAAG,OAAO,WAAW,CAAC,CAAC,CAAC,MAAM,KAAK,CAAC,CAAC,MAAM,eAAe,SAAS;CACvE,OAAO,eAAe,SAAS,KAAK,GAAG,SAAS;AAClD;AAEA,SAAS,YAAY,MAAc,UAAsC;CACvE,IAAI,SAAS,SAAS,UAAU;EAI9B,MAAM,OAAO,4BAA4B,KAAK,IAAI,CAAC,GAAG;EACtD,IAAI,MAAM,OAAO,WAAW,IAAI;EAChC,MAAM,MAAM,cAAc,MAAM,eAAe;EAC/C,IAAI,KAAK,OAAO,WAAW,GAAG;CAChC;CAKA,MAAM,cAAc,cAAc,MAAM,cAAc;CACtD,IAAI,aACF,OAAO,WAAW,YAAY,QAAQ,sBAAsB,EAAE,CAAC;CAEjE,MAAM,YAAY,cAAc,MAAM,mBAAmB;CACzD,IAAI,WACF,OAAO,WAAW,UAAU,QAAQ,sBAAsB,EAAE,CAAC;CAE/D,OAAO,WAAW,IAAI;AACxB;AAEA,SAAS,qBAAqB,MAAkC;CAI9D,MAAM,QAAQ,mCAAmC,KAAK,IAAI,CAAC,GAAG;CAC9D,IAAI,OAAO;EACT,MAAM,IAAI,OAAO,SAAS,OAAO,EAAE;EACnC,IAAI,OAAO,SAAS,CAAC,KAAK,IAAI,QAAQ,sBAAK,IAAI,KAAK,EAAA,CAAE,eAAe,IAAI,GACvE,OAAO,IAAI,KAAK,KAAK,IAAI,GAAG,IAAI,EAAE,CAAC,CAAC,CAAC,YAAY;CAErD;AAEF;AAEA,SAAS,sBAAsB,UAAwC;CACrE,IAAI,SAAS,SAAS,UAAU,OAAO,CAAC,2BAA2B,kBAAkB;CACrF,OAAO,CAAC,kBAAkB;AAC5B;;;;;;;;;;;;;;;;;;;;;;;ACjKA,MAAM,WAAW;AACjB,MAAM,YAAY,GAAG,SAAS;;AAoB9B,MAAa,sBAAsB;CAAC;CAAkB;CAAuB;AAAkB;AAE/F,SAAgB,4BACd,UAAwC,CAAC,GACxB;CACjB,MAAM,KAAK,QAAQ,MAAM;CACzB,MAAM,eAAe,QAAQ,gBAAgB;CAC7C,OAAO;EACL;EACA,MAAM;EACN,aACE;EACF,MAAM,MAAM,MAA+C;GACzD,MAAM,MAA2B,CAAC;GAClC,MAAM,QAAQ,KAAK,SAAS,OAAO;GAEnC,IAAI,gBAAgB,IAAI,SAAS,OAC/B,IAAI,KAAK,MAAM,WAAW,IAAI,IAAI,CAAC;GAErC,KAAK,MAAM,QAAQ,QAAQ,gBAAgB,CAAC,GAAG;IAC7C,IAAI,IAAI,UAAU,OAAO;IACzB,IAAI,KAAK,MAAM,iBAAiB,IAAI,MAAM,IAAI,CAAC;GACjD;GACA,KAAK,MAAM,QAAQ,QAAQ,qBAAqB,CAAC,GAAG;IAClD,IAAI,IAAI,UAAU,OAAO;IACzB,IAAI,KAAK,MAAM,sBAAsB,IAAI,MAAM,IAAI,CAAC;GACtD;GACA,OAAO;EACT;CACF;AACF;AAEA,eAAe,WAAW,UAAkB,MAA6C;CACvF,MAAM,WAAW,MAAM,YAAY,WAAW;EAAE,QAAQ,KAAK;EAAQ,UAAU,KAAK;CAAS,CAAC;CAO9F,MAAM,QALU,SAAS,KAAK,MAAM,2BAAY,KAAK,CAAC,EAAA,CAI/B,KAAK,MAAM,WAAW,CAAC,CAC5B,CAAC,CAChB,QAAQ,MAAM,qBAAqB,KAAK,CAAC,CAAC,CAAC,CAC3C,KAAK,MAAM,CAAC,CACZ,MAAM,GAAG,GAAO;CAEnB,MAAM,aAAa,SAAS,cAAc,KAAK,SAAS;CACxD,OAAO;EACL,IAAI;EACJ,OAAO;EACP;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV,KAAK;GACL,iBAAiB,SAAS;GAC1B,WAAW,SAAS;GACpB,cAAc;GACd;GACA,oBACE,SAAS,uBAAuB,aAAa,KAAA,IAAY;EAC7D;EACA,gBAAgB;EAChB,UAAU;GAAE;GAAU,QAAQ,SAAS;GAAQ,WAAW,SAAS;GAAW,MAAM;EAAQ;CAC9F;AACF;AAEA,eAAe,iBACb,UACA,MACA,MAC4B;CAC5B,MAAM,MAAM,GAAG,SAAS,gBAAgB,KAAK,QAAQ,QAAQ,EAAE;CAC/D,MAAM,WAAW,MAAM,YAAY,KAAK;EAAE,QAAQ,KAAK;EAAQ,UAAU,KAAK;CAAS,CAAC;CAExF,MAAM,QAAQ,aAAa,SAAS,MAAM,mBAAmB,MAAM;CACnE,MAAM,OAAO,mBAAmB,SAAS,IAAI;CAC7C,MAAM,aAAa,SAAS,cAAc,KAAK,SAAS;CAExD,OAAO;EACL,IAAI,eAAe;EACnB;EACA;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV;GACA,iBAAiB,oBAAoB,SAAS,IAAI,KAAK,SAAS;GAChE,WAAW,SAAS;GACpB,cAAc;GACd;GACA,oBACE,SAAS,uBAAuB,aAAa,KAAA,IAAY;EAC7D;EACA,gBAAgB;EAChB,UAAU;GACR;GACA,QAAQ,SAAS;GACjB,WAAW,SAAS;GACpB,MAAM;GACN;EACF;CACF;AACF;AAEA,eAAe,sBACb,UACA,MACA,MAC4B;CAC5B,MAAM,MAAM,GAAG,WAAW,KAAK,WAAW,GAAG,IAAI,OAAO,IAAI;CAC5D,MAAM,WAAW,MAAM,YAAY,KAAK;EAAE,QAAQ,KAAK;EAAQ,UAAU,KAAK;CAAS,CAAC;CACxF,MAAM,OAAO,mBAAmB,SAAS,IAAI;CAC7C,MAAM,aAAa,SAAS,cAAc,KAAK,SAAS;CACxD,OAAO;EACL,IAAI,YAAY;EAChB,OAAO,aAAa,SAAS,MAAM,yBAAyB,MAAM;EAClE;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV;GACA,iBAAiB,SAAS;GAC1B,WAAW,SAAS;GACpB,cAAc;GACd;GACA,oBACE,SAAS,uBACR,aAAa,KAAA,IAAY;EAC9B;EACA,gBAAgB,CAAC,GAAG,qBAAqB,qBAAqB;EAC9D,UAAU;GACR;GACA,QAAQ,SAAS;GACjB,WAAW,SAAS;GACpB,MAAM;GACN;EACF;CACF;AACF;AAEA,SAAS,aAAa,MAAc,UAA0B;CAC5D,MAAM,KAAK,gEAAgE,KAAK,IAAI,CAAC,GAAG;CACxF,IAAI,IAAI,OAAO,WAAW,EAAE;CAC5B,MAAM,QAAQ,8BAA8B,KAAK,IAAI,CAAC,GAAG;CACzD,IAAI,OAAO,OAAO,WAAW,KAAK,CAAC,CAAC,MAAM,KAAK,CAAC,CAAC,MAAM;CACvD,OAAO;AACT;AAEA,SAAS,mBAAmB,MAAsB;CAGhD,MAAM,OAAO,2BAA2B,KAAK,IAAI,CAAC,GAAG;CACrD,IAAI,MAKF,OAAO,WAJO,KACX,QAAQ,yBAAyB,EAAE,CAAC,CACpC,QAAQ,+BAA+B,EAAE,CAAC,CAC1C,QAAQ,+BAA+B,EACpB,CAAC,CAAC,CAAC,MAAM,GAAG,GAAO;CAE3C,MAAM,OAAO,2BAA2B,KAAK,IAAI,CAAC,GAAG;CACrD,OAAO,OAAO,WAAW,IAAI,CAAC,CAAC,MAAM,GAAG,GAAO,IAAI,WAAW,IAAI,CAAC,CAAC,MAAM,GAAG,GAAO;AACtF;AAEA,SAAS,oBAAoB,MAAkC;CAG7D,MAAM,IAAI,mCAAmC,KAAK,IAAI;CACtD,IAAI,IAAI,IAAI;EACV,MAAM,OAAO,OAAO,SAAS,EAAE,IAAI,EAAE;EACrC,IAAI,OAAO,SAAS,IAAI,KAAK,QAAQ,OAAQ,yBAAQ,IAAI,KAAK,EAAA,CAAE,eAAe,IAAI,GACjF,OAAO,IAAI,KAAK,KAAK,IAAI,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,YAAY;CAEtD;AAEF;AAEA,SAAS,WAAW,OAAuB;CACzC,OAAO,WAAW,KAAK;AACzB;;;ACpKA,SAAgB,qBAAqB,QAA+C;CAClF,MAAM,KAAK,OAAO,MAAM,aAAa,OAAO,MAAM,YAAY;CAE9D,OAAO;EACL;EACA,MAHW,OAAO,QAAQ,GAAG,OAAO,MAAM;EAI1C,aAAa,GAAG,OAAO,MAAM;EAC7B,MAAM,MAAM,MAA+C;GACzD,MAAM,QAAQ,KAAK,SAAS,OAAO,SAAS;GAC5C,MAAM,WAAW,OAAO,SAAS,MAAM,GAAG,KAAK;GAC/C,MAAM,MAA2B,CAAC;GAClC,KAAK,MAAM,UAAU,UACnB,IAAI,KAAK,MAAM,YAAY,IAAI,QAAQ,QAAQ,IAAI,CAAC;GAEtD,OAAO;EACT;CACF;AACF;AAEA,eAAe,YACb,UACA,QACA,QACA,MAC4B;CAC5B,MAAM,MAAM,QAAQ,OAAO,SAAS,OAAO,IAAI;CAC/C,MAAM,WAAW,MAAM,YAAY,KAAK;EAAE,QAAQ,KAAK;EAAQ,UAAU,KAAK;CAAS,CAAC;CAExF,MAAM,OAAO,SAAS,aAAa,kBAAkB,SAAS,MAAM,OAAO,QAAQ,IAAI;CACvF,MAAM,aAAa,SAAS,cAAc,KAAK,SAAS;CAExD,OAAO;EACL,IAAI,OAAO;EACX,OAAO,OAAO;EACd;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV;GACA,iBAAiB,SAAS;GAC1B,WAAW,SAAS;GACpB,cAAc,MAAM,OAAO,MAAM,YAAY;GAC7C;GACA,oBACE,SAAS,uBAAuB,aAAa,KAAA,IAAY;EAC7D;EACA,gBAAgB,OAAO,kBAAkB;GACvC;GACA;GACA;EACF;EACA,UAAU;GACR;GACA,QAAQ,SAAS;GACjB,WAAW,SAAS;GACpB,OAAO,OAAO;EAChB;CACF;AACF;AAEA,SAAS,kBAAkB,MAAc,UAA8C;CACrF,IAAI,SAAS,SAAS,SAAS;EAC7B,MAAM,OAAO,2BAA2B,KAAK,IAAI,CAAC,GAAG;EACrD,OAAO,WAAW,QAAQ,IAAI,CAAC,CAAC,MAAM,GAAG,GAAO;CAClD;CACA,IAAI,SAAS,SAAS,SAAS;EAC7B,MAAM,IAAI,SAAS,MAAM,KAAK,IAAI,CAAC,GAAG;EACtC,OAAO,IAAI,WAAW,CAAC,CAAC,CAAC,MAAM,GAAG,GAAO,IAAI;CAC/C;CACA,IAAI,SAAS,SAAS,MAAM;EAC1B,MAAM,UAAU,SAAS,MAAM,QAAQ,uBAAuB,MAAM;EAKpE,MAAM,QAAQ,IAJM,OAClB,sCAAsC,QAAQ,iCAC9C,GAEkB,CAAC,CAAC,KAAK,IAAI,CAAC,GAAG;EACnC,OAAO,QAAQ,WAAW,KAAK,CAAC,CAAC,MAAM,GAAG,GAAO,IAAI;CACvD;CACA,MAAM,UAAU,SAAS,MAAM,QAAQ,uBAAuB,MAAM;CAKpE,MAAM,QAAQ,IAJM,OAClB,kDAAkD,QAAQ,0CAC1D,GAEkB,CAAC,CAAC,KAAK,IAAI,CAAC,GAAG;CACnC,OAAO,QAAQ,WAAW,KAAK,CAAC,CAAC,MAAM,GAAG,GAAO,IAAI;AACvD;AAEA,SAAS,QAAQ,MAAc,MAAsB;CACnD,IAAI;EACF,OAAO,IAAI,IAAI,MAAM,KAAK,SAAS,GAAG,IAAI,OAAO,GAAG,KAAK,EAAE,CAAC,CAAC,SAAS;CACxE,QAAQ;EACN,OAAO,GAAG,KAAK,QAAQ,QAAQ,EAAE,EAAE,GAAG,KAAK,QAAQ,QAAQ,EAAE;CAC/D;AACF"}
1
+ {"version":3,"file":"index.js","names":["BASE_URL","extractTitle"],"sources":["../../src/sources/html.ts","../../src/sources/http.ts","../../src/sources/cornell-lii.ts","../../src/sources/irs-publications.ts","../../src/sources/state-sos.ts"],"sourcesContent":["/**\n * Minimal HTML helpers used by the shipped sources.\n *\n * Deliberately not a full DOM parser: every authority we ship against\n * (Cornell LII, IRS.gov, state SOS portals) has well-behaved server-rendered\n * HTML where regex-based extraction is correct and cheap. Bringing in cheerio\n * would add a 1.5MB dependency to a package whose purpose is shipping\n * primitives, not parsing arbitrary web pages.\n *\n * If a future source needs real DOM traversal, it should depend on its own\n * parser locally rather than promoting one into the package-wide deps.\n *\n * @stable\n */\n\n/**\n * Strip HTML tags, collapse whitespace, decode common entities.\n *\n * Preserves paragraph and line breaks (`</p>`, `<br>`, `</li>`, `</div>`,\n * `</h*>`) as `\\n` so statute text retains its subsection structure.\n */\nexport function htmlToText(html: string): string {\n return html\n .replace(/<script[\\s\\S]*?<\\/script>/gi, '')\n .replace(/<style[\\s\\S]*?<\\/style>/gi, '')\n .replace(/<noscript[\\s\\S]*?<\\/noscript>/gi, '')\n .replace(/<!--([\\s\\S]*?)-->/g, '')\n .replace(/<\\s*br\\s*\\/?>/gi, '\\n')\n .replace(/<\\/(p|li|div|tr|h[1-6]|blockquote|section|article)>/gi, '\\n')\n .replace(/<[^>]+>/g, '')\n .replace(/&nbsp;/gi, ' ')\n .replace(/&amp;/gi, '&')\n .replace(/&lt;/gi, '<')\n .replace(/&gt;/gi, '>')\n .replace(/&quot;/gi, '\"')\n .replace(/&#39;/gi, \"'\")\n .replace(/&sect;/gi, '§')\n .replace(/&mdash;/gi, '—')\n .replace(/&ndash;/gi, '–')\n .replace(/&#(\\d+);/g, (_, code) => String.fromCodePoint(Number(code)))\n .replace(/&#x([0-9a-f]+);/gi, (_, code) => String.fromCodePoint(Number.parseInt(code, 16)))\n .split('\\n')\n .map((line) => line.replace(/[\\t  ]+/g, ' ').trim())\n .filter((line, idx, all) => !(line === '' && all[idx - 1] === ''))\n .join('\\n')\n .trim()\n}\n\n/** Extract the first match of a regex's first capture group, or undefined. */\nexport function firstMatch(html: string, pattern: RegExp): string | undefined {\n return pattern.exec(html)?.[1]?.trim()\n}\n\n/** Extract the inner HTML of the first matching tag with id `id`. */\nexport function innerHtmlById(html: string, id: string): string | undefined {\n const escaped = id.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n const tagPattern = new RegExp(\n `<([a-z][a-z0-9]*)\\\\b[^>]*\\\\sid=[\"']${escaped}[\"'][^>]*>([\\\\s\\\\S]*?)<\\\\/\\\\1>`,\n 'i',\n )\n return tagPattern.exec(html)?.[2]\n}\n\n/**\n * Extract every (href, text) pair matching the URL regex.\n * Returns absolute URLs by resolving against `baseUrl`.\n */\nexport function extractLinks(\n html: string,\n hrefPattern: RegExp,\n baseUrl: string,\n): { href: string; text: string }[] {\n const out: { href: string; text: string }[] = []\n const anchor = /<a\\b[^>]*\\shref=[\"']([^\"']+)[\"'][^>]*>([\\s\\S]*?)<\\/a>/gi\n for (const match of html.matchAll(anchor)) {\n const href = match[1]\n const inner = match[2]\n if (!href || !inner) continue\n if (!hrefPattern.test(href)) continue\n const text = htmlToText(inner)\n if (!text) continue\n try {\n out.push({ href: new URL(href, baseUrl).toString(), text })\n } catch {\n /* skip malformed URL */\n }\n }\n return out\n}\n","import { mkdir, readFile, stat, writeFile } from 'node:fs/promises'\nimport { dirname, join } from 'node:path'\nimport { sha256 } from '../ids'\n\n/**\n * Polite HTTP fetcher shared by remote sources.\n *\n * Independent sources share a per-origin throttle because rate-limited sites\n * may return block pages instead of 429 responses. Responses are cached by URL\n * because many publishers omit reliable ETag and Last-Modified headers. Bodies\n * are checked even after a 2xx response because captcha and block pages often\n * use successful status codes.\n */\n\n/** User-Agent string sent on every outbound request. */\nexport const POLITE_USER_AGENT =\n 'agent-knowledge (+https://github.com/tangle-network/agent-knowledge)'\n\n/** Minimum gap between successive requests to the same origin (ms). */\nexport const MIN_REQUEST_GAP_MS = 1_000\n\n/** Maximum response body we will buffer in memory (bytes). */\nexport const MAX_RESPONSE_BYTES = 8 * 1024 * 1024\n\nconst hostThrottle = new Map<string, Promise<void>>()\n\nexport interface PoliteFetchOptions {\n signal?: AbortSignal\n cacheDir?: string\n /**\n * Cache age beyond which we re-fetch. Default 1 hour, long enough to\n * batch a cron sweep across many selectors, short enough that hourly\n * authoritative-page changes get picked up next tick.\n */\n cacheTtlMs?: number\n /**\n * Extra request headers. The fetcher always sets `User-Agent` and\n * `Accept`; callers can add `Accept-Language` etc.\n */\n headers?: Record<string, string>\n}\n\nexport interface PoliteFetchResult {\n url: string\n status: number\n /** Decoded UTF-8 body. Truncated to `MAX_RESPONSE_BYTES`. */\n body: string\n /**\n * Best-effort source-attested timestamp. Reads `Last-Modified`,\n * falling back to `Date`, falling back to fetch time. Always ISO 8601.\n */\n sourceUpdatedAt: string\n fetchedAt: string\n /** True iff the response was satisfied from disk cache. */\n fromCache: boolean\n /**\n * False on: non-2xx status, captcha/block page heuristic match, or\n * decoded body below 200 chars from a host known to serve real content\n * (Cornell, IRS, state SOS). `unverifiableReason` carries the why.\n */\n verifiable: boolean\n unverifiableReason?: string\n}\n\n/**\n * Fetch one URL with per-host throttling, on-disk cache, and block-page\n * detection. Never throws on network/HTTP failure. It returns a result with\n * `verifiable: false` and `unverifiableReason` set so the caller can decide\n * whether to skip, retry, or surface.\n *\n * Throws ONLY on `AbortError` (caller asked to stop) and on cache-write\n * failures that indicate a misconfigured filesystem.\n */\nexport async function politeFetch(\n url: string,\n options: PoliteFetchOptions = {},\n): Promise<PoliteFetchResult> {\n const cacheTtl = options.cacheTtlMs ?? 60 * 60 * 1000\n const cached = options.cacheDir ? await readCache(options.cacheDir, url, cacheTtl) : undefined\n if (cached) return cached\n\n const host = safeHost(url)\n await throttleHost(host)\n\n const fetchedAt = new Date().toISOString()\n let response: Response\n try {\n response = await fetch(url, {\n signal: options.signal,\n redirect: 'follow',\n headers: {\n 'User-Agent': POLITE_USER_AGENT,\n Accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',\n 'Accept-Language': 'en-US,en;q=0.9',\n ...(options.headers ?? {}),\n },\n })\n } catch (error) {\n if ((error as { name?: string }).name === 'AbortError') throw error\n const result: PoliteFetchResult = {\n url,\n status: 0,\n body: '',\n sourceUpdatedAt: fetchedAt,\n fetchedAt,\n fromCache: false,\n verifiable: false,\n unverifiableReason: `network error: ${(error as Error).message}`,\n }\n if (options.cacheDir) await writeCache(options.cacheDir, url, result)\n return result\n }\n\n const text = await readBoundedText(response)\n const lastModified = response.headers.get('last-modified')\n const dateHeader = response.headers.get('date')\n const sourceUpdatedAt = parseHttpDate(lastModified) ?? parseHttpDate(dateHeader) ?? fetchedAt\n\n const result: PoliteFetchResult = {\n url,\n status: response.status,\n body: text,\n sourceUpdatedAt,\n fetchedAt,\n fromCache: false,\n verifiable: true,\n }\n\n if (response.status < 200 || response.status >= 300) {\n result.verifiable = false\n result.unverifiableReason = `non-2xx status: ${response.status}`\n } else if (looksLikeBlockPage(text)) {\n result.verifiable = false\n result.unverifiableReason = 'block-page heuristic matched'\n } else if (text.length < 200 && knownLargeAuthority(host)) {\n result.verifiable = false\n result.unverifiableReason = `body shorter than expected (${text.length} chars)`\n }\n\n if (options.cacheDir) await writeCache(options.cacheDir, url, result)\n return result\n}\n\n/** Reset the in-process throttle map. Test-only. */\nexport function __resetHttpThrottle(): void {\n hostThrottle.clear()\n}\n\nfunction safeHost(url: string): string {\n try {\n return new URL(url).host\n } catch {\n return 'unknown'\n }\n}\n\nasync function throttleHost(host: string): Promise<void> {\n const prev = hostThrottle.get(host) ?? Promise.resolve()\n let release: () => void = () => {}\n const next = new Promise<void>((resolve) => {\n release = resolve\n })\n hostThrottle.set(\n host,\n prev.then(() => next),\n )\n await prev\n setTimeout(release, MIN_REQUEST_GAP_MS)\n}\n\nasync function readBoundedText(response: Response): Promise<string> {\n if (!response.body) return ''\n const reader = response.body.getReader()\n const chunks: Uint8Array[] = []\n let total = 0\n while (true) {\n const { done, value } = await reader.read()\n if (done) break\n if (!value) continue\n total += value.length\n if (total > MAX_RESPONSE_BYTES) {\n // Stop reading; release the underlying connection.\n await reader.cancel()\n break\n }\n chunks.push(value)\n }\n const merged = new Uint8Array(Math.min(total, MAX_RESPONSE_BYTES))\n let offset = 0\n for (const chunk of chunks) {\n const take = Math.min(chunk.length, merged.length - offset)\n if (take <= 0) break\n merged.set(chunk.subarray(0, take), offset)\n offset += take\n }\n return new TextDecoder('utf-8', { fatal: false }).decode(merged)\n}\n\nfunction parseHttpDate(value: string | null): string | undefined {\n if (!value) return undefined\n const ms = Date.parse(value)\n return Number.isFinite(ms) ? new Date(ms).toISOString() : undefined\n}\n\n/** Cheap heuristic that catches CAPTCHA, WAF block pages, and \"Just a moment\" interstitials. */\nexport function looksLikeBlockPage(body: string): boolean {\n if (!body) return false\n const lower = body.toLowerCase()\n const markers = [\n 'verify you are human',\n 'please enable javascript and cookies',\n 'just a moment',\n 'access denied',\n 'request unsuccessful',\n 'cf-error-details',\n 'captcha',\n 'incapsula',\n 'pardon our interruption',\n ]\n for (const marker of markers) {\n if (lower.includes(marker)) return true\n }\n return false\n}\n\nfunction knownLargeAuthority(host: string): boolean {\n return (\n host.endsWith('law.cornell.edu') ||\n host.endsWith('irs.gov') ||\n host.endsWith('sos.ca.gov') ||\n host.endsWith('sos.state.tx.us') ||\n host.endsWith('sos.state.us')\n )\n}\n\nfunction cachePath(cacheDir: string, url: string): string {\n const key = sha256(url)\n return join(cacheDir, 'http', `${key.slice(0, 2)}`, `${key}.json`)\n}\n\nasync function readCache(\n cacheDir: string,\n url: string,\n ttlMs: number,\n): Promise<PoliteFetchResult | undefined> {\n const path = cachePath(cacheDir, url)\n try {\n const info = await stat(path)\n if (Date.now() - info.mtimeMs > ttlMs) return undefined\n const raw = await readFile(path, 'utf8')\n const parsed = JSON.parse(raw) as PoliteFetchResult\n return { ...parsed, fromCache: true }\n } catch {\n return undefined\n }\n}\n\nasync function writeCache(cacheDir: string, url: string, value: PoliteFetchResult): Promise<void> {\n const path = cachePath(cacheDir, url)\n await mkdir(dirname(path), { recursive: true })\n await writeFile(path, JSON.stringify(value), 'utf8')\n}\n","import { sha256 } from '../ids'\nimport { htmlToText, innerHtmlById } from './html'\nimport { politeFetch } from './http'\nimport type { FetchOpts, KnowledgeFragment, KnowledgeSource } from './types'\n\n/**\n * Cornell Legal Information Institute (LII) source.\n *\n * Pulls federal US Code sections and Wex encyclopedia entries — the two\n * Cornell LII surfaces an agent typically grounds against. The Wex\n * \"non-compete\" page is the canonical test case for the Ryan-LLC v. FTC\n * vacatur drift the continuous-ingestion story is designed to catch.\n *\n * @stable\n */\n\nconst BASE_URL = 'https://www.law.cornell.edu'\n\nexport interface CornellLiiSelector {\n /** Either 'uscode' or 'wex'. */\n kind: 'uscode' | 'wex'\n /**\n * For `uscode`: `<title>/<section>` (e.g. `'18/1836'` for DTSA).\n * For `wex`: the slug (e.g. `'non-compete'`).\n */\n path: string\n /**\n * Optional pre-declared eval dimensions affected by this section. If\n * omitted, defaults are chosen from `kind` + path heuristics.\n */\n dimensionHints?: string[]\n}\n\nexport interface CornellLiiSourceOptions {\n /**\n * Selectors to fetch on each `fetch()` call. The caller (a per-tenant\n * workspace config, typically) lists exactly the authorities they need\n * tracked. There is no auto-discovery; that would crawl Cornell at\n * cron speed, which is what the polite-fetch contract exists to avoid.\n */\n selectors: CornellLiiSelector[]\n /** Source id override; default is `'cornell-lii'`. */\n id?: string\n}\n\n/**\n * Build a Cornell LII source for the listed selectors.\n *\n * Example: track DTSA + non-compete:\n * ```\n * createCornellLiiSource({\n * selectors: [\n * { kind: 'uscode', path: '18/1836' },\n * { kind: 'wex', path: 'non-compete', dimensionHints: ['jurisdictional_accuracy'] },\n * ],\n * })\n * ```\n */\nexport function createCornellLiiSource(options: CornellLiiSourceOptions): KnowledgeSource {\n const id = options.id ?? 'cornell-lii'\n return {\n id,\n name: 'Cornell Legal Information Institute',\n description:\n 'Federal US Code sections (uscode/text/...) and Wex legal encyclopedia entries from law.cornell.edu.',\n async fetch(opts: FetchOpts): Promise<KnowledgeFragment[]> {\n const limit = opts.limit ?? options.selectors.length\n const selectors = options.selectors.slice(0, limit)\n const out: KnowledgeFragment[] = []\n for (const selector of selectors) {\n out.push(await fetchOne(id, selector, opts))\n }\n return out\n },\n }\n}\n\nasync function fetchOne(\n sourceId: string,\n selector: CornellLiiSelector,\n opts: FetchOpts,\n): Promise<KnowledgeFragment> {\n const path = selector.path.replace(/^\\/+/, '')\n const url =\n selector.kind === 'uscode' ? `${BASE_URL}/uscode/text/${path}` : `${BASE_URL}/wex/${path}`\n\n const response = await politeFetch(url, {\n signal: opts.signal,\n cacheDir: opts.cacheDir,\n })\n\n const fragmentId = `${selector.kind}:${selector.path}`\n const dimensionHints = selector.dimensionHints ?? defaultDimensionHints(selector)\n\n if (!response.verifiable) {\n return {\n id: fragmentId,\n title: `Cornell LII ${selector.kind} ${selector.path}`,\n body: '',\n bodyHash: sha256(''),\n provenance: {\n url,\n sourceUpdatedAt: response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable: false,\n unverifiableReason: response.unverifiableReason,\n },\n dimensionHints,\n metadata: { sourceId, status: response.status, fromCache: response.fromCache },\n }\n }\n\n const html = response.body\n const title = extractTitle(html, selector)\n const body = extractBody(html, selector)\n const effective = extractEffectiveDate(html) ?? response.sourceUpdatedAt\n\n const verifiable = body.length > 50\n return {\n id: fragmentId,\n title,\n body,\n bodyHash: sha256(body),\n provenance: {\n url,\n sourceUpdatedAt: effective,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable,\n unverifiableReason: verifiable ? undefined : 'extracted body too short',\n },\n dimensionHints,\n metadata: { sourceId, status: response.status, fromCache: response.fromCache },\n }\n}\n\nfunction extractTitle(html: string, selector: CornellLiiSelector): string {\n const h1 = /<h1[^>]*\\bid=[\"']page_title[\"'][^>]*>([\\s\\S]*?)<\\/h1>/i.exec(html)?.[1]\n if (h1) return htmlToText(h1)\n const t = /<title>([\\s\\S]*?)<\\/title>/i.exec(html)?.[1]\n if (t) return htmlToText(t).split(' | ')[0] ?? `Cornell LII ${selector.path}`\n return `Cornell LII ${selector.kind} ${selector.path}`\n}\n\nfunction extractBody(html: string, selector: CornellLiiSelector): string {\n if (selector.kind === 'uscode') {\n // The statute text lives inside a <text><div class=\"text\">…</div></text>\n // block on US Code section pages. Prefer it; fall back to #tab_default_1\n // which always contains the section body.\n const text = /<text>([\\s\\S]*?)<\\/text>/i.exec(html)?.[1]\n if (text) return htmlToText(text)\n const tab = innerHtmlById(html, 'tab_default_1')\n if (tab) return htmlToText(tab)\n }\n // Wex pages wrap the encyclopedia entry under <div id=\"main-content\"> (newer\n // Drupal template) or directly inside <div id=\"extracted-content\"> (older\n // template). Try both — lazy regex matching against a nested-div container\n // returns the wrong (shorter) slice, so we anchor on the leaf containers.\n const mainContent = innerHtmlById(html, 'main-content')\n if (mainContent) {\n return htmlToText(mainContent.replace(/<h1[\\s\\S]*?<\\/h1>/i, ''))\n }\n const extracted = innerHtmlById(html, 'extracted-content')\n if (extracted) {\n return htmlToText(extracted.replace(/<h1[\\s\\S]*?<\\/h1>/i, ''))\n }\n return htmlToText(html)\n}\n\nfunction extractEffectiveDate(html: string): string | undefined {\n // Cornell LII includes \"Editorial Notes\" / \"Amendments\" blocks with\n // dates; the most reliable machine-readable signal is the last\n // amendment year embedded near the section text.\n const amend = /Amendments[\\s\\S]{0,200}?(\\d{4})/i.exec(html)?.[1]\n if (amend) {\n const y = Number.parseInt(amend, 10)\n if (Number.isFinite(y) && y > 1900 && y <= new Date().getUTCFullYear() + 1) {\n return new Date(Date.UTC(y, 11, 31)).toISOString()\n }\n }\n return undefined\n}\n\nfunction defaultDimensionHints(selector: CornellLiiSelector): string[] {\n if (selector.kind === 'uscode') return ['jurisdictional_accuracy', 'citation_hygiene']\n return ['citation_hygiene']\n}\n","import { sha256 } from '../ids'\nimport { htmlToText } from './html'\nimport { politeFetch } from './http'\nimport type { FetchOpts, KnowledgeFragment, KnowledgeSource } from './types'\n\n/**\n * IRS publications source.\n *\n * Two surfaces:\n *\n * 1. The publications index at https://www.irs.gov/publications enumerates\n * every active publication with its revision year — a single fragment\n * with the full table lets change detection notice when a publication\n * year flips (e.g. Pub 15 (2025) → Pub 15 (2026)).\n *\n * 2. Individual publication landing pages at /publications/p<N>[<suffix>]\n * return one fragment per publication with summary text. Callers list\n * the publications they need tracked via `selectors`.\n *\n * Revenue procedures are fetched under their numbered URLs; the IRS does\n * not maintain a stable HTML index of rev-procs, so the caller passes the\n * specific rev-proc paths they care about.\n *\n * @stable\n */\n\nconst BASE_URL = 'https://www.irs.gov'\nconst INDEX_URL = `${BASE_URL}/publications`\n\nexport interface IrsPublicationsSourceOptions {\n /**\n * Specific publication slugs to fetch (e.g. `['p15', 'p17', 'p463']`).\n * When `includeIndex` is true (default), the publications index page is\n * also fetched as a single fragment so change detection can notice\n * year/revision shifts across the whole catalogue.\n */\n publications?: string[]\n /**\n * Revenue procedure paths to fetch (e.g. `['/irb/2024-31_IRB']`). The\n * caller passes the exact path; this source does not auto-discover.\n */\n revenueProcedures?: string[]\n includeIndex?: boolean\n id?: string\n}\n\n/** Default eval dimensions for IRS-sourced fragments. */\nexport const IRS_DIMENSION_HINTS = ['tax_compliance', 'regulatory_currency', 'citation_hygiene']\n\nexport function createIrsPublicationsSource(\n options: IrsPublicationsSourceOptions = {},\n): KnowledgeSource {\n const id = options.id ?? 'irs-publications'\n const includeIndex = options.includeIndex ?? true\n return {\n id,\n name: 'IRS Publications',\n description:\n 'Internal Revenue Service publications index and individual publication landing pages from irs.gov.',\n async fetch(opts: FetchOpts): Promise<KnowledgeFragment[]> {\n const out: KnowledgeFragment[] = []\n const limit = opts.limit ?? Number.POSITIVE_INFINITY\n\n if (includeIndex && out.length < limit) {\n out.push(await fetchIndex(id, opts))\n }\n for (const slug of options.publications ?? []) {\n if (out.length >= limit) break\n out.push(await fetchPublication(id, slug, opts))\n }\n for (const path of options.revenueProcedures ?? []) {\n if (out.length >= limit) break\n out.push(await fetchRevenueProcedure(id, path, opts))\n }\n return out\n },\n }\n}\n\nasync function fetchIndex(sourceId: string, opts: FetchOpts): Promise<KnowledgeFragment> {\n const response = await politeFetch(INDEX_URL, { signal: opts.signal, cacheDir: opts.cacheDir })\n const tablePattern = /<table[\\s\\S]*?<\\/table>/gi\n const matches = response.body.match(tablePattern) ?? []\n // Extract the table that lists current-year publications. IRS publishes\n // one table per year on the index; the most recent table is always the\n // first that mentions a year ≥ current.\n const tables = matches.map((t) => htmlToText(t))\n const body = tables\n .filter((t) => /Publication\\s*\\d+/i.test(t))\n .join('\\n\\n')\n .slice(0, 200_000)\n\n const verifiable = response.verifiable && body.length > 200\n return {\n id: 'index',\n title: 'IRS Publications Index',\n body,\n bodyHash: sha256(body),\n provenance: {\n url: INDEX_URL,\n sourceUpdatedAt: response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable,\n unverifiableReason:\n response.unverifiableReason ?? (verifiable ? undefined : 'no publication rows extracted'),\n },\n dimensionHints: IRS_DIMENSION_HINTS,\n metadata: { sourceId, status: response.status, fromCache: response.fromCache, kind: 'index' },\n }\n}\n\nasync function fetchPublication(\n sourceId: string,\n slug: string,\n opts: FetchOpts,\n): Promise<KnowledgeFragment> {\n const url = `${BASE_URL}/publications/${slug.replace(/^\\/+/, '')}`\n const response = await politeFetch(url, { signal: opts.signal, cacheDir: opts.cacheDir })\n\n const title = extractTitle(response.body, `IRS Publication ${slug}`)\n const body = extractMainContent(response.body)\n const verifiable = response.verifiable && body.length > 200\n\n return {\n id: `publication:${slug}`,\n title,\n body,\n bodyHash: sha256(body),\n provenance: {\n url,\n sourceUpdatedAt: extractRevisionDate(response.body) ?? response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable,\n unverifiableReason:\n response.unverifiableReason ?? (verifiable ? undefined : 'no publication body extracted'),\n },\n dimensionHints: IRS_DIMENSION_HINTS,\n metadata: {\n sourceId,\n status: response.status,\n fromCache: response.fromCache,\n kind: 'publication',\n slug,\n },\n }\n}\n\nasync function fetchRevenueProcedure(\n sourceId: string,\n path: string,\n opts: FetchOpts,\n): Promise<KnowledgeFragment> {\n const url = `${BASE_URL}${path.startsWith('/') ? path : `/${path}`}`\n const response = await politeFetch(url, { signal: opts.signal, cacheDir: opts.cacheDir })\n const body = extractMainContent(response.body)\n const verifiable = response.verifiable && body.length > 200\n return {\n id: `rev-proc:${path}`,\n title: extractTitle(response.body, `IRS Revenue Procedure ${path}`),\n body,\n bodyHash: sha256(body),\n provenance: {\n url,\n sourceUpdatedAt: response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: 'US-FED',\n verifiable,\n unverifiableReason:\n response.unverifiableReason ??\n (verifiable ? undefined : 'no revenue-procedure body extracted'),\n },\n dimensionHints: [...IRS_DIMENSION_HINTS, 'procedural_currency'],\n metadata: {\n sourceId,\n status: response.status,\n fromCache: response.fromCache,\n kind: 'rev-proc',\n path,\n },\n }\n}\n\nfunction extractTitle(html: string, fallback: string): string {\n const og = /<meta\\s+property=[\"']og:title[\"']\\s+content=[\"']([^\"']+)[\"']/i.exec(html)?.[1]\n if (og) return decodeHtml(og)\n const title = /<title>([\\s\\S]*?)<\\/title>/i.exec(html)?.[1]\n if (title) return htmlToText(title).split(' | ')[0] ?? fallback\n return fallback\n}\n\nfunction extractMainContent(html: string): string {\n // IRS uses Drupal — the main publication body is inside <main role=\"main\">\n // or under .field--name-body. We try main first; on miss, body.\n const main = /<main\\b[\\s\\S]*?<\\/main>/i.exec(html)?.[0]\n if (main) {\n const noNav = main\n .replace(/<nav[\\s\\S]*?<\\/nav>/gi, '')\n .replace(/<header[\\s\\S]*?<\\/header>/gi, '')\n .replace(/<footer[\\s\\S]*?<\\/footer>/gi, '')\n return htmlToText(noNav).slice(0, 200_000)\n }\n const body = /<body\\b[\\s\\S]*?<\\/body>/i.exec(html)?.[0]\n return body ? htmlToText(body).slice(0, 200_000) : htmlToText(html).slice(0, 200_000)\n}\n\nfunction extractRevisionDate(html: string): string | undefined {\n // IRS publication pages typically show \"Publication X (YYYY)\" in the title;\n // pulling the year gives a stable revision marker.\n const m = /Publication\\s+\\S+\\s*\\((\\d{4})\\)/i.exec(html)\n if (m?.[1]) {\n const year = Number.parseInt(m[1], 10)\n if (Number.isFinite(year) && year >= 2000 && year <= new Date().getUTCFullYear() + 1) {\n return new Date(Date.UTC(year, 0, 1)).toISOString()\n }\n }\n return undefined\n}\n\nfunction decodeHtml(value: string): string {\n return htmlToText(value)\n}\n","import { sha256 } from '../ids'\nimport { htmlToText } from './html'\nimport { politeFetch } from './http'\nimport type { FetchOpts, KnowledgeFragment, KnowledgeSource } from './types'\n\n/**\n * Generic Secretary-of-State source.\n *\n * Every US state SOS surfaces LLC/Corp formation requirements differently\n * (CA via static forms pages, DE via division of corporations pages, TX\n * via SOSDirect content pages). Rather than baking 50 state-specific\n * parsers into this package, the source takes a config that names the URL\n * pattern + CSS-equivalent selector + jurisdiction tag. Callers supply one\n * config per state they need tracked.\n *\n * The selector is interpreted as a substring/regex of an HTML element id\n * or class — see `StateSosSourceConfig` for the contract. This is\n * intentionally minimal; richer extraction belongs in a state-specific\n * adapter the consumer authors.\n *\n * @experimental Interface will likely grow as we add more state coverage.\n */\n\nexport interface StateSosEntity {\n /** Stable id for this fragment within the state (e.g. 'llc-formation', 'corp-formation'). */\n id: string\n /** Path under the configured `baseUrl` for this entity. */\n path: string\n /**\n * Extraction selector. Choose one:\n * - `{ kind: 'id', value: 'main-content' }` — innermost match of element with that id\n * - `{ kind: 'class', value: 'field--name-body' }` — innermost match of element with that class\n * - `{ kind: 'regex', value: /<article[\\s\\S]*?<\\/article>/i }` — raw regex\n * - `{ kind: 'whole' }` — full body, tags stripped (fallback for unstructured pages)\n */\n selector:\n | { kind: 'id'; value: string }\n | { kind: 'class'; value: string }\n | { kind: 'regex'; value: RegExp }\n | { kind: 'whole' }\n title: string\n /** Eval dimensions this entity feeds. */\n dimensionHints?: string[]\n}\n\nexport interface StateSosSourceConfig {\n /** US state postal code, e.g. 'CA', 'DE', 'TX'. */\n state: string\n /** Base URL for the state SOS — e.g. 'https://www.sos.ca.gov'. */\n baseUrl: string\n /** Entities this state exposes (LLC, Corp, etc). */\n entities: StateSosEntity[]\n /** Source id; default `state-sos:<state>`. */\n id?: string\n /** Display name; default `<state> Secretary of State`. */\n name?: string\n}\n\nexport function createStateSosSource(config: StateSosSourceConfig): KnowledgeSource {\n const id = config.id ?? `state-sos:${config.state.toLowerCase()}`\n const name = config.name ?? `${config.state} Secretary of State`\n return {\n id,\n name,\n description: `${config.state} Secretary of State filings and formation guidance pages.`,\n async fetch(opts: FetchOpts): Promise<KnowledgeFragment[]> {\n const limit = opts.limit ?? config.entities.length\n const entities = config.entities.slice(0, limit)\n const out: KnowledgeFragment[] = []\n for (const entity of entities) {\n out.push(await fetchEntity(id, config, entity, opts))\n }\n return out\n },\n }\n}\n\nasync function fetchEntity(\n sourceId: string,\n config: StateSosSourceConfig,\n entity: StateSosEntity,\n opts: FetchOpts,\n): Promise<KnowledgeFragment> {\n const url = joinUrl(config.baseUrl, entity.path)\n const response = await politeFetch(url, { signal: opts.signal, cacheDir: opts.cacheDir })\n\n const body = response.verifiable ? extractBySelector(response.body, entity.selector) : ''\n const verifiable = response.verifiable && body.length > 100\n\n return {\n id: entity.id,\n title: entity.title,\n body,\n bodyHash: sha256(body),\n provenance: {\n url,\n sourceUpdatedAt: response.sourceUpdatedAt,\n fetchedAt: response.fetchedAt,\n jurisdiction: `US-${config.state.toUpperCase()}`,\n verifiable,\n unverifiableReason:\n response.unverifiableReason ?? (verifiable ? undefined : 'extracted body too short'),\n },\n dimensionHints: entity.dimensionHints ?? [\n 'jurisdictional_accuracy',\n 'corporate_formation',\n 'citation_hygiene',\n ],\n metadata: {\n sourceId,\n status: response.status,\n fromCache: response.fromCache,\n state: config.state,\n },\n }\n}\n\nfunction extractBySelector(html: string, selector: StateSosEntity['selector']): string {\n if (selector.kind === 'whole') {\n const main = /<main\\b[\\s\\S]*?<\\/main>/i.exec(html)?.[0]\n return htmlToText(main ?? html).slice(0, 200_000)\n }\n if (selector.kind === 'regex') {\n const m = selector.value.exec(html)?.[0]\n return m ? htmlToText(m).slice(0, 200_000) : ''\n }\n if (selector.kind === 'id') {\n const escaped = selector.value.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n const pattern = new RegExp(\n `<([a-z][a-z0-9]*)\\\\b[^>]*\\\\sid=[\"']${escaped}[\"'][^>]*>([\\\\s\\\\S]*?)<\\\\/\\\\1>`,\n 'i',\n )\n const inner = pattern.exec(html)?.[2]\n return inner ? htmlToText(inner).slice(0, 200_000) : ''\n }\n const escaped = selector.value.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&')\n const pattern = new RegExp(\n `<([a-z][a-z0-9]*)\\\\b[^>]*\\\\sclass=[\"'][^\"']*\\\\b${escaped}\\\\b[^\"']*[\"'][^>]*>([\\\\s\\\\S]*?)<\\\\/\\\\1>`,\n 'i',\n )\n const inner = pattern.exec(html)?.[2]\n return inner ? htmlToText(inner).slice(0, 200_000) : ''\n}\n\nfunction joinUrl(base: string, path: string): string {\n try {\n return new URL(path, base.endsWith('/') ? base : `${base}/`).toString()\n } catch {\n return `${base.replace(/\\/+$/, '')}/${path.replace(/^\\/+/, '')}`\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;AAqBA,SAAgB,WAAW,MAAsB;CAC/C,OAAO,KACJ,QAAQ,+BAA+B,EAAE,CAAC,CAC1C,QAAQ,6BAA6B,EAAE,CAAC,CACxC,QAAQ,mCAAmC,EAAE,CAAC,CAC9C,QAAQ,sBAAsB,EAAE,CAAC,CACjC,QAAQ,mBAAmB,IAAI,CAAC,CAChC,QAAQ,yDAAyD,IAAI,CAAC,CACtE,QAAQ,YAAY,EAAE,CAAC,CACvB,QAAQ,YAAY,GAAG,CAAC,CACxB,QAAQ,WAAW,GAAG,CAAC,CACvB,QAAQ,UAAU,GAAG,CAAC,CACtB,QAAQ,UAAU,GAAG,CAAC,CACtB,QAAQ,YAAY,IAAG,CAAC,CACxB,QAAQ,WAAW,GAAG,CAAC,CACvB,QAAQ,YAAY,GAAG,CAAC,CACxB,QAAQ,aAAa,GAAG,CAAC,CACzB,QAAQ,aAAa,GAAG,CAAC,CACzB,QAAQ,cAAc,GAAG,SAAS,OAAO,cAAc,OAAO,IAAI,CAAC,CAAC,CAAC,CACrE,QAAQ,sBAAsB,GAAG,SAAS,OAAO,cAAc,OAAO,SAAS,MAAM,EAAE,CAAC,CAAC,CAAC,CAC1F,MAAM,IAAI,CAAC,CACX,KAAK,SAAS,KAAK,QAAQ,YAAY,GAAG,CAAC,CAAC,KAAK,CAAC,CAAC,CACnD,QAAQ,MAAM,KAAK,QAAQ,EAAE,SAAS,MAAM,IAAI,MAAM,OAAO,GAAG,CAAC,CACjE,KAAK,IAAI,CAAC,CACV,KAAK;AACV;;AAGA,SAAgB,WAAW,MAAc,SAAqC;CAC5E,OAAO,QAAQ,KAAK,IAAI,CAAC,GAAG,EAAE,EAAE,KAAK;AACvC;;AAGA,SAAgB,cAAc,MAAc,IAAgC;CAC1E,MAAM,UAAU,GAAG,QAAQ,uBAAuB,MAAM;CAKxD,OAAO,IAJgB,OACrB,sCAAsC,QAAQ,iCAC9C,GAEc,CAAC,CAAC,KAAK,IAAI,CAAC,GAAG;AACjC;;;;;AAMA,SAAgB,aACd,MACA,aACA,SACkC;CAClC,MAAM,MAAwC,CAAC;CAE/C,KAAK,MAAM,SAAS,KAAK,SAAS,yDAAM,GAAG;EACzC,MAAM,OAAO,MAAM;EACnB,MAAM,QAAQ,MAAM;EACpB,IAAI,CAAC,QAAQ,CAAC,OAAO;EACrB,IAAI,CAAC,YAAY,KAAK,IAAI,GAAG;EAC7B,MAAM,OAAO,WAAW,KAAK;EAC7B,IAAI,CAAC,MAAM;EACX,IAAI;GACF,IAAI,KAAK;IAAE,MAAM,IAAI,IAAI,MAAM,OAAO,CAAC,CAAC,SAAS;IAAG;GAAK,CAAC;EAC5D,QAAQ,CAER;CACF;CACA,OAAO;AACT;;;;;;;;;;;;;ACzEA,MAAa,oBACX;;AAGF,MAAa,qBAAqB;;AAGlC,MAAa,qBAAqB;AAElC,MAAM,+BAAe,IAAI,IAA2B;;;;;;;;;;AAiDpD,eAAsB,YACpB,KACA,UAA8B,CAAC,GACH;CAC5B,MAAM,WAAW,QAAQ,cAAc;CACvC,MAAM,SAAS,QAAQ,WAAW,MAAM,UAAU,QAAQ,UAAU,KAAK,QAAQ,IAAI,KAAA;CACrF,IAAI,QAAQ,OAAO;CAEnB,MAAM,OAAO,SAAS,GAAG;CACzB,MAAM,aAAa,IAAI;CAEvB,MAAM,6BAAY,IAAI,KAAK,EAAA,CAAE,YAAY;CACzC,IAAI;CACJ,IAAI;EACF,WAAW,MAAM,MAAM,KAAK;GAC1B,QAAQ,QAAQ;GAChB,UAAU;GACV,SAAS;IACP,cAAc;IACd,QAAQ;IACR,mBAAmB;IACnB,GAAI,QAAQ,WAAW,CAAC;GAC1B;EACF,CAAC;CACH,SAAS,OAAO;EACd,IAAK,MAA4B,SAAS,cAAc,MAAM;EAC9D,MAAM,SAA4B;GAChC;GACA,QAAQ;GACR,MAAM;GACN,iBAAiB;GACjB;GACA,WAAW;GACX,YAAY;GACZ,oBAAoB,kBAAmB,MAAgB;EACzD;EACA,IAAI,QAAQ,UAAU,MAAM,WAAW,QAAQ,UAAU,KAAK,MAAM;EACpE,OAAO;CACT;CAEA,MAAM,OAAO,MAAM,gBAAgB,QAAQ;CAC3C,MAAM,eAAe,SAAS,QAAQ,IAAI,eAAe;CACzD,MAAM,aAAa,SAAS,QAAQ,IAAI,MAAM;CAC9C,MAAM,kBAAkB,cAAc,YAAY,KAAK,cAAc,UAAU,KAAK;CAEpF,MAAM,SAA4B;EAChC;EACA,QAAQ,SAAS;EACjB,MAAM;EACN;EACA;EACA,WAAW;EACX,YAAY;CACd;CAEA,IAAI,SAAS,SAAS,OAAO,SAAS,UAAU,KAAK;EACnD,OAAO,aAAa;EACpB,OAAO,qBAAqB,mBAAmB,SAAS;CAC1D,OAAO,IAAI,mBAAmB,IAAI,GAAG;EACnC,OAAO,aAAa;EACpB,OAAO,qBAAqB;CAC9B,OAAO,IAAI,KAAK,SAAS,OAAO,oBAAoB,IAAI,GAAG;EACzD,OAAO,aAAa;EACpB,OAAO,qBAAqB,+BAA+B,KAAK,OAAO;CACzE;CAEA,IAAI,QAAQ,UAAU,MAAM,WAAW,QAAQ,UAAU,KAAK,MAAM;CACpE,OAAO;AACT;;AAGA,SAAgB,sBAA4B;CAC1C,aAAa,MAAM;AACrB;AAEA,SAAS,SAAS,KAAqB;CACrC,IAAI;EACF,OAAO,IAAI,IAAI,GAAG,CAAC,CAAC;CACtB,QAAQ;EACN,OAAO;CACT;AACF;AAEA,eAAe,aAAa,MAA6B;CACvD,MAAM,OAAO,aAAa,IAAI,IAAI,KAAK,QAAQ,QAAQ;CACvD,IAAI,gBAA4B,CAAC;CACjC,MAAM,OAAO,IAAI,SAAe,YAAY;EAC1C,UAAU;CACZ,CAAC;CACD,aAAa,IACX,MACA,KAAK,WAAW,IAAI,CACtB;CACA,MAAM;CACN,WAAW,SAAS,kBAAkB;AACxC;AAEA,eAAe,gBAAgB,UAAqC;CAClE,IAAI,CAAC,SAAS,MAAM,OAAO;CAC3B,MAAM,SAAS,SAAS,KAAK,UAAU;CACvC,MAAM,SAAuB,CAAC;CAC9B,IAAI,QAAQ;CACZ,OAAO,MAAM;EACX,MAAM,EAAE,MAAM,UAAU,MAAM,OAAO,KAAK;EAC1C,IAAI,MAAM;EACV,IAAI,CAAC,OAAO;EACZ,SAAS,MAAM;EACf,IAAI,QAAA,SAA4B;GAE9B,MAAM,OAAO,OAAO;GACpB;EACF;EACA,OAAO,KAAK,KAAK;CACnB;CACA,MAAM,SAAS,IAAI,WAAW,KAAK,IAAI,OAAO,kBAAkB,CAAC;CACjE,IAAI,SAAS;CACb,KAAK,MAAM,SAAS,QAAQ;EAC1B,MAAM,OAAO,KAAK,IAAI,MAAM,QAAQ,OAAO,SAAS,MAAM;EAC1D,IAAI,QAAQ,GAAG;EACf,OAAO,IAAI,MAAM,SAAS,GAAG,IAAI,GAAG,MAAM;EAC1C,UAAU;CACZ;CACA,OAAO,IAAI,YAAY,SAAS,EAAE,OAAO,MAAM,CAAC,CAAC,CAAC,OAAO,MAAM;AACjE;AAEA,SAAS,cAAc,OAA0C;CAC/D,IAAI,CAAC,OAAO,OAAO,KAAA;CACnB,MAAM,KAAK,KAAK,MAAM,KAAK;CAC3B,OAAO,OAAO,SAAS,EAAE,IAAI,IAAI,KAAK,EAAE,CAAC,CAAC,YAAY,IAAI,KAAA;AAC5D;;AAGA,SAAgB,mBAAmB,MAAuB;CACxD,IAAI,CAAC,MAAM,OAAO;CAClB,MAAM,QAAQ,KAAK,YAAY;CAY/B,KAAK,MAAM,UAAU;EAVnB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CAEyB,GACzB,IAAI,MAAM,SAAS,MAAM,GAAG,OAAO;CAErC,OAAO;AACT;AAEA,SAAS,oBAAoB,MAAuB;CAClD,OACE,KAAK,SAAS,iBAAiB,KAC/B,KAAK,SAAS,SAAS,KACvB,KAAK,SAAS,YAAY,KAC1B,KAAK,SAAS,iBAAiB,KAC/B,KAAK,SAAS,cAAc;AAEhC;AAEA,SAAS,UAAU,UAAkB,KAAqB;CACxD,MAAM,MAAM,OAAO,GAAG;CACtB,OAAO,KAAK,UAAU,QAAQ,GAAG,IAAI,MAAM,GAAG,CAAC,KAAK,GAAG,IAAI,MAAM;AACnE;AAEA,eAAe,UACb,UACA,KACA,OACwC;CACxC,MAAM,OAAO,UAAU,UAAU,GAAG;CACpC,IAAI;EACF,MAAM,OAAO,MAAM,KAAK,IAAI;EAC5B,IAAI,KAAK,IAAI,IAAI,KAAK,UAAU,OAAO,OAAO,KAAA;EAC9C,MAAM,MAAM,MAAM,SAAS,MAAM,MAAM;EAEvC,OAAO;GAAE,GADM,KAAK,MAAM,GACT;GAAG,WAAW;EAAK;CACtC,QAAQ;EACN;CACF;AACF;AAEA,eAAe,WAAW,UAAkB,KAAa,OAAyC;CAChG,MAAM,OAAO,UAAU,UAAU,GAAG;CACpC,MAAM,MAAM,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAC9C,MAAM,UAAU,MAAM,KAAK,UAAU,KAAK,GAAG,MAAM;AACrD;;;;;;;;;;;;;ACrPA,MAAMA,aAAW;;;;;;;;;;;;;;AA0CjB,SAAgB,uBAAuB,SAAmD;CACxF,MAAM,KAAK,QAAQ,MAAM;CACzB,OAAO;EACL;EACA,MAAM;EACN,aACE;EACF,MAAM,MAAM,MAA+C;GACzD,MAAM,QAAQ,KAAK,SAAS,QAAQ,UAAU;GAC9C,MAAM,YAAY,QAAQ,UAAU,MAAM,GAAG,KAAK;GAClD,MAAM,MAA2B,CAAC;GAClC,KAAK,MAAM,YAAY,WACrB,IAAI,KAAK,MAAM,SAAS,IAAI,UAAU,IAAI,CAAC;GAE7C,OAAO;EACT;CACF;AACF;AAEA,eAAe,SACb,UACA,UACA,MAC4B;CAC5B,MAAM,OAAO,SAAS,KAAK,QAAQ,QAAQ,EAAE;CAC7C,MAAM,MACJ,SAAS,SAAS,WAAW,GAAGA,WAAS,eAAe,SAAS,GAAGA,WAAS,OAAO;CAEtF,MAAM,WAAW,MAAM,YAAY,KAAK;EACtC,QAAQ,KAAK;EACb,UAAU,KAAK;CACjB,CAAC;CAED,MAAM,aAAa,GAAG,SAAS,KAAK,GAAG,SAAS;CAChD,MAAM,iBAAiB,SAAS,kBAAkB,sBAAsB,QAAQ;CAEhF,IAAI,CAAC,SAAS,YACZ,OAAO;EACL,IAAI;EACJ,OAAO,eAAe,SAAS,KAAK,GAAG,SAAS;EAChD,MAAM;EACN,UAAU,OAAO,EAAE;EACnB,YAAY;GACV;GACA,iBAAiB,SAAS;GAC1B,WAAW,SAAS;GACpB,cAAc;GACd,YAAY;GACZ,oBAAoB,SAAS;EAC/B;EACA;EACA,UAAU;GAAE;GAAU,QAAQ,SAAS;GAAQ,WAAW,SAAS;EAAU;CAC/E;CAGF,MAAM,OAAO,SAAS;CACtB,MAAM,QAAQC,eAAa,MAAM,QAAQ;CACzC,MAAM,OAAO,YAAY,MAAM,QAAQ;CACvC,MAAM,YAAY,qBAAqB,IAAI,KAAK,SAAS;CAEzD,MAAM,aAAa,KAAK,SAAS;CACjC,OAAO;EACL,IAAI;EACJ;EACA;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV;GACA,iBAAiB;GACjB,WAAW,SAAS;GACpB,cAAc;GACd;GACA,oBAAoB,aAAa,KAAA,IAAY;EAC/C;EACA;EACA,UAAU;GAAE;GAAU,QAAQ,SAAS;GAAQ,WAAW,SAAS;EAAU;CAC/E;AACF;AAEA,SAASA,eAAa,MAAc,UAAsC;CACxE,MAAM,KAAK,yDAAyD,KAAK,IAAI,CAAC,GAAG;CACjF,IAAI,IAAI,OAAO,WAAW,EAAE;CAC5B,MAAM,IAAI,8BAA8B,KAAK,IAAI,CAAC,GAAG;CACrD,IAAI,GAAG,OAAO,WAAW,CAAC,CAAC,CAAC,MAAM,KAAK,CAAC,CAAC,MAAM,eAAe,SAAS;CACvE,OAAO,eAAe,SAAS,KAAK,GAAG,SAAS;AAClD;AAEA,SAAS,YAAY,MAAc,UAAsC;CACvE,IAAI,SAAS,SAAS,UAAU;EAI9B,MAAM,OAAO,4BAA4B,KAAK,IAAI,CAAC,GAAG;EACtD,IAAI,MAAM,OAAO,WAAW,IAAI;EAChC,MAAM,MAAM,cAAc,MAAM,eAAe;EAC/C,IAAI,KAAK,OAAO,WAAW,GAAG;CAChC;CAKA,MAAM,cAAc,cAAc,MAAM,cAAc;CACtD,IAAI,aACF,OAAO,WAAW,YAAY,QAAQ,sBAAsB,EAAE,CAAC;CAEjE,MAAM,YAAY,cAAc,MAAM,mBAAmB;CACzD,IAAI,WACF,OAAO,WAAW,UAAU,QAAQ,sBAAsB,EAAE,CAAC;CAE/D,OAAO,WAAW,IAAI;AACxB;AAEA,SAAS,qBAAqB,MAAkC;CAI9D,MAAM,QAAQ,mCAAmC,KAAK,IAAI,CAAC,GAAG;CAC9D,IAAI,OAAO;EACT,MAAM,IAAI,OAAO,SAAS,OAAO,EAAE;EACnC,IAAI,OAAO,SAAS,CAAC,KAAK,IAAI,QAAQ,sBAAK,IAAI,KAAK,EAAA,CAAE,eAAe,IAAI,GACvE,OAAO,IAAI,KAAK,KAAK,IAAI,GAAG,IAAI,EAAE,CAAC,CAAC,CAAC,YAAY;CAErD;AAEF;AAEA,SAAS,sBAAsB,UAAwC;CACrE,IAAI,SAAS,SAAS,UAAU,OAAO,CAAC,2BAA2B,kBAAkB;CACrF,OAAO,CAAC,kBAAkB;AAC5B;;;;;;;;;;;;;;;;;;;;;;;ACjKA,MAAM,WAAW;AACjB,MAAM,YAAY,GAAG,SAAS;;AAoB9B,MAAa,sBAAsB;CAAC;CAAkB;CAAuB;AAAkB;AAE/F,SAAgB,4BACd,UAAwC,CAAC,GACxB;CACjB,MAAM,KAAK,QAAQ,MAAM;CACzB,MAAM,eAAe,QAAQ,gBAAgB;CAC7C,OAAO;EACL;EACA,MAAM;EACN,aACE;EACF,MAAM,MAAM,MAA+C;GACzD,MAAM,MAA2B,CAAC;GAClC,MAAM,QAAQ,KAAK,SAAS,OAAO;GAEnC,IAAI,gBAAgB,IAAI,SAAS,OAC/B,IAAI,KAAK,MAAM,WAAW,IAAI,IAAI,CAAC;GAErC,KAAK,MAAM,QAAQ,QAAQ,gBAAgB,CAAC,GAAG;IAC7C,IAAI,IAAI,UAAU,OAAO;IACzB,IAAI,KAAK,MAAM,iBAAiB,IAAI,MAAM,IAAI,CAAC;GACjD;GACA,KAAK,MAAM,QAAQ,QAAQ,qBAAqB,CAAC,GAAG;IAClD,IAAI,IAAI,UAAU,OAAO;IACzB,IAAI,KAAK,MAAM,sBAAsB,IAAI,MAAM,IAAI,CAAC;GACtD;GACA,OAAO;EACT;CACF;AACF;AAEA,eAAe,WAAW,UAAkB,MAA6C;CACvF,MAAM,WAAW,MAAM,YAAY,WAAW;EAAE,QAAQ,KAAK;EAAQ,UAAU,KAAK;CAAS,CAAC;CAO9F,MAAM,QALU,SAAS,KAAK,MAAM,2BAAY,KAAK,CAAC,EAAA,CAI/B,KAAK,MAAM,WAAW,CAAC,CAC5B,CAAC,CAChB,QAAQ,MAAM,qBAAqB,KAAK,CAAC,CAAC,CAAC,CAC3C,KAAK,MAAM,CAAC,CACZ,MAAM,GAAG,GAAO;CAEnB,MAAM,aAAa,SAAS,cAAc,KAAK,SAAS;CACxD,OAAO;EACL,IAAI;EACJ,OAAO;EACP;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV,KAAK;GACL,iBAAiB,SAAS;GAC1B,WAAW,SAAS;GACpB,cAAc;GACd;GACA,oBACE,SAAS,uBAAuB,aAAa,KAAA,IAAY;EAC7D;EACA,gBAAgB;EAChB,UAAU;GAAE;GAAU,QAAQ,SAAS;GAAQ,WAAW,SAAS;GAAW,MAAM;EAAQ;CAC9F;AACF;AAEA,eAAe,iBACb,UACA,MACA,MAC4B;CAC5B,MAAM,MAAM,GAAG,SAAS,gBAAgB,KAAK,QAAQ,QAAQ,EAAE;CAC/D,MAAM,WAAW,MAAM,YAAY,KAAK;EAAE,QAAQ,KAAK;EAAQ,UAAU,KAAK;CAAS,CAAC;CAExF,MAAM,QAAQ,aAAa,SAAS,MAAM,mBAAmB,MAAM;CACnE,MAAM,OAAO,mBAAmB,SAAS,IAAI;CAC7C,MAAM,aAAa,SAAS,cAAc,KAAK,SAAS;CAExD,OAAO;EACL,IAAI,eAAe;EACnB;EACA;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV;GACA,iBAAiB,oBAAoB,SAAS,IAAI,KAAK,SAAS;GAChE,WAAW,SAAS;GACpB,cAAc;GACd;GACA,oBACE,SAAS,uBAAuB,aAAa,KAAA,IAAY;EAC7D;EACA,gBAAgB;EAChB,UAAU;GACR;GACA,QAAQ,SAAS;GACjB,WAAW,SAAS;GACpB,MAAM;GACN;EACF;CACF;AACF;AAEA,eAAe,sBACb,UACA,MACA,MAC4B;CAC5B,MAAM,MAAM,GAAG,WAAW,KAAK,WAAW,GAAG,IAAI,OAAO,IAAI;CAC5D,MAAM,WAAW,MAAM,YAAY,KAAK;EAAE,QAAQ,KAAK;EAAQ,UAAU,KAAK;CAAS,CAAC;CACxF,MAAM,OAAO,mBAAmB,SAAS,IAAI;CAC7C,MAAM,aAAa,SAAS,cAAc,KAAK,SAAS;CACxD,OAAO;EACL,IAAI,YAAY;EAChB,OAAO,aAAa,SAAS,MAAM,yBAAyB,MAAM;EAClE;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV;GACA,iBAAiB,SAAS;GAC1B,WAAW,SAAS;GACpB,cAAc;GACd;GACA,oBACE,SAAS,uBACR,aAAa,KAAA,IAAY;EAC9B;EACA,gBAAgB,CAAC,GAAG,qBAAqB,qBAAqB;EAC9D,UAAU;GACR;GACA,QAAQ,SAAS;GACjB,WAAW,SAAS;GACpB,MAAM;GACN;EACF;CACF;AACF;AAEA,SAAS,aAAa,MAAc,UAA0B;CAC5D,MAAM,KAAK,gEAAgE,KAAK,IAAI,CAAC,GAAG;CACxF,IAAI,IAAI,OAAO,WAAW,EAAE;CAC5B,MAAM,QAAQ,8BAA8B,KAAK,IAAI,CAAC,GAAG;CACzD,IAAI,OAAO,OAAO,WAAW,KAAK,CAAC,CAAC,MAAM,KAAK,CAAC,CAAC,MAAM;CACvD,OAAO;AACT;AAEA,SAAS,mBAAmB,MAAsB;CAGhD,MAAM,OAAO,2BAA2B,KAAK,IAAI,CAAC,GAAG;CACrD,IAAI,MAKF,OAAO,WAJO,KACX,QAAQ,yBAAyB,EAAE,CAAC,CACpC,QAAQ,+BAA+B,EAAE,CAAC,CAC1C,QAAQ,+BAA+B,EACxB,CAAK,CAAC,CAAC,MAAM,GAAG,GAAO;CAE3C,MAAM,OAAO,2BAA2B,KAAK,IAAI,CAAC,GAAG;CACrD,OAAO,OAAO,WAAW,IAAI,CAAC,CAAC,MAAM,GAAG,GAAO,IAAI,WAAW,IAAI,CAAC,CAAC,MAAM,GAAG,GAAO;AACtF;AAEA,SAAS,oBAAoB,MAAkC;CAG7D,MAAM,IAAI,mCAAmC,KAAK,IAAI;CACtD,IAAI,IAAI,IAAI;EACV,MAAM,OAAO,OAAO,SAAS,EAAE,IAAI,EAAE;EACrC,IAAI,OAAO,SAAS,IAAI,KAAK,QAAQ,OAAQ,yBAAQ,IAAI,KAAK,EAAA,CAAE,eAAe,IAAI,GACjF,OAAO,IAAI,KAAK,KAAK,IAAI,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,YAAY;CAEtD;AAEF;AAEA,SAAS,WAAW,OAAuB;CACzC,OAAO,WAAW,KAAK;AACzB;;;ACpKA,SAAgB,qBAAqB,QAA+C;CAClF,MAAM,KAAK,OAAO,MAAM,aAAa,OAAO,MAAM,YAAY;CAE9D,OAAO;EACL;EACA,MAHW,OAAO,QAAQ,GAAG,OAAO,MAAM;EAI1C,aAAa,GAAG,OAAO,MAAM;EAC7B,MAAM,MAAM,MAA+C;GACzD,MAAM,QAAQ,KAAK,SAAS,OAAO,SAAS;GAC5C,MAAM,WAAW,OAAO,SAAS,MAAM,GAAG,KAAK;GAC/C,MAAM,MAA2B,CAAC;GAClC,KAAK,MAAM,UAAU,UACnB,IAAI,KAAK,MAAM,YAAY,IAAI,QAAQ,QAAQ,IAAI,CAAC;GAEtD,OAAO;EACT;CACF;AACF;AAEA,eAAe,YACb,UACA,QACA,QACA,MAC4B;CAC5B,MAAM,MAAM,QAAQ,OAAO,SAAS,OAAO,IAAI;CAC/C,MAAM,WAAW,MAAM,YAAY,KAAK;EAAE,QAAQ,KAAK;EAAQ,UAAU,KAAK;CAAS,CAAC;CAExF,MAAM,OAAO,SAAS,aAAa,kBAAkB,SAAS,MAAM,OAAO,QAAQ,IAAI;CACvF,MAAM,aAAa,SAAS,cAAc,KAAK,SAAS;CAExD,OAAO;EACL,IAAI,OAAO;EACX,OAAO,OAAO;EACd;EACA,UAAU,OAAO,IAAI;EACrB,YAAY;GACV;GACA,iBAAiB,SAAS;GAC1B,WAAW,SAAS;GACpB,cAAc,MAAM,OAAO,MAAM,YAAY;GAC7C;GACA,oBACE,SAAS,uBAAuB,aAAa,KAAA,IAAY;EAC7D;EACA,gBAAgB,OAAO,kBAAkB;GACvC;GACA;GACA;EACF;EACA,UAAU;GACR;GACA,QAAQ,SAAS;GACjB,WAAW,SAAS;GACpB,OAAO,OAAO;EAChB;CACF;AACF;AAEA,SAAS,kBAAkB,MAAc,UAA8C;CACrF,IAAI,SAAS,SAAS,SAAS;EAC7B,MAAM,OAAO,2BAA2B,KAAK,IAAI,CAAC,GAAG;EACrD,OAAO,WAAW,QAAQ,IAAI,CAAC,CAAC,MAAM,GAAG,GAAO;CAClD;CACA,IAAI,SAAS,SAAS,SAAS;EAC7B,MAAM,IAAI,SAAS,MAAM,KAAK,IAAI,CAAC,GAAG;EACtC,OAAO,IAAI,WAAW,CAAC,CAAC,CAAC,MAAM,GAAG,GAAO,IAAI;CAC/C;CACA,IAAI,SAAS,SAAS,MAAM;EAC1B,MAAM,UAAU,SAAS,MAAM,QAAQ,uBAAuB,MAAM;EAKpE,MAAM,QAAQ,IAJM,OAClB,sCAAsC,QAAQ,iCAC9C,GAEkB,CAAC,CAAC,KAAK,IAAI,CAAC,GAAG;EACnC,OAAO,QAAQ,WAAW,KAAK,CAAC,CAAC,MAAM,GAAG,GAAO,IAAI;CACvD;CACA,MAAM,UAAU,SAAS,MAAM,QAAQ,uBAAuB,MAAM;CAKpE,MAAM,QAAQ,IAJM,OAClB,kDAAkD,QAAQ,0CAC1D,GAEkB,CAAC,CAAC,KAAK,IAAI,CAAC,GAAG;CACnC,OAAO,QAAQ,WAAW,KAAK,CAAC,CAAC,MAAM,GAAG,GAAO,IAAI;AACvD;AAEA,SAAS,QAAQ,MAAc,MAAsB;CACnD,IAAI;EACF,OAAO,IAAI,IAAI,MAAM,KAAK,SAAS,GAAG,IAAI,OAAO,GAAG,KAAK,EAAE,CAAC,CAAC,SAAS;CACxE,QAAQ;EACN,OAAO,GAAG,KAAK,QAAQ,QAAQ,EAAE,EAAE,GAAG,KAAK,QAAQ,QAAQ,EAAE;CAC/D;AACF"}
@@ -1,38 +1,37 @@
1
1
  import { d as KnowledgeGraphNode, l as KnowledgeGraph, u as KnowledgeGraphEdge } from "../types-m2QB86fF.js";
2
2
  //#region src/viz/index.d.ts
3
- interface KnowledgeVizNode extends KnowledgeGraphNode {
3
+ export interface KnowledgeVizNode extends KnowledgeGraphNode {
4
4
  degree: number;
5
5
  community: number;
6
6
  }
7
- interface KnowledgeVizEdge extends KnowledgeGraphEdge {
7
+ export interface KnowledgeVizEdge extends KnowledgeGraphEdge {
8
8
  id: string;
9
9
  }
10
- interface KnowledgeCommunity {
10
+ export interface KnowledgeCommunity {
11
11
  id: number;
12
12
  nodeIds: string[];
13
13
  topTitles: string[];
14
14
  cohesion: number;
15
15
  }
16
- interface KnowledgeVizGraph {
16
+ export interface KnowledgeVizGraph {
17
17
  nodes: KnowledgeVizNode[];
18
18
  edges: KnowledgeVizEdge[];
19
19
  communities: KnowledgeCommunity[];
20
20
  }
21
- interface KnowledgeGap {
21
+ export interface KnowledgeGap {
22
22
  type: 'isolated-node' | 'sparse-community' | 'bridge-node';
23
23
  title: string;
24
24
  nodeIds: string[];
25
25
  suggestion: string;
26
26
  }
27
- interface SurprisingConnection {
27
+ export interface SurprisingConnection {
28
28
  source: KnowledgeVizNode;
29
29
  target: KnowledgeVizNode;
30
30
  score: number;
31
31
  reasons: string[];
32
32
  }
33
- declare function toKnowledgeVizGraph(graph: KnowledgeGraph): KnowledgeVizGraph;
34
- declare function detectKnowledgeGaps(graph: KnowledgeVizGraph, limit?: number): KnowledgeGap[];
35
- declare function findSurprisingConnections(graph: KnowledgeVizGraph, limit?: number): SurprisingConnection[];
33
+ export declare function toKnowledgeVizGraph(graph: KnowledgeGraph): KnowledgeVizGraph;
34
+ export declare function detectKnowledgeGaps(graph: KnowledgeVizGraph, limit?: number): KnowledgeGap[];
35
+ export declare function findSurprisingConnections(graph: KnowledgeVizGraph, limit?: number): SurprisingConnection[];
36
36
  //#endregion
37
- export { KnowledgeCommunity, KnowledgeGap, KnowledgeVizEdge, KnowledgeVizGraph, KnowledgeVizNode, SurprisingConnection, detectKnowledgeGaps, findSurprisingConnections, toKnowledgeVizGraph };
38
37
  //# sourceMappingURL=index.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/viz/index.ts"],"mappings":";;UAEiB,yBAAyB;EACxC;EACA;;UAGe,yBAAyB;EACxC;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf,OAAO;EACP,OAAO;EACP,aAAa;;UAGE;EACf;EACA;EACA;EACA;;UAGe;EACf,QAAQ;EACR,QAAQ;EACR;EACA;;iBAGc,oBAAoB,OAAO,iBAAiB;iBAmB5C,oBAAoB,OAAO,mBAAmB,iBAAa;iBA+C3D,0BACd,OAAO,mBACP,iBACC"}
1
+ {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/viz/index.ts"],"mappings":";;iBAEiB,yBAAyB;EACxC;EACA;;iBAGe,yBAAyB;EACxC;;iBAGe;EACf;EACA;EACA;EACA;;iBAGe;EACf,OAAO;EACP,OAAO;EACP,aAAa;;iBAGE;EACf;EACA;EACA;EACA;;iBAGe;EACf,QAAQ;EACR,QAAQ;EACR;EACA;;wBAGc,oBAAoB,OAAO,iBAAiB;wBAmB5C,oBAAoB,OAAO,mBAAmB,iBAAa;wBA+C3D,0BACd,OAAO,mBACP,iBACC"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-knowledge",
3
- "version": "17.1.10",
3
+ "version": "18.0.0",
4
4
  "description": "Build, search, evaluate, and improve source-backed knowledge bases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-knowledge#readme",
6
6
  "repository": {
@@ -81,57 +81,32 @@
81
81
  "dependencies": {
82
82
  "@types/proper-lockfile": "4.1.4",
83
83
  "proper-lockfile": "4.1.2",
84
- "zod": "4.5.4"
84
+ "zod": "4.6.5",
85
+ "@tangle-network/tcloud": ">=0.8.0 <0.9.0"
85
86
  },
86
87
  "peerDependencies": {
87
- "@tangle-network/agent-eval": ">=0.182.0 <0.200.0",
88
- "@tangle-network/agent-interface": "^2.0.0"
88
+ "@tangle-network/agent-eval": ">=0.201.0 <0.204.0",
89
+ "@tangle-network/agent-interface": "^2.15.0"
89
90
  },
90
91
  "devDependencies": {
91
92
  "@arethetypeswrong/cli": "^0.18.5",
92
- "@biomejs/biome": "^2.5.11",
93
- "@neo4j-labs/agent-memory": "0.4.1",
94
- "@tangle-network/agent-eval": "0.199.0",
95
- "@tangle-network/agent-interface": "2.0.0",
96
- "@types/node": "^26.4.0",
97
- "mem0ai": "3.1.7",
98
- "oxc-parser": "0.147.0",
93
+ "@biomejs/biome": "^2.5.14",
94
+ "@neo4j-labs/agent-memory": "0.5.0",
95
+ "@tangle-network/agent-eval": "0.203.0",
96
+ "@tangle-network/agent-interface": "2.15.0",
97
+ "@types/node": "^26.6.2",
98
+ "mem0ai": "3.3.0",
99
+ "oxc-parser": "0.151.0",
99
100
  "publint": "^0.3.24",
100
- "tsdown": "^0.22.14",
101
+ "tsdown": "^0.23.0",
101
102
  "typescript": "^7.0.2",
102
- "vite": "8.2.2",
103
- "vitest": "^4.1.11",
104
- "yaml": "2.9.0"
105
- },
106
- "pnpm": {
107
- "ignoredBuiltDependencies": [
108
- "@ax-llm/ax",
109
- "esbuild"
110
- ],
111
- "onlyBuiltDependencies": [
112
- "better-sqlite3"
113
- ],
114
- "minimumReleaseAge": 4320,
115
- "minimumReleaseAgeExclude": [
116
- "@tangle-network/agent-core",
117
- "@tangle-network/agent-eval",
118
- "@tangle-network/agent-interface",
119
- "@tangle-network/agent-trace-contract",
120
- "esbuild",
121
- "vite",
122
- "zod@4.5.4"
123
- ],
124
- "overrides": {
125
- "@hono/node-server": "2.0.12",
126
- "esbuild": "0.28.1",
127
- "hono": "4.12.32",
128
- "vite": "8.2.2",
129
- "ws": "8.21.1"
130
- }
103
+ "vite": "8.3.1",
104
+ "vitest": "^5.0.1",
105
+ "yaml": "2.9.1"
131
106
  },
132
107
  "engines": {
133
- "node": ">=20.19.0"
108
+ "node": ">=22.12.0"
134
109
  },
135
110
  "license": "MIT",
136
- "packageManager": "pnpm@10.34.5"
111
+ "packageManager": "pnpm@12.6.0"
137
112
  }