mcp-scraper 0.88.2 → 0.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/CHANGELOG.md +36 -15
  2. package/README.md +6 -5
  3. package/THIRD_PARTY_NOTICES.html +203 -0
  4. package/dist/analytics-repository-2BMT5JNE.js +1 -0
  5. package/dist/bin/api-server.js +2 -41
  6. package/dist/bin/mcp-scraper-cli.js +39 -756
  7. package/dist/bin/mcp-scraper-core.js +1 -60
  8. package/dist/bin/mcp-scraper-install.js +2 -25
  9. package/dist/bin/mcp-stdio-server.js +1 -19
  10. package/dist/bin/paa-harvest.js +1 -41
  11. package/dist/chunk-3GP5CYZX.js +1 -0
  12. package/dist/chunk-4AI7DOS7.js +59 -0
  13. package/dist/chunk-4FROKQJN.js +1 -0
  14. package/dist/chunk-7YGVI5J4.js +21710 -0
  15. package/dist/chunk-CCYWSJNG.js +16 -0
  16. package/dist/chunk-CFI6CXIV.js +182 -0
  17. package/dist/chunk-E2WRWV3A.js +1 -0
  18. package/dist/chunk-GMWKPIYX.js +1172 -0
  19. package/dist/chunk-HDPYG3XV.js +102 -0
  20. package/dist/chunk-HE45FFBU.js +1 -0
  21. package/dist/chunk-HUV2WTRW.js +1 -0
  22. package/dist/chunk-KJQXUZ4Y.js +4 -0
  23. package/dist/chunk-L4CGLFPU.js +4 -0
  24. package/dist/chunk-M22MM4N4.js +84 -0
  25. package/dist/chunk-MASR22K4.js +73 -0
  26. package/dist/chunk-MZN4U5BL.js +1 -0
  27. package/dist/chunk-PUHFVA7P.js +1280 -0
  28. package/dist/chunk-QPWPR5XG.js +10 -0
  29. package/dist/chunk-TMB56NCA.js +1 -0
  30. package/dist/chunk-TXENITMS.js +20 -0
  31. package/dist/chunk-W2BVJ7S2.js +13 -0
  32. package/dist/chunk-WO3N5FH2.js +5 -0
  33. package/dist/chunk-WSCGYRWA.js +2595 -0
  34. package/dist/chunk-X54CQLK2.js +1 -0
  35. package/dist/chunk-XLWNEVUZ.js +27 -0
  36. package/dist/chunk-XPZVJIZ2.js +100 -0
  37. package/dist/chunk-YQZGZBB4.js +1 -0
  38. package/dist/chunk-Z2QGQJS2.js +1 -0
  39. package/dist/db-F2MX63GI.js +1 -0
  40. package/dist/extract-bundle-SNUIHM3J.js +26 -0
  41. package/dist/gmail-service-BZ3H75XC.js +1 -0
  42. package/dist/index.cjs +21750 -6045
  43. package/dist/index.d.cts +14 -14
  44. package/dist/index.d.ts +14 -14
  45. package/dist/index.js +18 -315
  46. package/dist/lead-list-enrichment-repository-S2H3U7T7.js +1 -0
  47. package/dist/location-data-repository-OTWHWMV6.js +1 -0
  48. package/dist/server-RFR2A5UJ.js +7303 -0
  49. package/dist/site-extract-repository-SE776XDC.js +1 -0
  50. package/dist/worker-XUDSM3AL.js +1 -0
  51. package/package.json +17 -124
  52. package/dist/analytics-repository-GGJJCVVP.js +0 -194
  53. package/dist/chunk-4QMUF6XM.js +0 -1013
  54. package/dist/chunk-6HAV7LCE.js +0 -265
  55. package/dist/chunk-ABF2CGOZ.js +0 -113
  56. package/dist/chunk-C5Z4OFKW.js +0 -404
  57. package/dist/chunk-DNM65UCK.js +0 -299
  58. package/dist/chunk-EQGTEHLZ.js +0 -592
  59. package/dist/chunk-F5GQJWZU.js +0 -732
  60. package/dist/chunk-GGZEC22A.js +0 -215
  61. package/dist/chunk-GXBZXWXB.js +0 -184
  62. package/dist/chunk-IHXAXYIS.js +0 -843
  63. package/dist/chunk-K3Z5AQYE.js +0 -683
  64. package/dist/chunk-K45K75OF.js +0 -6
  65. package/dist/chunk-LFW2FRPJ.js +0 -224
  66. package/dist/chunk-MZDNZQWT.js +0 -2078
  67. package/dist/chunk-NVUKO5NN.js +0 -256
  68. package/dist/chunk-OM7HVEJ3.js +0 -26
  69. package/dist/chunk-OPQIGAFB.js +0 -286
  70. package/dist/chunk-OZJMVCDK.js +0 -16
  71. package/dist/chunk-P7FWOMU7.js +0 -505
  72. package/dist/chunk-PGJQDMC2.js +0 -383
  73. package/dist/chunk-PKZS6SHW.js +0 -33139
  74. package/dist/chunk-RJ7JVYKU.js +0 -68
  75. package/dist/chunk-S24LFPL7.js +0 -5262
  76. package/dist/chunk-T3MZISOF.js +0 -240
  77. package/dist/chunk-UZPTGUDV.js +0 -1915
  78. package/dist/chunk-X623GTBV.js +0 -8290
  79. package/dist/chunk-YXNDOQXN.js +0 -4018
  80. package/dist/db-Z34LPZNR.js +0 -284
  81. package/dist/extract-bundle-565SBZCR.js +0 -1003
  82. package/dist/gmail-service-E6ALS7JG.js +0 -25
  83. package/dist/lead-list-enrichment-repository-36RPVV6N.js +0 -67
  84. package/dist/location-data-repository-WPRG62GE.js +0 -34
  85. package/dist/server-SQZ3A7SY.js +0 -86606
  86. package/dist/site-extract-repository-VYFZASPU.js +0 -69
  87. package/dist/worker-LDCAULWL.js +0 -146
@@ -1,683 +0,0 @@
1
- import {
2
- csvRecords
3
- } from "./chunk-RJ7JVYKU.js";
4
- import {
5
- getDb
6
- } from "./chunk-YXNDOQXN.js";
7
-
8
- // src/api/location-data-repository.ts
9
- import { createHash, randomUUID } from "crypto";
10
- var HOSTED_ZIP_DATASET_KIND = "us_zip_groups";
11
- var HOSTED_CENSUS_DATASET_PREFIX = "us_census_places";
12
- var MIN_NATIONWIDE_ZIP_COUNT = 25e3;
13
- var REQUIRED_LOCATION_STATE_CODES = [
14
- "AL",
15
- "AK",
16
- "AZ",
17
- "AR",
18
- "CA",
19
- "CO",
20
- "CT",
21
- "DE",
22
- "DC",
23
- "FL",
24
- "GA",
25
- "HI",
26
- "ID",
27
- "IL",
28
- "IN",
29
- "IA",
30
- "KS",
31
- "KY",
32
- "LA",
33
- "ME",
34
- "MD",
35
- "MA",
36
- "MI",
37
- "MN",
38
- "MS",
39
- "MO",
40
- "MT",
41
- "NE",
42
- "NV",
43
- "NH",
44
- "NJ",
45
- "NM",
46
- "NY",
47
- "NC",
48
- "ND",
49
- "OH",
50
- "OK",
51
- "OR",
52
- "PA",
53
- "RI",
54
- "SC",
55
- "SD",
56
- "TN",
57
- "TX",
58
- "UT",
59
- "VT",
60
- "VA",
61
- "WA",
62
- "WV",
63
- "WI",
64
- "WY"
65
- ];
66
- var STATE_FIPS_BY_ABBR = {
67
- AL: "01",
68
- AK: "02",
69
- AZ: "04",
70
- AR: "05",
71
- CA: "06",
72
- CO: "08",
73
- CT: "09",
74
- DE: "10",
75
- DC: "11",
76
- FL: "12",
77
- GA: "13",
78
- HI: "15",
79
- ID: "16",
80
- IL: "17",
81
- IN: "18",
82
- IA: "19",
83
- KS: "20",
84
- KY: "21",
85
- LA: "22",
86
- ME: "23",
87
- MD: "24",
88
- MA: "25",
89
- MI: "26",
90
- MN: "27",
91
- MS: "28",
92
- MO: "29",
93
- MT: "30",
94
- NE: "31",
95
- NV: "32",
96
- NH: "33",
97
- NJ: "34",
98
- NM: "35",
99
- NY: "36",
100
- NC: "37",
101
- ND: "38",
102
- OH: "39",
103
- OK: "40",
104
- OR: "41",
105
- PA: "42",
106
- RI: "44",
107
- SC: "45",
108
- SD: "46",
109
- TN: "47",
110
- TX: "48",
111
- UT: "49",
112
- VT: "50",
113
- VA: "51",
114
- WA: "53",
115
- WV: "54",
116
- WI: "55",
117
- WY: "56"
118
- };
119
- var REQUIRED_LOCATION_STATE_SET = new Set(REQUIRED_LOCATION_STATE_CODES);
120
- var IMPORT_BATCH_SIZE = 200;
121
- var MAX_SOURCE_URL_LENGTH = 2e3;
122
- var schemaDb = null;
123
- var schemaPromise = null;
124
- function normalizeHostedCityKey(value) {
125
- return value.toLowerCase().replace(/[^a-z0-9]+/g, " ").trim();
126
- }
127
- function stringArray(value) {
128
- if (value == null) return [];
129
- try {
130
- const parsed = JSON.parse(String(value));
131
- return Array.isArray(parsed) ? parsed.map(String) : [];
132
- } catch {
133
- return [];
134
- }
135
- }
136
- function rowToDataset(row) {
137
- return {
138
- id: String(row.id),
139
- kind: String(row.dataset_kind),
140
- sourceUrl: row.source_url == null ? null : String(row.source_url),
141
- sha256: String(row.sha256),
142
- rowCount: Number(row.row_count ?? 0),
143
- groupCount: Number(row.group_count ?? 0),
144
- stateCount: Number(row.state_count ?? 0),
145
- sourceRowCount: Number(row.source_row_count ?? row.row_count ?? 0),
146
- rejectedRowCount: Number(row.rejected_row_count ?? 0),
147
- zipCount: Number(row.zip_count ?? 0),
148
- coveredStates: stringArray(row.covered_states_json),
149
- coverageComplete: Number(row.coverage_complete ?? 0) === 1,
150
- status: String(row.status),
151
- createdAt: String(row.created_at),
152
- completedAt: row.completed_at == null ? null : String(row.completed_at),
153
- error: row.error == null ? null : String(row.error)
154
- };
155
- }
156
- function censusDatasetKind(state) {
157
- return `${HOSTED_CENSUS_DATASET_PREFIX}:${state}`;
158
- }
159
- function normalizedSourceUrl(value) {
160
- const trimmed = value?.trim();
161
- if (!trimmed) return null;
162
- if (trimmed.length > MAX_SOURCE_URL_LENGTH) throw new Error(`sourceUrl must be ${MAX_SOURCE_URL_LENGTH} characters or fewer`);
163
- let parsed;
164
- try {
165
- parsed = new URL(trimmed);
166
- } catch {
167
- throw new Error("sourceUrl must be a valid URL");
168
- }
169
- parsed.username = "";
170
- parsed.password = "";
171
- parsed.search = "";
172
- parsed.hash = "";
173
- return parsed.toString();
174
- }
175
- function normalizedZip(value) {
176
- const digits = value.trim();
177
- if (!/^\d{1,5}$/.test(digits)) return null;
178
- return digits.padStart(5, "0");
179
- }
180
- function normalizedState(value) {
181
- const state = value.trim().toUpperCase();
182
- return REQUIRED_LOCATION_STATE_SET.has(state) ? state : null;
183
- }
184
- function csvColumn(record, names) {
185
- for (const name of names) {
186
- const value = record[name];
187
- if (value !== void 0) return value;
188
- }
189
- return "";
190
- }
191
- function numberOrNull(value) {
192
- if (value === void 0 || value.trim() === "") return null;
193
- const parsed = Number(value);
194
- return Number.isFinite(parsed) && parsed >= 0 ? Math.trunc(parsed) : null;
195
- }
196
- function displayCityFromCensus(name) {
197
- if (/^Nashville-Davidson metropolitan government/i.test(name)) return "Nashville";
198
- return name.replace(/\s+(city|town|village|municipality|borough)$/i, "").trim();
199
- }
200
- function parseHostedCensusPlacesCsv(csv, expectedStateFips) {
201
- const records = csvRecords(csv.replace(/^\uFEFF/, ""));
202
- if (records.length === 0) throw new Error("Census place CSV contains no data rows.");
203
- const headers = new Set(Object.keys(records[0] ?? {}).map((header) => header.trim()));
204
- const required = ["SUMLEV", "STATE", "NAME", "POPESTIMATE2020", "POPESTIMATE2021", "POPESTIMATE2022", "POPESTIMATE2023", "POPESTIMATE2024", "POPESTIMATE2025"];
205
- const missing = required.filter((header) => !headers.has(header));
206
- if (missing.length > 0) throw new Error(`Census place CSV is missing required columns: ${missing.join(", ")}.`);
207
- const placeRecords = records.filter((record) => record.SUMLEV?.trim() === "162");
208
- if (expectedStateFips) {
209
- const expected = expectedStateFips.padStart(2, "0");
210
- const mismatched = placeRecords.filter((record) => record.STATE?.trim().padStart(2, "0") !== expected);
211
- if (mismatched.length > 0) {
212
- const received = [...new Set(mismatched.map((record) => record.STATE?.trim() || "(blank)"))].slice(0, 5);
213
- throw new Error(`Census place CSV STATE does not match requested FIPS ${expected}; received ${received.join(", ")}.`);
214
- }
215
- }
216
- const candidates = placeRecords.map((record) => {
217
- const censusName = record.NAME?.trim() ?? "";
218
- if (!censusName) return null;
219
- const city = displayCityFromCensus(censusName);
220
- const cityKey = normalizeHostedCityKey(city);
221
- if (!cityKey) return null;
222
- const populations = {
223
- 2020: numberOrNull(record.POPESTIMATE2020),
224
- 2021: numberOrNull(record.POPESTIMATE2021),
225
- 2022: numberOrNull(record.POPESTIMATE2022),
226
- 2023: numberOrNull(record.POPESTIMATE2023),
227
- 2024: numberOrNull(record.POPESTIMATE2024),
228
- 2025: numberOrNull(record.POPESTIMATE2025)
229
- };
230
- if (Object.values(populations).every((value) => value === null)) return null;
231
- return {
232
- city,
233
- cityKey,
234
- censusName,
235
- estimatesBase2020: numberOrNull(record.ESTIMATESBASE2020),
236
- populations
237
- };
238
- }).filter((place) => place !== null);
239
- const placesByCityKey = /* @__PURE__ */ new Map();
240
- const populationRank = (place) => place.populations[2025] ?? place.populations[2024] ?? place.populations[2023] ?? place.populations[2022] ?? place.populations[2021] ?? place.populations[2020] ?? place.estimatesBase2020 ?? -1;
241
- for (const candidate of candidates) {
242
- const current = placesByCityKey.get(candidate.cityKey);
243
- if (!current || populationRank(candidate) > populationRank(current) || populationRank(candidate) === populationRank(current) && candidate.censusName.localeCompare(current.censusName) < 0) {
244
- placesByCityKey.set(candidate.cityKey, candidate);
245
- }
246
- }
247
- const places = [...placesByCityKey.values()].sort((a, b) => a.city.localeCompare(b.city));
248
- if (places.length === 0) throw new Error("Census place CSV did not contain any valid place population rows.");
249
- return { sourceRowCount: records.length, places };
250
- }
251
- function parseHostedZipGroupsCsv(csv) {
252
- const records = csvRecords(csv.replace(/^\uFEFF/, ""));
253
- if (records.length === 0) throw new Error("Location ZIP CSV contains no data rows.");
254
- const first = records[0] ?? {};
255
- const headers = new Set(Object.keys(first).map((header) => header.trim().toLowerCase()));
256
- const hasState = ["state_abbr", "state", "state_id"].some((header) => headers.has(header));
257
- const hasCity = ["city", "primary_city"].some((header) => headers.has(header));
258
- const hasZip = ["zipcode", "zip", "zip_code"].some((header) => headers.has(header));
259
- if (!hasState || !hasCity || !hasZip) {
260
- throw new Error("Location ZIP CSV must include state_abbr/state, city, and zipcode/zip columns.");
261
- }
262
- const grouped = /* @__PURE__ */ new Map();
263
- let rowCount = 0;
264
- let rejectedRowCount = 0;
265
- const states = /* @__PURE__ */ new Set();
266
- const uniqueZips = /* @__PURE__ */ new Set();
267
- for (const rawRecord of records) {
268
- const record = Object.fromEntries(Object.entries(rawRecord).map(([key2, value]) => [key2.trim().toLowerCase(), value]));
269
- const state = normalizedState(csvColumn(record, ["state_abbr", "state", "state_id"]));
270
- const city = csvColumn(record, ["city", "primary_city"]).trim();
271
- const zip = normalizedZip(csvColumn(record, ["zipcode", "zip", "zip_code"]));
272
- const county = csvColumn(record, ["county", "county_name"]).trim();
273
- if (!state || !REQUIRED_LOCATION_STATE_SET.has(state) || !city || !zip) {
274
- rejectedRowCount += 1;
275
- continue;
276
- }
277
- const cityKey = normalizeHostedCityKey(city);
278
- if (!cityKey) {
279
- rejectedRowCount += 1;
280
- continue;
281
- }
282
- const key = `${state}\0${cityKey}`;
283
- let group = grouped.get(key);
284
- if (!group) {
285
- group = { city, cityKey, state, zips: /* @__PURE__ */ new Set(), counties: /* @__PURE__ */ new Set() };
286
- grouped.set(key, group);
287
- }
288
- group.zips.add(zip);
289
- if (county) group.counties.add(county);
290
- states.add(state);
291
- uniqueZips.add(zip);
292
- rowCount += 1;
293
- }
294
- if (grouped.size === 0) throw new Error("Location ZIP CSV did not contain any valid US city/ZIP rows.");
295
- return {
296
- sourceRowCount: records.length,
297
- rowCount,
298
- rejectedRowCount,
299
- zipCount: uniqueZips.size,
300
- stateCount: states.size,
301
- coveredStates: [...states].sort(),
302
- groups: [...grouped.values()].map((group) => ({
303
- city: group.city,
304
- cityKey: group.cityKey,
305
- state: group.state,
306
- zips: [...group.zips].sort(),
307
- counties: [...group.counties].sort()
308
- })).sort((a, b) => a.state.localeCompare(b.state) || a.city.localeCompare(b.city))
309
- };
310
- }
311
- function assertNationwideZipCoverage(parsed) {
312
- const covered = new Set(parsed.coveredStates);
313
- const missingStates = REQUIRED_LOCATION_STATE_CODES.filter((state) => !covered.has(state));
314
- const errors = [];
315
- if (missingStates.length > 0) errors.push(`missing required states: ${missingStates.join(", ")}`);
316
- if (parsed.zipCount < MIN_NATIONWIDE_ZIP_COUNT) {
317
- errors.push(`only ${parsed.zipCount} unique ZIPs; at least ${MIN_NATIONWIDE_ZIP_COUNT} are required`);
318
- }
319
- if (errors.length > 0) {
320
- throw new Error(`Location ZIP snapshot is not nationwide and was not activated (${errors.join("; ")}; ${parsed.rejectedRowCount} rejected rows).`);
321
- }
322
- }
323
- async function ensureHostedLocationDataSchema() {
324
- const db = getDb();
325
- if (schemaDb !== db) {
326
- schemaDb = db;
327
- schemaPromise = null;
328
- }
329
- if (!schemaPromise) {
330
- schemaPromise = (async () => {
331
- await db.execute(`
332
- CREATE TABLE IF NOT EXISTS location_data_imports (
333
- id TEXT PRIMARY KEY,
334
- dataset_kind TEXT NOT NULL,
335
- source_url TEXT,
336
- sha256 TEXT NOT NULL,
337
- row_count INTEGER NOT NULL DEFAULT 0,
338
- group_count INTEGER NOT NULL DEFAULT 0,
339
- state_count INTEGER NOT NULL DEFAULT 0,
340
- source_row_count INTEGER NOT NULL DEFAULT 0,
341
- rejected_row_count INTEGER NOT NULL DEFAULT 0,
342
- zip_count INTEGER NOT NULL DEFAULT 0,
343
- covered_states_json TEXT NOT NULL DEFAULT '[]',
344
- coverage_complete INTEGER NOT NULL DEFAULT 0,
345
- status TEXT NOT NULL,
346
- error TEXT,
347
- created_at TEXT NOT NULL DEFAULT (datetime('now')),
348
- completed_at TEXT
349
- )
350
- `);
351
- const importColumns = new Set((await db.execute(`PRAGMA table_info(location_data_imports)`)).rows.map((row) => String(row.name)));
352
- const missingImportColumns = [
353
- ["source_row_count", `INTEGER NOT NULL DEFAULT 0`],
354
- ["rejected_row_count", `INTEGER NOT NULL DEFAULT 0`],
355
- ["zip_count", `INTEGER NOT NULL DEFAULT 0`],
356
- ["covered_states_json", `TEXT NOT NULL DEFAULT '[]'`],
357
- ["coverage_complete", `INTEGER NOT NULL DEFAULT 0`]
358
- ];
359
- for (const [column, definition] of missingImportColumns) {
360
- if (!importColumns.has(column)) await db.execute(`ALTER TABLE location_data_imports ADD COLUMN ${column} ${definition}`);
361
- }
362
- await db.execute(`CREATE INDEX IF NOT EXISTS location_data_imports_kind_created ON location_data_imports(dataset_kind, created_at DESC)`);
363
- await db.execute(`
364
- CREATE TABLE IF NOT EXISTS location_zip_groups (
365
- import_id TEXT NOT NULL REFERENCES location_data_imports(id) ON DELETE CASCADE,
366
- state_abbr TEXT NOT NULL,
367
- city_key TEXT NOT NULL,
368
- city TEXT NOT NULL,
369
- zips_json TEXT NOT NULL,
370
- counties_json TEXT NOT NULL,
371
- PRIMARY KEY(import_id, state_abbr, city_key)
372
- )
373
- `);
374
- await db.execute(`CREATE INDEX IF NOT EXISTS location_zip_groups_state ON location_zip_groups(import_id, state_abbr, city_key)`);
375
- await db.execute(`
376
- CREATE TABLE IF NOT EXISTS location_census_places (
377
- import_id TEXT NOT NULL REFERENCES location_data_imports(id) ON DELETE CASCADE,
378
- state_abbr TEXT NOT NULL,
379
- city_key TEXT NOT NULL,
380
- city TEXT NOT NULL,
381
- census_name TEXT NOT NULL,
382
- estimates_base_2020 INTEGER,
383
- population_2020 INTEGER,
384
- population_2021 INTEGER,
385
- population_2022 INTEGER,
386
- population_2023 INTEGER,
387
- population_2024 INTEGER,
388
- population_2025 INTEGER,
389
- PRIMARY KEY(import_id, state_abbr, city_key)
390
- )
391
- `);
392
- await db.execute(`CREATE INDEX IF NOT EXISTS location_census_places_state ON location_census_places(import_id, state_abbr, city_key)`);
393
- await db.execute(`
394
- CREATE TABLE IF NOT EXISTS location_data_active (
395
- dataset_kind TEXT PRIMARY KEY,
396
- import_id TEXT NOT NULL REFERENCES location_data_imports(id),
397
- activated_at TEXT NOT NULL DEFAULT (datetime('now'))
398
- )
399
- `);
400
- })().catch((error) => {
401
- schemaPromise = null;
402
- throw error;
403
- });
404
- }
405
- await schemaPromise;
406
- }
407
- async function getActiveHostedDataset(kind) {
408
- await ensureHostedLocationDataSchema();
409
- const result = await getDb().execute({
410
- sql: `
411
- SELECT i.*
412
- FROM location_data_active a
413
- JOIN location_data_imports i ON i.id = a.import_id
414
- WHERE a.dataset_kind = ? AND i.status = 'ready'
415
- LIMIT 1
416
- `,
417
- args: [kind]
418
- });
419
- const row = result.rows[0];
420
- return row ? rowToDataset(row) : null;
421
- }
422
- async function getActiveHostedLocationDataset() {
423
- return getActiveHostedDataset(HOSTED_ZIP_DATASET_KIND);
424
- }
425
- async function listActiveHostedLocationDatasets() {
426
- await ensureHostedLocationDataSchema();
427
- const result = await getDb().execute(`
428
- SELECT i.*
429
- FROM location_data_active a
430
- JOIN location_data_imports i ON i.id = a.import_id
431
- WHERE i.status = 'ready'
432
- ORDER BY i.dataset_kind ASC
433
- `);
434
- return result.rows.map((row) => rowToDataset(row));
435
- }
436
- async function getHostedZipGroups(stateInput) {
437
- const state = normalizedState(stateInput);
438
- if (!state) throw new Error("state must be a two-letter US state abbreviation");
439
- const dataset = await getActiveHostedLocationDataset();
440
- if (!dataset) return { dataset: null, groups: /* @__PURE__ */ new Map() };
441
- const result = await getDb().execute({
442
- sql: `
443
- SELECT city, city_key, state_abbr, zips_json, counties_json
444
- FROM location_zip_groups
445
- WHERE import_id = ? AND state_abbr = ?
446
- ORDER BY city COLLATE NOCASE ASC
447
- `,
448
- args: [dataset.id, state]
449
- });
450
- const groups = /* @__PURE__ */ new Map();
451
- for (const row of result.rows) {
452
- const cityKey = String(row.city_key);
453
- groups.set(cityKey, {
454
- city: String(row.city),
455
- cityKey,
456
- state: String(row.state_abbr),
457
- zips: JSON.parse(String(row.zips_json)),
458
- counties: JSON.parse(String(row.counties_json))
459
- });
460
- }
461
- return { dataset, groups };
462
- }
463
- async function importHostedZipGroupsCsv(input) {
464
- await ensureHostedLocationDataSchema();
465
- const sourceUrl = normalizedSourceUrl(input.sourceUrl);
466
- const sha256 = createHash("sha256").update(input.csv).digest("hex");
467
- const active = await getActiveHostedLocationDataset();
468
- if (active?.sha256 === sha256) return { dataset: active, duplicate: true };
469
- const parsed = parseHostedZipGroupsCsv(input.csv);
470
- assertNationwideZipCoverage(parsed);
471
- const id = `loc_${randomUUID().replaceAll("-", "")}`;
472
- const db = getDb();
473
- await db.execute({
474
- sql: `
475
- INSERT INTO location_data_imports (
476
- id, dataset_kind, source_url, sha256, row_count, group_count, state_count,
477
- source_row_count, rejected_row_count, zip_count, covered_states_json, coverage_complete, status
478
- ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 1, 'loading')
479
- `,
480
- args: [
481
- id,
482
- HOSTED_ZIP_DATASET_KIND,
483
- sourceUrl,
484
- sha256,
485
- parsed.rowCount,
486
- parsed.groups.length,
487
- parsed.stateCount,
488
- parsed.sourceRowCount,
489
- parsed.rejectedRowCount,
490
- parsed.zipCount,
491
- JSON.stringify(parsed.coveredStates)
492
- ]
493
- });
494
- try {
495
- for (let offset = 0; offset < parsed.groups.length; offset += IMPORT_BATCH_SIZE) {
496
- const batch = parsed.groups.slice(offset, offset + IMPORT_BATCH_SIZE).map((group) => ({
497
- sql: `
498
- INSERT INTO location_zip_groups (
499
- import_id, state_abbr, city_key, city, zips_json, counties_json
500
- ) VALUES (?, ?, ?, ?, ?, ?)
501
- `,
502
- args: [id, group.state, group.cityKey, group.city, JSON.stringify(group.zips), JSON.stringify(group.counties)]
503
- }));
504
- await db.batch(batch, "write");
505
- }
506
- await db.batch([
507
- {
508
- sql: `UPDATE location_data_imports SET status = 'ready', completed_at = datetime('now'), error = NULL WHERE id = ? AND status = 'loading'`,
509
- args: [id]
510
- },
511
- {
512
- sql: `
513
- INSERT INTO location_data_active (dataset_kind, import_id, activated_at)
514
- VALUES (?, ?, datetime('now'))
515
- ON CONFLICT(dataset_kind) DO UPDATE SET import_id = excluded.import_id, activated_at = excluded.activated_at
516
- `,
517
- args: [HOSTED_ZIP_DATASET_KIND, id]
518
- }
519
- ], "write");
520
- } catch (error) {
521
- const message = error instanceof Error ? error.message : String(error);
522
- await db.execute({
523
- sql: `UPDATE location_data_imports SET status = 'failed', error = ?, completed_at = datetime('now') WHERE id = ?`,
524
- args: [message.slice(0, 1e3), id]
525
- }).catch(() => void 0);
526
- throw error;
527
- }
528
- const result = await db.execute({ sql: "SELECT * FROM location_data_imports WHERE id = ? LIMIT 1", args: [id] });
529
- const row = result.rows[0];
530
- if (!row) throw new Error("Hosted location import completed without a durable metadata row.");
531
- return { dataset: rowToDataset(row), duplicate: false };
532
- }
533
- async function importHostedCensusPlacesCsv(input) {
534
- await ensureHostedLocationDataSchema();
535
- const state = normalizedState(input.state);
536
- if (!state) throw new Error("state must be a two-letter US state abbreviation");
537
- const kind = censusDatasetKind(state);
538
- const sourceUrl = normalizedSourceUrl(input.sourceUrl);
539
- const sha256 = createHash("sha256").update(input.csv).digest("hex");
540
- const active = await getActiveHostedDataset(kind);
541
- if (active?.sha256 === sha256) return { dataset: active, duplicate: true };
542
- const stateFips = STATE_FIPS_BY_ABBR[state];
543
- const parsed = parseHostedCensusPlacesCsv(input.csv, stateFips);
544
- const id = `loc_${randomUUID().replaceAll("-", "")}`;
545
- const db = getDb();
546
- await db.execute({
547
- sql: `
548
- INSERT INTO location_data_imports (
549
- id, dataset_kind, source_url, sha256, row_count, group_count, state_count,
550
- source_row_count, rejected_row_count, zip_count, covered_states_json, coverage_complete, status
551
- ) VALUES (?, ?, ?, ?, ?, ?, 1, ?, 0, 0, ?, 1, 'loading')
552
- `,
553
- args: [id, kind, sourceUrl, sha256, parsed.places.length, parsed.places.length, parsed.sourceRowCount, JSON.stringify([state])]
554
- });
555
- try {
556
- for (let offset = 0; offset < parsed.places.length; offset += IMPORT_BATCH_SIZE) {
557
- const batch = parsed.places.slice(offset, offset + IMPORT_BATCH_SIZE).map((place) => ({
558
- sql: `
559
- INSERT INTO location_census_places (
560
- import_id, state_abbr, city_key, city, census_name, estimates_base_2020,
561
- population_2020, population_2021, population_2022, population_2023,
562
- population_2024, population_2025
563
- ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
564
- `,
565
- args: [
566
- id,
567
- state,
568
- place.cityKey,
569
- place.city,
570
- place.censusName,
571
- place.estimatesBase2020,
572
- place.populations[2020],
573
- place.populations[2021],
574
- place.populations[2022],
575
- place.populations[2023],
576
- place.populations[2024],
577
- place.populations[2025]
578
- ]
579
- }));
580
- await db.batch(batch, "write");
581
- }
582
- await db.batch([
583
- {
584
- sql: `UPDATE location_data_imports SET status = 'ready', completed_at = datetime('now'), error = NULL WHERE id = ? AND status = 'loading'`,
585
- args: [id]
586
- },
587
- {
588
- sql: `
589
- INSERT INTO location_data_active (dataset_kind, import_id, activated_at)
590
- VALUES (?, ?, datetime('now'))
591
- ON CONFLICT(dataset_kind) DO UPDATE SET import_id = excluded.import_id, activated_at = excluded.activated_at
592
- `,
593
- args: [kind, id]
594
- }
595
- ], "write");
596
- } catch (error) {
597
- const message = error instanceof Error ? error.message : String(error);
598
- await db.execute({
599
- sql: `UPDATE location_data_imports SET status = 'failed', error = ?, completed_at = datetime('now') WHERE id = ?`,
600
- args: [message.slice(0, 1e3), id]
601
- }).catch(() => void 0);
602
- throw error;
603
- }
604
- const result = await db.execute({ sql: "SELECT * FROM location_data_imports WHERE id = ? LIMIT 1", args: [id] });
605
- const row = result.rows[0];
606
- if (!row) throw new Error("Hosted Census import completed without a durable metadata row.");
607
- return { dataset: rowToDataset(row), duplicate: false };
608
- }
609
- async function queryHostedLocationMarkets(input) {
610
- const state = normalizedState(input.state);
611
- if (!state) throw new Error("state must be a two-letter US state abbreviation");
612
- if (!Number.isSafeInteger(input.minPopulation) || input.minPopulation < 0) throw new Error("minPopulation must be a non-negative integer");
613
- if (!Number.isSafeInteger(input.maxResults) || input.maxResults < 1 || input.maxResults > 100) throw new Error("maxResults must be between 1 and 100");
614
- const populationDataset = await getActiveHostedDataset(censusDatasetKind(state));
615
- const needsZipDataset = input.includeZipGroups || Boolean(input.zip);
616
- const zipLookup = needsZipDataset ? await getHostedZipGroups(state) : { dataset: null, groups: /* @__PURE__ */ new Map() };
617
- if (!populationDataset) {
618
- return { markets: [], populationDataset: null, zipDataset: zipLookup.dataset, warnings: [] };
619
- }
620
- const result = await getDb().execute({
621
- sql: `
622
- SELECT *
623
- FROM location_census_places
624
- WHERE import_id = ? AND state_abbr = ?
625
- ORDER BY city COLLATE NOCASE ASC
626
- `,
627
- args: [populationDataset.id, state]
628
- });
629
- const cityQuery = normalizeHostedCityKey(input.cityQuery ?? "");
630
- const zip = input.zip?.trim();
631
- const populationColumn = `population_${input.populationYear}`;
632
- const markets = result.rows.map((row) => {
633
- const populationValue = row[populationColumn];
634
- if (populationValue == null) return null;
635
- const population = Number(populationValue);
636
- if (!Number.isSafeInteger(population) || population < input.minPopulation) return null;
637
- const city = String(row.city);
638
- const normalizedKey = String(row.city_key);
639
- if (cityQuery && !normalizedKey.includes(cityQuery)) return null;
640
- const zipGroup = zipLookup.groups.get(normalizedKey);
641
- const zips = zipGroup?.zips ?? [];
642
- if (zip && !zips.includes(zip)) return null;
643
- return {
644
- city,
645
- state,
646
- location: `${city}, ${state}`,
647
- cityKey: `${city}|${state}`,
648
- censusName: String(row.census_name),
649
- population,
650
- populationYear: input.populationYear,
651
- estimatesBase2020: row.estimates_base_2020 == null ? null : Number(row.estimates_base_2020),
652
- zips: input.includeZipGroups ? zips : [],
653
- counties: input.includeZipGroups ? zipGroup?.counties ?? [] : []
654
- };
655
- }).filter((market) => market !== null).sort((a, b) => b.population - a.population || a.city.localeCompare(b.city)).slice(0, input.maxResults);
656
- const warnings = [];
657
- if (input.includeZipGroups && zipLookup.dataset && markets.some((market) => market.zips.length === 0)) {
658
- warnings.push("Some hosted Census places did not match the hosted ZIP city groups.");
659
- }
660
- return {
661
- markets,
662
- populationDataset,
663
- zipDataset: zipLookup.dataset,
664
- warnings
665
- };
666
- }
667
-
668
- export {
669
- HOSTED_ZIP_DATASET_KIND,
670
- HOSTED_CENSUS_DATASET_PREFIX,
671
- MIN_NATIONWIDE_ZIP_COUNT,
672
- REQUIRED_LOCATION_STATE_CODES,
673
- parseHostedCensusPlacesCsv,
674
- parseHostedZipGroupsCsv,
675
- ensureHostedLocationDataSchema,
676
- getActiveHostedDataset,
677
- getActiveHostedLocationDataset,
678
- listActiveHostedLocationDatasets,
679
- getHostedZipGroups,
680
- importHostedZipGroupsCsv,
681
- importHostedCensusPlacesCsv,
682
- queryHostedLocationMarkets
683
- };
@@ -1,6 +0,0 @@
1
- // src/version.ts
2
- var PACKAGE_VERSION = "0.88.2";
3
-
4
- export {
5
- PACKAGE_VERSION
6
- };