@gscdump/cli 3.8.0 → 4.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/README.md +115 -30
  2. package/bin/gscdump.mjs +21 -1
  3. package/dist/analysis-local.mjs +172 -53
  4. package/dist/auth-state.mjs +21 -6
  5. package/dist/auth.mjs +27 -18
  6. package/dist/bing-data.mjs +6 -2
  7. package/dist/bing-hosted.mjs +1 -1
  8. package/dist/cli-args.mjs +8 -0
  9. package/dist/cli.d.mts +5 -1
  10. package/dist/cli.mjs +41 -10
  11. package/dist/cloud-google.mjs +8 -1
  12. package/dist/command-meta.mjs +4 -4
  13. package/dist/command-registry.mjs +86 -10
  14. package/dist/commands/analyze.mjs +84 -43
  15. package/dist/commands/auth.mjs +46 -37
  16. package/dist/commands/bing.mjs +3 -3
  17. package/dist/commands/compact.mjs +12 -8
  18. package/dist/commands/config.mjs +14 -14
  19. package/dist/commands/doctor.mjs +26 -26
  20. package/dist/commands/dump.mjs +203 -105
  21. package/dist/commands/entities.mjs +10 -113
  22. package/dist/commands/gc.mjs +4 -3
  23. package/dist/commands/indexing-urls.mjs +185 -0
  24. package/dist/commands/indexing.mjs +12 -8
  25. package/dist/commands/init.mjs +77 -18
  26. package/dist/commands/inspect.mjs +97 -110
  27. package/dist/commands/mcp.mjs +37 -57
  28. package/dist/commands/papercut.mjs +1 -1
  29. package/dist/commands/profile.mjs +9 -17
  30. package/dist/commands/query.mjs +219 -210
  31. package/dist/commands/report.mjs +68 -85
  32. package/dist/commands/rollups.mjs +9 -6
  33. package/dist/commands/sitemaps.mjs +51 -104
  34. package/dist/commands/sites.mjs +40 -24
  35. package/dist/commands/stats.mjs +14 -17
  36. package/dist/commands/store-purge.mjs +9 -5
  37. package/dist/commands/store.mjs +0 -12
  38. package/dist/commands/sync.mjs +774 -296
  39. package/dist/config.mjs +7 -4
  40. package/dist/context.mjs +51 -21
  41. package/dist/coverage.mjs +251 -0
  42. package/dist/dump-bing.mjs +71 -0
  43. package/dist/dump-writers.mjs +246 -0
  44. package/dist/error-handler.mjs +108 -38
  45. package/dist/filters.mjs +125 -0
  46. package/dist/hosted-site.mjs +91 -0
  47. package/dist/inspect-urls.mjs +76 -0
  48. package/dist/inspection-record.mjs +93 -0
  49. package/dist/local-entities.mjs +658 -0
  50. package/dist/mcp/errors.mjs +71 -18
  51. package/dist/mcp/handlers/diagnostics.mjs +15 -12
  52. package/dist/mcp/handlers/reports.mjs +33 -42
  53. package/dist/mcp/server/index.mjs +80 -38
  54. package/dist/mcp/types.mjs +7 -7
  55. package/dist/package.mjs +1 -1
  56. package/dist/quota-ledger.mjs +370 -0
  57. package/dist/render/analysis.mjs +8 -1
  58. package/dist/render/report.mjs +8 -1
  59. package/dist/request-pacer.mjs +29 -0
  60. package/dist/route.mjs +301 -0
  61. package/dist/sitemap.mjs +4 -0
  62. package/dist/sql-views.mjs +117 -0
  63. package/dist/store-sites.mjs +134 -0
  64. package/dist/sync-plan.mjs +129 -0
  65. package/dist/sync-run.mjs +96 -0
  66. package/dist/table-sources.mjs +57 -0
  67. package/dist/token-info.mjs +54 -0
  68. package/dist/utils.mjs +28 -19
  69. package/dist/window.mjs +159 -0
  70. package/package.json +17 -14
  71. package/skills/gscdump/SKILL.md +172 -37
  72. package/dist/commands/export.mjs +0 -76
  73. package/dist/native-duckdb.mjs +0 -46
@@ -10,6 +10,8 @@ It keeps a local Parquet Store for Google rows. Every command has `--help`.
10
10
  For `query`, `-s` means `--site`, `-d` means `--dimensions`, and `-f` means `--format`.
11
11
  Use `--start` and `--end` for dates. `--site=SITE` also works.
12
12
  Use each option once, with either its short or long spelling.
13
+ Put options after the subcommand name: `gscdump store stats --json`, not `gscdump store --json stats`.
14
+ A failed command prints one `Error:` line to stderr and exits 1.
13
15
  Example: `gscdump query --site=SITE --start=DATE --end=DATE -d page -f json`.
14
16
 
15
17
  ## Start each task
@@ -17,9 +19,11 @@ Example: `gscdump query --site=SITE --start=DATE --end=DATE -d page -f json`.
17
19
  1. Before reading traffic, run `gscdump auth status --json`. Do this even when the user says authentication works.
18
20
  2. Keep the requested Site, dates, dimensions, and task scope. A request for pages does not need query dimensions.
19
21
  3. Before local queries, check coverage with `gscdump store stats --site SITE --json`.
20
- Use `gscdump sync --site SITE --status --json` when you need sync-state details.
22
+ Use `gscdump sync --site SITE --status --json` when you need coverage, gaps, or sync-state details.
23
+ On a fresh Store, `store stats` exits 1 and says it has no data. Continue with the bounded sync.
21
24
  4. Read the table dimensions and watermarks. Sync only missing tables and the requested dates, once per task.
22
25
  5. Use `sync --json`. Read its completion result before deciding what to do next. Never repeat a successful sync.
26
+ 6. If the user asks for saved rows, check `meta.source: "local"` in the query result. A successful query can answer live when its table has no synced data.
23
27
 
24
28
  If the task only asks about deletion, explain the scope and ask for consent.
25
29
  You may read Store metadata with `store stats` and `sync --status`.
@@ -29,6 +33,8 @@ Call the local data directory the Store in your answer.
29
33
  ## Authentication mode
30
34
 
31
35
  Check `gscdump auth status --json` before queries. Reuse the user's selected mode.
36
+ In local mode, `googleAuthenticated: true` means Google accepted the credentials. When it is false, `googleError` says why.
37
+ In cloud mode, `hostedSync` lists each gscdump.com Site with `syncStatus` and `syncProgress`. `gscdump sites` shows the same progress. The CLI does not read the hosted Store: cloud mode queries go to the live API through gscdump.com.
32
38
 
33
39
  | Mode | Credentials | Query path |
34
40
  | --- | --- | --- |
@@ -68,21 +74,45 @@ After cloud Bing login opens a browser, use `bing status --site s_SITE_ID` to co
68
74
  Hosted Bing commands use the API's plan and preview access rules.
69
75
  Hosted connection verification uses `bing verify --site s_SITE_ID`.
70
76
  Google Indexing API and Site Verification commands require local mode.
71
- Hosted sitemap membership and history require hosted credentials.
77
+ Hosted sitemap reads and `indexing urls` require hosted credentials. Their `--site` takes a Site URL, such as `example.com`.
72
78
 
73
79
  ## Data boundaries
74
80
 
75
81
  - `sync`, `query --live`, `analyze --live`, `report --live`, `sites`,
76
82
  `sitemaps`, and `inspect` use the selected authentication mode.
77
83
  - Google Indexing API requests require local credentials. `indexing quota` only prints documented limits and needs no authentication.
78
- - `query`, `analyze`, `report`, `dump`, and `store` read the local Store by
79
- default. If the Store has no rows for the Site, sync first or pass `--live`.
84
+ - `dump` and `store` read the local Store only.
85
+ - `query`, `analyze`, and `report` pick a source for each run. See [Routing](#routing).
80
86
  - Google returns a 2 to 3 day data delay. Default windows end three days ago.
81
87
  - Google omits low-volume rows. Pagination cannot recover them.
82
88
  - URL Inspection is limited to 2,000 requests per Site per day.
83
89
  - `indexing submit` and `indexing remove` are only for job posting and
84
90
  livestream pages. Google rejects other content.
85
91
 
92
+ ## Routing
93
+
94
+ Login is optional. A user logs in (Google or hosted) or syncs a local Store.
95
+ `query`, `analyze`, and `report` choose one source for each run. One run never
96
+ mixes Store rows and live rows.
97
+
98
+ | Store data for the Site | Google connected | Result |
99
+ |---|---|---|
100
+ | Covers every date the run needs | any | Answers from the Store, also while a sync runs |
101
+ | None for the tables the run needs | yes | Answers from the live API. stderr says so, and JSON has `meta.source: "live"` |
102
+ | None | no | Stops. Next command: `gscdump init` |
103
+ | Some dates missing | yes | Stops with the exact `gscdump sync` command, or pass `--live` |
104
+ | Some dates missing, a sync is running | yes | Stops with `Sync running: 41 of 90 days done.` Run again later, or pass `--live` |
105
+
106
+ - `--live` always asks Search Console. It needs Google auth.
107
+ - `query --sql` reads the Store only. It stops when no table it names has data.
108
+ - JSON output carries `meta.source`: `local` or `live`.
109
+ - A stop with `--format json` or `--json` prints `{ "error": { "code", "message", "nextCommand" } }` on stdout and exits 1.
110
+ Codes: `NOT_CONNECTED`, `STORE_RANGE_NOT_COVERED` (with `missingDates`), `SYNC_RUNNING` (with `sync.done` and `sync.total`), `NO_SYNCED_DATA`, `STORE_ONLY`, `LIVE_ONLY`.
111
+ Partial coverage and a running sync are normal progress. Run `nextCommand`, or tell the user to.
112
+ - A sync is running only while its heartbeat is recent. A killed sync does not block reads.
113
+ - Every Search Analytics, URL Inspection, and Indexing API call spends the shared quota ledger in the data dir.
114
+ When a quota is spent, the command stops at once and says when the quota resets.
115
+
86
116
  ## Get the binary
87
117
 
88
118
  ```sh
@@ -134,10 +164,18 @@ Cloud mode requires hosted access. Pro is free during beta, then paid after laun
134
164
 
135
165
  ## Site identifiers
136
166
 
137
- For Google, use the exact value that `gscdump sites` prints.
167
+ For Google, pass the Site as the user writes it: `--site example.com`.
168
+ `https://example.com`, `www.example.com`, and `Example.com` resolve to the same Site.
169
+ Do not add `sc-domain:`.
138
170
 
139
- - Domain property: `sc-domain:example.com`
140
- - URL-prefix property: `https://example.com/` (trailing slash included)
171
+ - The CLI checks Sites in the Store first. It asks Search Console only when the Store has no match and auth exists.
172
+ - A full Site URL that names a property exactly, such as `https://example.com/`, picks that property.
173
+ - Otherwise, if a domain property and a URL-prefix property both match, the CLI picks the one with Store data, then the domain property.
174
+ - If a URL-prefix property has a path, include the path: `--site example.com/blog`.
175
+ - If the input is a subdomain inside a domain property, the command fails. The error names the parent Site and a `--page` filter to use.
176
+ - If the input matches more than one Site, the command fails and lists them. Pass one of the listed Site URLs.
177
+ - Without `--site` and `defaultSite`, a command in a terminal shows a picker.
178
+ Without a terminal, the command fails with `Pass --site. Sites: ...`, unless only one Site is available.
141
179
 
142
180
  For cloud Bing commands, use a Site ID from `gscdump bing sites`, such as `s_SITE_ID`.
143
181
  For local Bing commands, use the full verified Site URL from `gscdump bing sites --mode local`.
@@ -146,7 +184,7 @@ Bing commands require their own explicit `--site`; the Google `defaultSite` sett
146
184
  Set a default once to drop `--site` from later commands:
147
185
 
148
186
  ```sh
149
- gscdump config set defaultSite sc-domain:example.com
187
+ gscdump config set defaultSite example.com
150
188
  ```
151
189
 
152
190
  ## Output
@@ -174,17 +212,17 @@ Do not rewrite rows, estimate metrics, or add manually calculated totals.
174
212
  | `gscdump query` | Rows by page, query, date, country, or device |
175
213
  | `gscdump analyze <id>` | One Analyzer over the Store or live rows |
176
214
  | `gscdump report <id>` | A Report that composes several Analyzers |
177
- | `gscdump inspect <url>` | URL Inspection with Indexing Evidence |
215
+ | `gscdump inspect <url...>` | URL Inspection with Indexing Evidence, saved to the Store |
178
216
  | `gscdump sitemaps` | List, submit, delete, and probe sitemaps |
179
- | `gscdump indexing` | Indexing API notifications and quota |
180
- | `gscdump dump` | Export Store tables to Parquet, CSV, JSON, or NDJSON |
217
+ | `gscdump indexing` | Indexing API notifications and quota; hosted URL Inspection results |
218
+ | `gscdump dump` | Export Store tables, inspections, sitemaps, and Bing data as Parquet, CSV, JSON, NDJSON, SQLite, or DuckDB |
181
219
  | `gscdump store` | Store stats, compaction, garbage collection, resets |
182
- | `gscdump entities` | Snapshot URL inspections into the entity store |
220
+ | `gscdump entities` | Read saved inspections; snapshot Indexing API metadata |
183
221
  | `gscdump config` | Defaults such as `defaultSite`, `dataDir`, `defaultLimit` |
184
222
  | `gscdump profile` | Separate credential and config directories |
185
223
  | `gscdump auth` | `status`, `login`, `logout`, `refresh` |
186
224
  | `gscdump doctor` | Health checks for auth, scopes, Store, and reachability |
187
- | `gscdump init` | Interactive first-time setup |
225
+ | `gscdump init` | First-time setup. Without a terminal it never prompts: it uses BYOK env credentials or fails with the auth command |
188
226
  | `gscdump mcp` | Start Google MCP tools with the selected authentication |
189
227
  | `gscdump skill install` | Copy this skill into an agent skill directory |
190
228
  | `gscdump papercut` | Report a CLI problem to gscdump.com |
@@ -192,63 +230,145 @@ Do not rewrite rows, estimate metrics, or add manually calculated totals.
192
230
  `gscdump login`, `gscdump logout`, and `gscdump status` are top-level aliases of
193
231
  the matching `auth` subcommands.
194
232
  The MCP server does not expose Bing tools. Use `gscdump bing` commands through this skill.
233
+ `gscdump mcp` starts without authentication. Its Google tools then return an error with the command to run.
234
+ A failed Google request returns its status, Google's explanation, and the next step in the tool result.
195
235
 
196
236
  ## Sync before local analysis
197
237
 
198
238
  ```sh
199
- gscdump store stats --site sc-domain:example.com --json
200
- gscdump sync --site sc-domain:example.com --days 90 \
201
- --tables pages,queries,page_queries,countries --json
202
- gscdump sync --site sc-domain:example.com --status --json
239
+ gscdump store stats --site example.com --json
240
+ gscdump sync --site example.com --status --json
241
+ gscdump sync --site example.com --json
203
242
  ```
204
243
 
205
- - Pass an explicit `--tables` list. The default list has a known daily-totals
206
- limitation.
207
- - `--full` backfills the 450 days Google keeps.
244
+ - A plain sync catches up. Each table runs from its oldest synced date to the
245
+ latest date Google has finalized (Pacific time, about 3 days late). A table
246
+ with no history starts 28 days back. Newest dates come first.
247
+ - `--days N`, `--start`, and `--end` pick a range instead. `--full` fetches the
248
+ 16 months Google keeps, plus 14 days Google often still serves.
249
+ - Sync covers every table and search type by default. Pass `--tables` and
250
+ `--types` to sync less. Sync skips table and type pairs Google cannot answer.
251
+ - Sync paces Google calls: 8 in flight and 600 per minute across all tables.
252
+ `--requests-per-minute N` changes the rate.
253
+ Sync does not retry a quota 403. The quota ledger stops the run instead.
254
+ - Every call goes through a quota ledger in the Store directory. If Google
255
+ refuses a call for quota, or the run reaches `--max-calls N`, sync stops,
256
+ keeps the rest `pending`, and exits 0. `status` in `sync --json` is then
257
+ `partial` and `stopped` says why. Run the same command later to continue.
258
+ Exit 1 means real failures: read `failed` dates in `sync --status --json`.
259
+ - Sync also saves the sitemap list, sitemap URLs, and URL Inspection results.
260
+ It inspects up to 50 due URLs per run: never-inspected sitemap URLs first,
261
+ then pages with impressions, then the oldest results. `--inspect-limit N`
262
+ changes that; Google allows 2,000 per Site per day. `--no-sitemaps` and
263
+ `--no-inspections` skip those steps.
264
+ - `coverage` in `sync --json` and `sync --status --json` says how much the
265
+ Store holds. Partial coverage is normal progress. Never report data as
266
+ complete unless its `kind` is `complete`.
267
+ - A day Google still updates stays `pending`; the next sync fetches it again.
208
268
  - Sync skips completed dates. `--force` refreshes them. `--retry-failed`
209
- reruns only failed dates.
210
- - `--dry-run` prints the planned work without calling Google.
211
- - Use the user's date range. The 90-day example does not authorize a wider sync.
269
+ reruns only failed dates. A plain sync also retries failed dates.
270
+ - `--dry-run` prints the planned dates and the fewest calls without calling Google.
271
+ - `--all-sites` syncs every verified Site, one after another.
272
+ - Use the user's date range. If the user names a range, pass `--start` and `--end`, not `--full`.
273
+ - If the user excludes rollups, pass `--no-rollups` on the sync command.
212
274
  - Empty Store metadata is expected before the first sync. It does not prove zero traffic.
213
275
 
214
276
  ## Query rows
215
277
 
216
278
  ```sh
217
- gscdump query --site sc-domain:example.com --dimensions page,query \
279
+ gscdump query --site example.com --dimensions page,query \
218
280
  --start 2026-08-01 --end 2026-08-28 --limit 1000 --format json
219
281
  ```
220
282
 
221
283
  - Dimension names are singular: `page`, `query`, `date`, `country`, `device`.
284
+ - A page breakdown uses `--tables pages` for sync and `-d page` for query.
285
+ `-d page,query` needs `page_queries`; syncing only `pages` does not fill that table.
222
286
  - Filters: `--query`, `--page`, `--country`, `--device`,
223
287
  `--search-appearance`. Prefixes: bare equals, `~` contains, `!~` not
224
288
  contains, `re:` regex, `!re:` not regex, `!` not equals.
225
- - `--live` bypasses the Store. `--type` selects a search type.
289
+ - `--page` takes a path or a full URL. The Store compares paths.
290
+ - Without dates, `query` reads the 28 days ending on the newest synced day.
291
+ - `--live` bypasses the Store. `--type` selects a search type. The default is `web`.
226
292
  `--data-state` and `--aggregation-type` apply to live mode only.
227
293
  - Metrics already include clicks, impressions, CTR, and position. There is no `--metrics` option.
228
- - If Store coverage is missing, read the JSON error and its bounded `nextArgs` before syncing.
294
+ - If Store coverage is missing, read the JSON error and run its `nextCommand`. It syncs only the missing dates and tables.
229
295
  Do not switch dimensions to make a failed query succeed.
230
296
  - `--explain` prints the request body or planned SQL without executing.
231
- - `--sql` runs raw DuckDB SQL over the Store with `{{FILES}}` as the file list.
297
+
298
+ ## SQL over the Store
299
+
300
+ ```sh
301
+ gscdump query --schema --format json
302
+ gscdump query --format json --sql "SELECT search_type, SUM(clicks) AS clicks,
303
+ gsc_position(sum_position, impressions) AS position
304
+ FROM pages WHERE date >= DATE '2026-08-01' GROUP BY search_type"
305
+ ```
306
+
307
+ - `--sql` runs DuckDB SQL over one view per Store table: `pages`, `queries`,
308
+ `page_queries`, `countries`, `dates`, `hourly_pages`, and the
309
+ `search_appearance*` tables. Join views on `site`, `search_type`, `url`, and `date`.
310
+ - `--schema` lists each view, its columns, its Sites, and its date range.
311
+ - Every view has `site` (the Site URL) and `search_type`. The Store keeps
312
+ every search type, so filter or group by `search_type`. A plain `SUM` adds
313
+ web, image, and Discover rows together.
314
+ - `url` holds the page path. `page` is the same value.
315
+ - `sum_position` is the zero-based position times impressions. Use
316
+ `gsc_position(sum_position, impressions)` for the average position. It adds 1
317
+ and weights by impressions. Never average a per-row position.
318
+ - The views cover every Site in the Store. `--site` and `--type` narrow them.
319
+ - Dates return as `YYYY-MM-DD`. Integers return as numbers.
320
+ - If a query names a table with no synced data, the JSON has a `warnings` list.
321
+
322
+ ## Export the Store
323
+
324
+ ```sh
325
+ gscdump dump --site example.com --format parquet --out ./export
326
+ gscdump dump --all-sites --format sqlite --out ./export
327
+ ```
328
+
329
+ - `dump` reads only the Store. It never calls Google to fill a gap.
330
+ - Every exported row has `site` and `search_type`.
331
+ - File formats write `<site>/<search_type>/<table>.<ext>` and
332
+ `<site>/<dataset>.<ext>` for inspections, sitemaps, and Indexing API metadata.
333
+ - `csv`, `json`, and `ndjson` rows also have `position`: `sum_position / impressions + 1`.
334
+ - `sqlite` and `duckdb` write one file, `gscdump.sqlite` or `gscdump.duckdb`,
335
+ with one table per dataset for every Site and search type.
336
+ - `manifest.json` lists every dataset with its row count, plus the coverage that `sync --status --json` reports. Partial coverage is progress: daily sync fills the rest.
337
+ - `sites.json` lists each exported Site URL with its Store ID.
232
338
 
233
339
  ## Analyze and report
234
340
 
235
341
  ```sh
236
342
  gscdump report list --json
237
- gscdump report opportunities --site sc-domain:example.com --json
238
- gscdump report movers --site sc-domain:example.com --period 28d --vs prev-period --json
343
+ gscdump report opportunities --site example.com --json
344
+ gscdump report movers --site example.com --period 28d --vs prev-period --json
239
345
  gscdump analyze list --json
240
- gscdump analyze striking-distance --site sc-domain:example.com --json
346
+ gscdump analyze striking-distance --site example.com --json
241
347
  ```
242
348
 
243
349
  - Report ids: `brand`, `growth`, `health`, `movers`, `opportunities`,
244
350
  `pre-publish`, `risks`, `triage`.
245
- - `--period` takes `7d`, `28d`, `90d`, `mtd`, `ytd`, or `custom` with
246
- `--start` and `--end`. `--vs` takes `none`, `prev-period`, or `yoy`.
351
+ - `--period` takes `7d`, `28d`, `30d`, `90d`, `180d`, `365d`, `mtd`, `qtd`,
352
+ `ytd`, `last-quarter`, or `custom`. `--start`/`--end` without `--period`
353
+ select a custom window. `--vs` takes `none`, `prev-period`, or `yoy`
354
+ (same weekdays 52 weeks earlier).
355
+ - Windows end on the newest synced day, or three days ago (Pacific time)
356
+ with `--live`. They never end on today.
247
357
  - `report <id> --explain` prints the plan without credentials or data.
248
358
  - `triage` needs `--target <page-or-query> --target-kind page|query`.
249
359
  `pre-publish` needs `--topic`. `brand` needs `--brand-terms 'a,b'`.
250
- - Analyzers take `--start` and `--end`. `movers` and `decay` also take
251
- `--prev-start` and `--prev-end`. `--period` and `--vs` belong to `report`.
360
+ - Analyzers take `--period`, `--start` and `--end`. `movers` and `decay`
361
+ compare with the previous period by default. Pass `--prev-start` and
362
+ `--prev-end` together to override it. `--vs` belongs to `report`.
363
+ - `--limit` caps the rows returned. It never caps the rows read.
364
+ `--fetch-budget` caps each live fetch (default 25000, max 100000).
365
+ - A `! Partial data` warning, or `meta.coverage.kind: "truncated"` in JSON,
366
+ means a live fetch hit its budget. Say the result is partial, or rerun
367
+ with a larger `--fetch-budget`.
368
+ - Local `analyze` and `report` runs need every day of the current and
369
+ comparison windows synced. If a day is missing, failed or pending, the
370
+ run stops and prints the `gscdump sync --site ... --start ... --end ...
371
+ --tables ...` command that fills it. Run it, or pass `--live`.
252
372
  - SQL-only Analyzers need Store rows. `--live` runs row-based Analyzers
253
373
  against Google.
254
374
  - Results name candidates for review. They do not prove why traffic changed.
@@ -256,15 +376,30 @@ gscdump analyze striking-distance --site sc-domain:example.com --json
256
376
  ## Inspect and index
257
377
 
258
378
  ```sh
259
- gscdump inspect https://example.com/page --site sc-domain:example.com --json
260
- gscdump inspect batch --site sc-domain:example.com --file urls.txt --json
379
+ gscdump inspect https://example.com/page https://example.com/other --site example.com --json
380
+ gscdump inspect --site example.com --file urls.txt --json
261
381
  gscdump indexing quota --json
262
382
  ```
263
383
 
264
- Inspection spends Google's separate 2,000 requests per Site per day quota.
384
+ Inspection spends Google's separate quota: 2,000 requests per day and 600 per minute for each property.
385
+ `inspect` refuses more than 2,000 URLs in one run. It saves each result to the Store.
386
+ On a quota error it stops and reports `remaining`. It exits 1 when any URL fails or remains.
265
387
  `indexing quota` describes Indexing API limits. It does not report remaining URL Inspection requests.
266
388
  Report the Indexing Evidence fields as Google returned them.
267
389
 
390
+ ## Find URLs Google has not indexed (hosted)
391
+
392
+ ```sh
393
+ gscdump indexing urls --site example.com --status not_indexed --json
394
+ gscdump indexing urls --site example.com --status not_indexed --all --format csv
395
+ ```
396
+
397
+ - The command reads URL Inspection results that gscdump.com already saved. It spends no inspection quota.
398
+ - `--status` takes `indexed`, `not_indexed`, or `pending`. `--search` keeps URLs that contain the text.
399
+ - Each row lists the sitemaps that contain the URL.
400
+ - Pages hold 100 rows by default and 500 at most. Use `--offset` for the next page, or `--all` for every page.
401
+ - With local authentication, the command fails. Pipe `gscdump sitemaps urls <sitemap-url>` into `gscdump inspect --site <site>` instead.
402
+
268
403
  ## Report a papercut
269
404
 
270
405
  If CLI behavior blocks or slows your work, report it once per distinct
@@ -1,76 +0,0 @@
1
- import { OUTPUT_ARGS, applyOutputMode, displayPath } from "../utils.mjs";
2
- import { allTables } from "../local-store.mjs";
3
- import { createCommandContext } from "../context.mjs";
4
- import { materializeParquetTables } from "../native-duckdb.mjs";
5
- import { defineCommand } from "citty";
6
- import path from "node:path";
7
- async function exportToDuckDB(opts) {
8
- const outPath = path.resolve(opts.outPath);
9
- const entries = await opts.engine.listLive({
10
- userId: opts.userId,
11
- siteId: opts.siteId
12
- });
13
- const inputs = [];
14
- for (const table of allTables()) {
15
- const tableEntries = entries.filter((entry) => entry.table === table);
16
- if (tableEntries.length > 0) inputs.push({
17
- table,
18
- filePaths: tableEntries.map((entry) => path.join(opts.dataDir, entry.objectKey))
19
- });
20
- }
21
- const tables = await materializeParquetTables(outPath, inputs, opts.force);
22
- return {
23
- outPath,
24
- tables,
25
- totalRows: tables.reduce((acc, t) => acc + t.rows, 0)
26
- };
27
- }
28
- const exportCommand = defineCommand({
29
- meta: {
30
- name: "export",
31
- description: "Pack live Parquet partitions into a single .duckdb file for portable distribution (browser attach, CDN serving, etc.)"
32
- },
33
- args: {
34
- out: {
35
- type: "string",
36
- required: true,
37
- description: "Output path for the .duckdb file"
38
- },
39
- site: {
40
- type: "string",
41
- description: "Limit export to a single site URL (omit to include all)"
42
- },
43
- force: {
44
- type: "boolean",
45
- default: false,
46
- description: "Overwrite the output file if it already exists"
47
- },
48
- ...OUTPUT_ARGS
49
- },
50
- async run({ args }) {
51
- const { json } = applyOutputMode(args);
52
- const store = (await createCommandContext({ needsStore: true })).store;
53
- const siteId = args.site ? store.siteIdFor(args.site) : void 0;
54
- const result = await exportToDuckDB({
55
- engine: store.engine,
56
- dataDir: store.dataDir,
57
- userId: store.userId,
58
- siteId,
59
- outPath: args.out,
60
- force: args.force
61
- });
62
- if (json) {
63
- console.log(JSON.stringify(result, null, 2));
64
- return;
65
- }
66
- if (result.tables.length === 0) {
67
- console.log(`\n No data to export. Run \`gscdump sync\` first.`);
68
- return;
69
- }
70
- for (const t of result.tables) console.log(` ${t.table.padEnd(15)} ${String(t.files).padStart(4)} parquet → ${t.table} (${t.rows.toLocaleString()} rows)`);
71
- console.log(`\n Exported ${result.tables.length} table(s), ${result.totalRows.toLocaleString()} rows → ${displayPath(result.outPath)}`);
72
- console.log(`\n Attach from DuckDB: \x1B[36mATTACH '${result.outPath}' AS gsc (READ_ONLY); SELECT * FROM gsc.pages LIMIT 10;\x1B[0m`);
73
- console.log(` Attach in a browser: use DuckDB-WASM registerFileBuffer + \x1B[36mATTACH 'gsc.duckdb' AS gsc (READ_ONLY)\x1B[0m`);
74
- }
75
- });
76
- export { exportCommand, exportToDuckDB };
@@ -1,46 +0,0 @@
1
- import { rm } from "node:fs/promises";
2
- import { dateColumnsFor } from "@gscdump/engine/schema";
3
- import { sqlEscape } from "@gscdump/engine/sql";
4
- import { dateReplaceClause } from "@gscdump/engine/sql-fragments";
5
- function parquetFileListSql(filePaths) {
6
- return filePaths.map((filePath) => `'${sqlEscape(filePath)}'`).join(", ");
7
- }
8
- async function loadDuckDB() {
9
- return import("@duckdb/node-api");
10
- }
11
- async function readParquetRows(filePaths, table) {
12
- const { DuckDBInstance } = await loadDuckDB();
13
- const instance = await DuckDBInstance.create(":memory:");
14
- const conn = await instance.connect();
15
- try {
16
- const replace = dateReplaceClause(dateColumnsFor(table), "string");
17
- return (await conn.runAndReadAll(`SELECT * ${replace} FROM read_parquet([${parquetFileListSql(filePaths)}], union_by_name=true)`)).getRowObjects();
18
- } finally {
19
- conn.closeSync();
20
- instance.closeSync();
21
- }
22
- }
23
- async function materializeParquetTables(outPath, tables, force = false) {
24
- if (force) await rm(outPath, { force: true });
25
- const { DuckDBInstance } = await loadDuckDB();
26
- const instance = await DuckDBInstance.create(outPath);
27
- const conn = await instance.connect();
28
- const results = [];
29
- try {
30
- for (const input of tables) {
31
- const replace = dateReplaceClause(dateColumnsFor(input.table), "date");
32
- await conn.run(`CREATE OR REPLACE TABLE ${input.table} AS SELECT * ${replace} FROM read_parquet([${parquetFileListSql(input.filePaths)}], union_by_name=true)`);
33
- const rows = (await conn.runAndReadAll(`SELECT count(*)::BIGINT AS n FROM ${input.table}`)).getRowObjects();
34
- results.push({
35
- table: input.table,
36
- files: input.filePaths.length,
37
- rows: Number(rows[0]?.n ?? 0)
38
- });
39
- }
40
- } finally {
41
- conn.closeSync();
42
- instance.closeSync();
43
- }
44
- return results;
45
- }
46
- export { materializeParquetTables, readParquetRows };