@gscdump/cli 3.8.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +113 -30
- package/bin/gscdump.mjs +21 -1
- package/dist/analysis-local.mjs +172 -53
- package/dist/auth-state.mjs +21 -6
- package/dist/auth.mjs +27 -18
- package/dist/bing-data.mjs +6 -2
- package/dist/bing-hosted.mjs +1 -1
- package/dist/cli-args.mjs +8 -0
- package/dist/cli.d.mts +5 -1
- package/dist/cli.mjs +41 -10
- package/dist/cloud-google.mjs +8 -1
- package/dist/command-meta.mjs +4 -4
- package/dist/command-registry.mjs +86 -10
- package/dist/commands/analyze.mjs +84 -43
- package/dist/commands/auth.mjs +46 -37
- package/dist/commands/bing.mjs +3 -3
- package/dist/commands/compact.mjs +12 -8
- package/dist/commands/config.mjs +14 -14
- package/dist/commands/doctor.mjs +26 -26
- package/dist/commands/dump.mjs +203 -105
- package/dist/commands/entities.mjs +10 -113
- package/dist/commands/gc.mjs +4 -3
- package/dist/commands/indexing-urls.mjs +185 -0
- package/dist/commands/indexing.mjs +12 -8
- package/dist/commands/init.mjs +77 -18
- package/dist/commands/inspect.mjs +97 -110
- package/dist/commands/mcp.mjs +37 -57
- package/dist/commands/papercut.mjs +1 -1
- package/dist/commands/profile.mjs +9 -17
- package/dist/commands/query.mjs +219 -210
- package/dist/commands/report.mjs +68 -85
- package/dist/commands/rollups.mjs +9 -6
- package/dist/commands/sitemaps.mjs +51 -104
- package/dist/commands/sites.mjs +40 -24
- package/dist/commands/stats.mjs +14 -17
- package/dist/commands/store-purge.mjs +9 -5
- package/dist/commands/store.mjs +0 -12
- package/dist/commands/sync.mjs +774 -296
- package/dist/config.mjs +7 -4
- package/dist/context.mjs +51 -21
- package/dist/coverage.mjs +251 -0
- package/dist/dump-bing.mjs +71 -0
- package/dist/dump-writers.mjs +246 -0
- package/dist/error-handler.mjs +108 -38
- package/dist/filters.mjs +125 -0
- package/dist/hosted-site.mjs +91 -0
- package/dist/inspect-urls.mjs +76 -0
- package/dist/inspection-record.mjs +93 -0
- package/dist/local-entities.mjs +658 -0
- package/dist/mcp/errors.mjs +71 -18
- package/dist/mcp/handlers/diagnostics.mjs +15 -12
- package/dist/mcp/handlers/reports.mjs +33 -42
- package/dist/mcp/server/index.mjs +80 -38
- package/dist/mcp/types.mjs +7 -7
- package/dist/package.mjs +1 -1
- package/dist/quota-ledger.mjs +370 -0
- package/dist/render/analysis.mjs +8 -1
- package/dist/render/report.mjs +8 -1
- package/dist/request-pacer.mjs +29 -0
- package/dist/route.mjs +295 -0
- package/dist/sitemap.mjs +4 -0
- package/dist/sql-views.mjs +117 -0
- package/dist/store-sites.mjs +134 -0
- package/dist/sync-plan.mjs +129 -0
- package/dist/sync-run.mjs +96 -0
- package/dist/table-sources.mjs +57 -0
- package/dist/token-info.mjs +54 -0
- package/dist/utils.mjs +28 -19
- package/dist/window.mjs +159 -0
- package/package.json +17 -14
- package/skills/gscdump/SKILL.md +167 -37
- package/dist/commands/export.mjs +0 -76
- package/dist/native-duckdb.mjs +0 -46
package/skills/gscdump/SKILL.md
CHANGED
|
@@ -10,6 +10,8 @@ It keeps a local Parquet Store for Google rows. Every command has `--help`.
|
|
|
10
10
|
For `query`, `-s` means `--site`, `-d` means `--dimensions`, and `-f` means `--format`.
|
|
11
11
|
Use `--start` and `--end` for dates. `--site=SITE` also works.
|
|
12
12
|
Use each option once, with either its short or long spelling.
|
|
13
|
+
Put options after the subcommand name: `gscdump store stats --json`, not `gscdump store --json stats`.
|
|
14
|
+
A failed command prints one `Error:` line to stderr and exits 1.
|
|
13
15
|
Example: `gscdump query --site=SITE --start=DATE --end=DATE -d page -f json`.
|
|
14
16
|
|
|
15
17
|
## Start each task
|
|
@@ -17,7 +19,7 @@ Example: `gscdump query --site=SITE --start=DATE --end=DATE -d page -f json`.
|
|
|
17
19
|
1. Before reading traffic, run `gscdump auth status --json`. Do this even when the user says authentication works.
|
|
18
20
|
2. Keep the requested Site, dates, dimensions, and task scope. A request for pages does not need query dimensions.
|
|
19
21
|
3. Before local queries, check coverage with `gscdump store stats --site SITE --json`.
|
|
20
|
-
Use `gscdump sync --site SITE --status --json` when you need sync-state details.
|
|
22
|
+
Use `gscdump sync --site SITE --status --json` when you need coverage, gaps, or sync-state details.
|
|
21
23
|
4. Read the table dimensions and watermarks. Sync only missing tables and the requested dates, once per task.
|
|
22
24
|
5. Use `sync --json`. Read its completion result before deciding what to do next. Never repeat a successful sync.
|
|
23
25
|
|
|
@@ -29,6 +31,8 @@ Call the local data directory the Store in your answer.
|
|
|
29
31
|
## Authentication mode
|
|
30
32
|
|
|
31
33
|
Check `gscdump auth status --json` before queries. Reuse the user's selected mode.
|
|
34
|
+
In local mode, `googleAuthenticated: true` means Google accepted the credentials. When it is false, `googleError` says why.
|
|
35
|
+
In cloud mode, `hostedSync` lists each gscdump.com Site with `syncStatus` and `syncProgress`. `gscdump sites` shows the same progress. The CLI does not read the hosted Store: cloud mode queries go to the live API through gscdump.com.
|
|
32
36
|
|
|
33
37
|
| Mode | Credentials | Query path |
|
|
34
38
|
| --- | --- | --- |
|
|
@@ -68,21 +72,45 @@ After cloud Bing login opens a browser, use `bing status --site s_SITE_ID` to co
|
|
|
68
72
|
Hosted Bing commands use the API's plan and preview access rules.
|
|
69
73
|
Hosted connection verification uses `bing verify --site s_SITE_ID`.
|
|
70
74
|
Google Indexing API and Site Verification commands require local mode.
|
|
71
|
-
Hosted sitemap
|
|
75
|
+
Hosted sitemap reads and `indexing urls` require hosted credentials. Their `--site` takes a Site URL, such as `example.com`.
|
|
72
76
|
|
|
73
77
|
## Data boundaries
|
|
74
78
|
|
|
75
79
|
- `sync`, `query --live`, `analyze --live`, `report --live`, `sites`,
|
|
76
80
|
`sitemaps`, and `inspect` use the selected authentication mode.
|
|
77
81
|
- Google Indexing API requests require local credentials. `indexing quota` only prints documented limits and needs no authentication.
|
|
78
|
-
- `
|
|
79
|
-
|
|
82
|
+
- `dump` and `store` read the local Store only.
|
|
83
|
+
- `query`, `analyze`, and `report` pick a source for each run. See [Routing](#routing).
|
|
80
84
|
- Google returns a 2 to 3 day data delay. Default windows end three days ago.
|
|
81
85
|
- Google omits low-volume rows. Pagination cannot recover them.
|
|
82
86
|
- URL Inspection is limited to 2,000 requests per Site per day.
|
|
83
87
|
- `indexing submit` and `indexing remove` are only for job posting and
|
|
84
88
|
livestream pages. Google rejects other content.
|
|
85
89
|
|
|
90
|
+
## Routing
|
|
91
|
+
|
|
92
|
+
Login is optional. A user logs in (Google or hosted) or syncs a local Store.
|
|
93
|
+
`query`, `analyze`, and `report` choose one source for each run. One run never
|
|
94
|
+
mixes Store rows and live rows.
|
|
95
|
+
|
|
96
|
+
| Store data for the Site | Google connected | Result |
|
|
97
|
+
|---|---|---|
|
|
98
|
+
| Covers every date the run needs | any | Answers from the Store, also while a sync runs |
|
|
99
|
+
| None for the tables the run needs | yes | Answers from the live API. stderr says so, and JSON has `meta.source: "live"` |
|
|
100
|
+
| None | no | Stops. Next command: `gscdump init` |
|
|
101
|
+
| Some dates missing | yes | Stops with the exact `gscdump sync` command, or pass `--live` |
|
|
102
|
+
| Some dates missing, a sync is running | yes | Stops with `Sync running: 41 of 90 days done.` Run again later, or pass `--live` |
|
|
103
|
+
|
|
104
|
+
- `--live` always asks Search Console. It needs Google auth.
|
|
105
|
+
- `query --sql` reads the Store only. It stops when no table it names has data.
|
|
106
|
+
- JSON output carries `meta.source`: `local` or `live`.
|
|
107
|
+
- A stop with `--format json` or `--json` prints `{ "error": { "code", "message", "nextCommand" } }` on stdout and exits 1.
|
|
108
|
+
Codes: `NOT_CONNECTED`, `STORE_RANGE_NOT_COVERED` (with `missingDates`), `SYNC_RUNNING` (with `sync.done` and `sync.total`), `NO_SYNCED_DATA`, `STORE_ONLY`, `LIVE_ONLY`.
|
|
109
|
+
Partial coverage and a running sync are normal progress. Run `nextCommand`, or tell the user to.
|
|
110
|
+
- A sync is running only while its heartbeat is recent. A killed sync does not block reads.
|
|
111
|
+
- Every Search Analytics, URL Inspection, and Indexing API call spends the shared quota ledger in the data dir.
|
|
112
|
+
When a quota is spent, the command stops at once and says when the quota resets.
|
|
113
|
+
|
|
86
114
|
## Get the binary
|
|
87
115
|
|
|
88
116
|
```sh
|
|
@@ -134,10 +162,18 @@ Cloud mode requires hosted access. Pro is free during beta, then paid after laun
|
|
|
134
162
|
|
|
135
163
|
## Site identifiers
|
|
136
164
|
|
|
137
|
-
For Google,
|
|
165
|
+
For Google, pass the Site as the user writes it: `--site example.com`.
|
|
166
|
+
`https://example.com`, `www.example.com`, and `Example.com` resolve to the same Site.
|
|
167
|
+
Do not add `sc-domain:`.
|
|
138
168
|
|
|
139
|
-
-
|
|
140
|
-
- URL
|
|
169
|
+
- The CLI checks Sites in the Store first. It asks Search Console only when the Store has no match and auth exists.
|
|
170
|
+
- A full Site URL that names a property exactly, such as `https://example.com/`, picks that property.
|
|
171
|
+
- Otherwise, if a domain property and a URL-prefix property both match, the CLI picks the one with Store data, then the domain property.
|
|
172
|
+
- If a URL-prefix property has a path, include the path: `--site example.com/blog`.
|
|
173
|
+
- If the input is a subdomain inside a domain property, the command fails. The error names the parent Site and a `--page` filter to use.
|
|
174
|
+
- If the input matches more than one Site, the command fails and lists them. Pass one of the listed Site URLs.
|
|
175
|
+
- Without `--site` and `defaultSite`, a command in a terminal shows a picker.
|
|
176
|
+
Without a terminal, the command fails with `Pass --site. Sites: ...`, unless only one Site is available.
|
|
141
177
|
|
|
142
178
|
For cloud Bing commands, use a Site ID from `gscdump bing sites`, such as `s_SITE_ID`.
|
|
143
179
|
For local Bing commands, use the full verified Site URL from `gscdump bing sites --mode local`.
|
|
@@ -146,7 +182,7 @@ Bing commands require their own explicit `--site`; the Google `defaultSite` sett
|
|
|
146
182
|
Set a default once to drop `--site` from later commands:
|
|
147
183
|
|
|
148
184
|
```sh
|
|
149
|
-
gscdump config set defaultSite
|
|
185
|
+
gscdump config set defaultSite example.com
|
|
150
186
|
```
|
|
151
187
|
|
|
152
188
|
## Output
|
|
@@ -174,17 +210,17 @@ Do not rewrite rows, estimate metrics, or add manually calculated totals.
|
|
|
174
210
|
| `gscdump query` | Rows by page, query, date, country, or device |
|
|
175
211
|
| `gscdump analyze <id>` | One Analyzer over the Store or live rows |
|
|
176
212
|
| `gscdump report <id>` | A Report that composes several Analyzers |
|
|
177
|
-
| `gscdump inspect <url
|
|
213
|
+
| `gscdump inspect <url...>` | URL Inspection with Indexing Evidence, saved to the Store |
|
|
178
214
|
| `gscdump sitemaps` | List, submit, delete, and probe sitemaps |
|
|
179
|
-
| `gscdump indexing` | Indexing API notifications and quota |
|
|
180
|
-
| `gscdump dump` | Export Store tables
|
|
215
|
+
| `gscdump indexing` | Indexing API notifications and quota; hosted URL Inspection results |
|
|
216
|
+
| `gscdump dump` | Export Store tables, inspections, sitemaps, and Bing data as Parquet, CSV, JSON, NDJSON, SQLite, or DuckDB |
|
|
181
217
|
| `gscdump store` | Store stats, compaction, garbage collection, resets |
|
|
182
|
-
| `gscdump entities` |
|
|
218
|
+
| `gscdump entities` | Read saved inspections; snapshot Indexing API metadata |
|
|
183
219
|
| `gscdump config` | Defaults such as `defaultSite`, `dataDir`, `defaultLimit` |
|
|
184
220
|
| `gscdump profile` | Separate credential and config directories |
|
|
185
221
|
| `gscdump auth` | `status`, `login`, `logout`, `refresh` |
|
|
186
222
|
| `gscdump doctor` | Health checks for auth, scopes, Store, and reachability |
|
|
187
|
-
| `gscdump init` |
|
|
223
|
+
| `gscdump init` | First-time setup. Without a terminal it never prompts: it uses BYOK env credentials or fails with the auth command |
|
|
188
224
|
| `gscdump mcp` | Start Google MCP tools with the selected authentication |
|
|
189
225
|
| `gscdump skill install` | Copy this skill into an agent skill directory |
|
|
190
226
|
| `gscdump papercut` | Report a CLI problem to gscdump.com |
|
|
@@ -192,29 +228,52 @@ Do not rewrite rows, estimate metrics, or add manually calculated totals.
|
|
|
192
228
|
`gscdump login`, `gscdump logout`, and `gscdump status` are top-level aliases of
|
|
193
229
|
the matching `auth` subcommands.
|
|
194
230
|
The MCP server does not expose Bing tools. Use `gscdump bing` commands through this skill.
|
|
231
|
+
`gscdump mcp` starts without authentication. Its Google tools then return an error with the command to run.
|
|
232
|
+
A failed Google request returns its status, Google's explanation, and the next step in the tool result.
|
|
195
233
|
|
|
196
234
|
## Sync before local analysis
|
|
197
235
|
|
|
198
236
|
```sh
|
|
199
|
-
gscdump store stats --site
|
|
200
|
-
gscdump sync --site
|
|
201
|
-
|
|
202
|
-
gscdump sync --site sc-domain:example.com --status --json
|
|
237
|
+
gscdump store stats --site example.com --json
|
|
238
|
+
gscdump sync --site example.com --status --json
|
|
239
|
+
gscdump sync --site example.com --json
|
|
203
240
|
```
|
|
204
241
|
|
|
205
|
-
-
|
|
206
|
-
|
|
207
|
-
|
|
242
|
+
- A plain sync catches up. Each table runs from its oldest synced date to the
|
|
243
|
+
latest date Google has finalized (Pacific time, about 3 days late). A table
|
|
244
|
+
with no history starts 28 days back. Newest dates come first.
|
|
245
|
+
- `--days N`, `--start`, and `--end` pick a range instead. `--full` fetches the
|
|
246
|
+
16 months Google keeps, plus 14 days Google often still serves.
|
|
247
|
+
- Sync covers every table and search type by default. Pass `--tables` and
|
|
248
|
+
`--types` to sync less. Sync skips table and type pairs Google cannot answer.
|
|
249
|
+
- Sync paces Google calls: 8 in flight and 600 per minute across all tables.
|
|
250
|
+
`--requests-per-minute N` changes the rate.
|
|
251
|
+
Sync does not retry a quota 403. The quota ledger stops the run instead.
|
|
252
|
+
- Every call goes through a quota ledger in the Store directory. If Google
|
|
253
|
+
refuses a call for quota, or the run reaches `--max-calls N`, sync stops,
|
|
254
|
+
keeps the rest `pending`, and exits 0. `status` in `sync --json` is then
|
|
255
|
+
`partial` and `stopped` says why. Run the same command later to continue.
|
|
256
|
+
Exit 1 means real failures: read `failed` dates in `sync --status --json`.
|
|
257
|
+
- Sync also saves the sitemap list, sitemap URLs, and URL Inspection results.
|
|
258
|
+
It inspects up to 50 due URLs per run: never-inspected sitemap URLs first,
|
|
259
|
+
then pages with impressions, then the oldest results. `--inspect-limit N`
|
|
260
|
+
changes that; Google allows 2,000 per Site per day. `--no-sitemaps` and
|
|
261
|
+
`--no-inspections` skip those steps.
|
|
262
|
+
- `coverage` in `sync --json` and `sync --status --json` says how much the
|
|
263
|
+
Store holds. Partial coverage is normal progress. Never report data as
|
|
264
|
+
complete unless its `kind` is `complete`.
|
|
265
|
+
- A day Google still updates stays `pending`; the next sync fetches it again.
|
|
208
266
|
- Sync skips completed dates. `--force` refreshes them. `--retry-failed`
|
|
209
|
-
reruns only failed dates.
|
|
210
|
-
- `--dry-run` prints the planned
|
|
211
|
-
-
|
|
267
|
+
reruns only failed dates. A plain sync also retries failed dates.
|
|
268
|
+
- `--dry-run` prints the planned dates and the fewest calls without calling Google.
|
|
269
|
+
- `--all-sites` syncs every verified Site, one after another.
|
|
270
|
+
- Use the user's date range. If the user names a range, pass `--start` and `--end`, not `--full`.
|
|
212
271
|
- Empty Store metadata is expected before the first sync. It does not prove zero traffic.
|
|
213
272
|
|
|
214
273
|
## Query rows
|
|
215
274
|
|
|
216
275
|
```sh
|
|
217
|
-
gscdump query --site
|
|
276
|
+
gscdump query --site example.com --dimensions page,query \
|
|
218
277
|
--start 2026-08-01 --end 2026-08-28 --limit 1000 --format json
|
|
219
278
|
```
|
|
220
279
|
|
|
@@ -222,33 +281,89 @@ gscdump query --site sc-domain:example.com --dimensions page,query \
|
|
|
222
281
|
- Filters: `--query`, `--page`, `--country`, `--device`,
|
|
223
282
|
`--search-appearance`. Prefixes: bare equals, `~` contains, `!~` not
|
|
224
283
|
contains, `re:` regex, `!re:` not regex, `!` not equals.
|
|
225
|
-
- `--
|
|
284
|
+
- `--page` takes a path or a full URL. The Store compares paths.
|
|
285
|
+
- Without dates, `query` reads the 28 days ending on the newest synced day.
|
|
286
|
+
- `--live` bypasses the Store. `--type` selects a search type. The default is `web`.
|
|
226
287
|
`--data-state` and `--aggregation-type` apply to live mode only.
|
|
227
288
|
- Metrics already include clicks, impressions, CTR, and position. There is no `--metrics` option.
|
|
228
|
-
- If Store coverage is missing, read the JSON error and its
|
|
289
|
+
- If Store coverage is missing, read the JSON error and run its `nextCommand`. It syncs only the missing dates and tables.
|
|
229
290
|
Do not switch dimensions to make a failed query succeed.
|
|
230
291
|
- `--explain` prints the request body or planned SQL without executing.
|
|
231
|
-
|
|
292
|
+
|
|
293
|
+
## SQL over the Store
|
|
294
|
+
|
|
295
|
+
```sh
|
|
296
|
+
gscdump query --schema --format json
|
|
297
|
+
gscdump query --format json --sql "SELECT search_type, SUM(clicks) AS clicks,
|
|
298
|
+
gsc_position(sum_position, impressions) AS position
|
|
299
|
+
FROM pages WHERE date >= DATE '2026-08-01' GROUP BY search_type"
|
|
300
|
+
```
|
|
301
|
+
|
|
302
|
+
- `--sql` runs DuckDB SQL over one view per Store table: `pages`, `queries`,
|
|
303
|
+
`page_queries`, `countries`, `dates`, `hourly_pages`, and the
|
|
304
|
+
`search_appearance*` tables. Join views on `site`, `search_type`, `url`, and `date`.
|
|
305
|
+
- `--schema` lists each view, its columns, its Sites, and its date range.
|
|
306
|
+
- Every view has `site` (the Site URL) and `search_type`. The Store keeps
|
|
307
|
+
every search type, so filter or group by `search_type`. A plain `SUM` adds
|
|
308
|
+
web, image, and Discover rows together.
|
|
309
|
+
- `url` holds the page path. `page` is the same value.
|
|
310
|
+
- `sum_position` is the zero-based position times impressions. Use
|
|
311
|
+
`gsc_position(sum_position, impressions)` for the average position. It adds 1
|
|
312
|
+
and weights by impressions. Never average a per-row position.
|
|
313
|
+
- The views cover every Site in the Store. `--site` and `--type` narrow them.
|
|
314
|
+
- Dates return as `YYYY-MM-DD`. Integers return as numbers.
|
|
315
|
+
- If a query names a table with no synced data, the JSON has a `warnings` list.
|
|
316
|
+
|
|
317
|
+
## Export the Store
|
|
318
|
+
|
|
319
|
+
```sh
|
|
320
|
+
gscdump dump --site example.com --format parquet --out ./export
|
|
321
|
+
gscdump dump --all-sites --format sqlite --out ./export
|
|
322
|
+
```
|
|
323
|
+
|
|
324
|
+
- `dump` reads only the Store. It never calls Google to fill a gap.
|
|
325
|
+
- Every exported row has `site` and `search_type`.
|
|
326
|
+
- File formats write `<site>/<search_type>/<table>.<ext>` and
|
|
327
|
+
`<site>/<dataset>.<ext>` for inspections, sitemaps, and Indexing API metadata.
|
|
328
|
+
- `csv`, `json`, and `ndjson` rows also have `position`: `sum_position / impressions + 1`.
|
|
329
|
+
- `sqlite` and `duckdb` write one file, `gscdump.sqlite` or `gscdump.duckdb`,
|
|
330
|
+
with one table per dataset for every Site and search type.
|
|
331
|
+
- `manifest.json` lists every dataset with its row count, plus the coverage that `sync --status --json` reports. Partial coverage is progress: daily sync fills the rest.
|
|
332
|
+
- `sites.json` lists each exported Site URL with its Store ID.
|
|
232
333
|
|
|
233
334
|
## Analyze and report
|
|
234
335
|
|
|
235
336
|
```sh
|
|
236
337
|
gscdump report list --json
|
|
237
|
-
gscdump report opportunities --site
|
|
238
|
-
gscdump report movers --site
|
|
338
|
+
gscdump report opportunities --site example.com --json
|
|
339
|
+
gscdump report movers --site example.com --period 28d --vs prev-period --json
|
|
239
340
|
gscdump analyze list --json
|
|
240
|
-
gscdump analyze striking-distance --site
|
|
341
|
+
gscdump analyze striking-distance --site example.com --json
|
|
241
342
|
```
|
|
242
343
|
|
|
243
344
|
- Report ids: `brand`, `growth`, `health`, `movers`, `opportunities`,
|
|
244
345
|
`pre-publish`, `risks`, `triage`.
|
|
245
|
-
- `--period` takes `7d`, `28d`, `90d`, `
|
|
246
|
-
|
|
346
|
+
- `--period` takes `7d`, `28d`, `30d`, `90d`, `180d`, `365d`, `mtd`, `qtd`,
|
|
347
|
+
`ytd`, `last-quarter`, or `custom`. `--start`/`--end` without `--period`
|
|
348
|
+
select a custom window. `--vs` takes `none`, `prev-period`, or `yoy`
|
|
349
|
+
(same weekdays 52 weeks earlier).
|
|
350
|
+
- Windows end on the newest synced day, or three days ago (Pacific time)
|
|
351
|
+
with `--live`. They never end on today.
|
|
247
352
|
- `report <id> --explain` prints the plan without credentials or data.
|
|
248
353
|
- `triage` needs `--target <page-or-query> --target-kind page|query`.
|
|
249
354
|
`pre-publish` needs `--topic`. `brand` needs `--brand-terms 'a,b'`.
|
|
250
|
-
- Analyzers take `--start` and `--end`. `movers` and `decay`
|
|
251
|
-
|
|
355
|
+
- Analyzers take `--period`, `--start` and `--end`. `movers` and `decay`
|
|
356
|
+
compare with the previous period by default. Pass `--prev-start` and
|
|
357
|
+
`--prev-end` together to override it. `--vs` belongs to `report`.
|
|
358
|
+
- `--limit` caps the rows returned. It never caps the rows read.
|
|
359
|
+
`--fetch-budget` caps each live fetch (default 25000, max 100000).
|
|
360
|
+
- A `! Partial data` warning, or `meta.coverage.kind: "truncated"` in JSON,
|
|
361
|
+
means a live fetch hit its budget. Say the result is partial, or rerun
|
|
362
|
+
with a larger `--fetch-budget`.
|
|
363
|
+
- Local `analyze` and `report` runs need every day of the current and
|
|
364
|
+
comparison windows synced. If a day is missing, failed or pending, the
|
|
365
|
+
run stops and prints the `gscdump sync --site ... --start ... --end ...
|
|
366
|
+
--tables ...` command that fills it. Run it, or pass `--live`.
|
|
252
367
|
- SQL-only Analyzers need Store rows. `--live` runs row-based Analyzers
|
|
253
368
|
against Google.
|
|
254
369
|
- Results name candidates for review. They do not prove why traffic changed.
|
|
@@ -256,15 +371,30 @@ gscdump analyze striking-distance --site sc-domain:example.com --json
|
|
|
256
371
|
## Inspect and index
|
|
257
372
|
|
|
258
373
|
```sh
|
|
259
|
-
gscdump inspect https://example.com/page --site
|
|
260
|
-
gscdump inspect
|
|
374
|
+
gscdump inspect https://example.com/page https://example.com/other --site example.com --json
|
|
375
|
+
gscdump inspect --site example.com --file urls.txt --json
|
|
261
376
|
gscdump indexing quota --json
|
|
262
377
|
```
|
|
263
378
|
|
|
264
|
-
Inspection spends Google's separate 2,000 requests per
|
|
379
|
+
Inspection spends Google's separate quota: 2,000 requests per day and 600 per minute for each property.
|
|
380
|
+
`inspect` refuses more than 2,000 URLs in one run. It saves each result to the Store.
|
|
381
|
+
On a quota error it stops and reports `remaining`. It exits 1 when any URL fails or remains.
|
|
265
382
|
`indexing quota` describes Indexing API limits. It does not report remaining URL Inspection requests.
|
|
266
383
|
Report the Indexing Evidence fields as Google returned them.
|
|
267
384
|
|
|
385
|
+
## Find URLs Google has not indexed (hosted)
|
|
386
|
+
|
|
387
|
+
```sh
|
|
388
|
+
gscdump indexing urls --site example.com --status not_indexed --json
|
|
389
|
+
gscdump indexing urls --site example.com --status not_indexed --all --format csv
|
|
390
|
+
```
|
|
391
|
+
|
|
392
|
+
- The command reads URL Inspection results that gscdump.com already saved. It spends no inspection quota.
|
|
393
|
+
- `--status` takes `indexed`, `not_indexed`, or `pending`. `--search` keeps URLs that contain the text.
|
|
394
|
+
- Each row lists the sitemaps that contain the URL.
|
|
395
|
+
- Pages hold 100 rows by default and 500 at most. Use `--offset` for the next page, or `--all` for every page.
|
|
396
|
+
- With local authentication, the command fails. Pipe `gscdump sitemaps urls <sitemap-url>` into `gscdump inspect --site <site>` instead.
|
|
397
|
+
|
|
268
398
|
## Report a papercut
|
|
269
399
|
|
|
270
400
|
If CLI behavior blocks or slows your work, report it once per distinct
|
package/dist/commands/export.mjs
DELETED
|
@@ -1,76 +0,0 @@
|
|
|
1
|
-
import { OUTPUT_ARGS, applyOutputMode, displayPath } from "../utils.mjs";
|
|
2
|
-
import { allTables } from "../local-store.mjs";
|
|
3
|
-
import { createCommandContext } from "../context.mjs";
|
|
4
|
-
import { materializeParquetTables } from "../native-duckdb.mjs";
|
|
5
|
-
import { defineCommand } from "citty";
|
|
6
|
-
import path from "node:path";
|
|
7
|
-
async function exportToDuckDB(opts) {
|
|
8
|
-
const outPath = path.resolve(opts.outPath);
|
|
9
|
-
const entries = await opts.engine.listLive({
|
|
10
|
-
userId: opts.userId,
|
|
11
|
-
siteId: opts.siteId
|
|
12
|
-
});
|
|
13
|
-
const inputs = [];
|
|
14
|
-
for (const table of allTables()) {
|
|
15
|
-
const tableEntries = entries.filter((entry) => entry.table === table);
|
|
16
|
-
if (tableEntries.length > 0) inputs.push({
|
|
17
|
-
table,
|
|
18
|
-
filePaths: tableEntries.map((entry) => path.join(opts.dataDir, entry.objectKey))
|
|
19
|
-
});
|
|
20
|
-
}
|
|
21
|
-
const tables = await materializeParquetTables(outPath, inputs, opts.force);
|
|
22
|
-
return {
|
|
23
|
-
outPath,
|
|
24
|
-
tables,
|
|
25
|
-
totalRows: tables.reduce((acc, t) => acc + t.rows, 0)
|
|
26
|
-
};
|
|
27
|
-
}
|
|
28
|
-
const exportCommand = defineCommand({
|
|
29
|
-
meta: {
|
|
30
|
-
name: "export",
|
|
31
|
-
description: "Pack live Parquet partitions into a single .duckdb file for portable distribution (browser attach, CDN serving, etc.)"
|
|
32
|
-
},
|
|
33
|
-
args: {
|
|
34
|
-
out: {
|
|
35
|
-
type: "string",
|
|
36
|
-
required: true,
|
|
37
|
-
description: "Output path for the .duckdb file"
|
|
38
|
-
},
|
|
39
|
-
site: {
|
|
40
|
-
type: "string",
|
|
41
|
-
description: "Limit export to a single site URL (omit to include all)"
|
|
42
|
-
},
|
|
43
|
-
force: {
|
|
44
|
-
type: "boolean",
|
|
45
|
-
default: false,
|
|
46
|
-
description: "Overwrite the output file if it already exists"
|
|
47
|
-
},
|
|
48
|
-
...OUTPUT_ARGS
|
|
49
|
-
},
|
|
50
|
-
async run({ args }) {
|
|
51
|
-
const { json } = applyOutputMode(args);
|
|
52
|
-
const store = (await createCommandContext({ needsStore: true })).store;
|
|
53
|
-
const siteId = args.site ? store.siteIdFor(args.site) : void 0;
|
|
54
|
-
const result = await exportToDuckDB({
|
|
55
|
-
engine: store.engine,
|
|
56
|
-
dataDir: store.dataDir,
|
|
57
|
-
userId: store.userId,
|
|
58
|
-
siteId,
|
|
59
|
-
outPath: args.out,
|
|
60
|
-
force: args.force
|
|
61
|
-
});
|
|
62
|
-
if (json) {
|
|
63
|
-
console.log(JSON.stringify(result, null, 2));
|
|
64
|
-
return;
|
|
65
|
-
}
|
|
66
|
-
if (result.tables.length === 0) {
|
|
67
|
-
console.log(`\n No data to export. Run \`gscdump sync\` first.`);
|
|
68
|
-
return;
|
|
69
|
-
}
|
|
70
|
-
for (const t of result.tables) console.log(` ${t.table.padEnd(15)} ${String(t.files).padStart(4)} parquet → ${t.table} (${t.rows.toLocaleString()} rows)`);
|
|
71
|
-
console.log(`\n Exported ${result.tables.length} table(s), ${result.totalRows.toLocaleString()} rows → ${displayPath(result.outPath)}`);
|
|
72
|
-
console.log(`\n Attach from DuckDB: \x1B[36mATTACH '${result.outPath}' AS gsc (READ_ONLY); SELECT * FROM gsc.pages LIMIT 10;\x1B[0m`);
|
|
73
|
-
console.log(` Attach in a browser: use DuckDB-WASM registerFileBuffer + \x1B[36mATTACH 'gsc.duckdb' AS gsc (READ_ONLY)\x1B[0m`);
|
|
74
|
-
}
|
|
75
|
-
});
|
|
76
|
-
export { exportCommand, exportToDuckDB };
|
package/dist/native-duckdb.mjs
DELETED
|
@@ -1,46 +0,0 @@
|
|
|
1
|
-
import { rm } from "node:fs/promises";
|
|
2
|
-
import { dateColumnsFor } from "@gscdump/engine/schema";
|
|
3
|
-
import { sqlEscape } from "@gscdump/engine/sql";
|
|
4
|
-
import { dateReplaceClause } from "@gscdump/engine/sql-fragments";
|
|
5
|
-
function parquetFileListSql(filePaths) {
|
|
6
|
-
return filePaths.map((filePath) => `'${sqlEscape(filePath)}'`).join(", ");
|
|
7
|
-
}
|
|
8
|
-
async function loadDuckDB() {
|
|
9
|
-
return import("@duckdb/node-api");
|
|
10
|
-
}
|
|
11
|
-
async function readParquetRows(filePaths, table) {
|
|
12
|
-
const { DuckDBInstance } = await loadDuckDB();
|
|
13
|
-
const instance = await DuckDBInstance.create(":memory:");
|
|
14
|
-
const conn = await instance.connect();
|
|
15
|
-
try {
|
|
16
|
-
const replace = dateReplaceClause(dateColumnsFor(table), "string");
|
|
17
|
-
return (await conn.runAndReadAll(`SELECT * ${replace} FROM read_parquet([${parquetFileListSql(filePaths)}], union_by_name=true)`)).getRowObjects();
|
|
18
|
-
} finally {
|
|
19
|
-
conn.closeSync();
|
|
20
|
-
instance.closeSync();
|
|
21
|
-
}
|
|
22
|
-
}
|
|
23
|
-
async function materializeParquetTables(outPath, tables, force = false) {
|
|
24
|
-
if (force) await rm(outPath, { force: true });
|
|
25
|
-
const { DuckDBInstance } = await loadDuckDB();
|
|
26
|
-
const instance = await DuckDBInstance.create(outPath);
|
|
27
|
-
const conn = await instance.connect();
|
|
28
|
-
const results = [];
|
|
29
|
-
try {
|
|
30
|
-
for (const input of tables) {
|
|
31
|
-
const replace = dateReplaceClause(dateColumnsFor(input.table), "date");
|
|
32
|
-
await conn.run(`CREATE OR REPLACE TABLE ${input.table} AS SELECT * ${replace} FROM read_parquet([${parquetFileListSql(input.filePaths)}], union_by_name=true)`);
|
|
33
|
-
const rows = (await conn.runAndReadAll(`SELECT count(*)::BIGINT AS n FROM ${input.table}`)).getRowObjects();
|
|
34
|
-
results.push({
|
|
35
|
-
table: input.table,
|
|
36
|
-
files: input.filePaths.length,
|
|
37
|
-
rows: Number(rows[0]?.n ?? 0)
|
|
38
|
-
});
|
|
39
|
-
}
|
|
40
|
-
} finally {
|
|
41
|
-
conn.closeSync();
|
|
42
|
-
instance.closeSync();
|
|
43
|
-
}
|
|
44
|
-
return results;
|
|
45
|
-
}
|
|
46
|
-
export { materializeParquetTables, readParquetRows };
|