@gscdump/cli 3.8.0 → 4.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +115 -30
- package/bin/gscdump.mjs +21 -1
- package/dist/analysis-local.mjs +172 -53
- package/dist/auth-state.mjs +21 -6
- package/dist/auth.mjs +27 -18
- package/dist/bing-data.mjs +6 -2
- package/dist/bing-hosted.mjs +1 -1
- package/dist/cli-args.mjs +8 -0
- package/dist/cli.d.mts +5 -1
- package/dist/cli.mjs +41 -10
- package/dist/cloud-google.mjs +8 -1
- package/dist/command-meta.mjs +4 -4
- package/dist/command-registry.mjs +86 -10
- package/dist/commands/analyze.mjs +84 -43
- package/dist/commands/auth.mjs +46 -37
- package/dist/commands/bing.mjs +3 -3
- package/dist/commands/compact.mjs +12 -8
- package/dist/commands/config.mjs +14 -14
- package/dist/commands/doctor.mjs +26 -26
- package/dist/commands/dump.mjs +203 -105
- package/dist/commands/entities.mjs +10 -113
- package/dist/commands/gc.mjs +4 -3
- package/dist/commands/indexing-urls.mjs +185 -0
- package/dist/commands/indexing.mjs +12 -8
- package/dist/commands/init.mjs +77 -18
- package/dist/commands/inspect.mjs +97 -110
- package/dist/commands/mcp.mjs +37 -57
- package/dist/commands/papercut.mjs +1 -1
- package/dist/commands/profile.mjs +9 -17
- package/dist/commands/query.mjs +219 -210
- package/dist/commands/report.mjs +68 -85
- package/dist/commands/rollups.mjs +9 -6
- package/dist/commands/sitemaps.mjs +51 -104
- package/dist/commands/sites.mjs +40 -24
- package/dist/commands/stats.mjs +14 -17
- package/dist/commands/store-purge.mjs +9 -5
- package/dist/commands/store.mjs +0 -12
- package/dist/commands/sync.mjs +774 -296
- package/dist/config.mjs +7 -4
- package/dist/context.mjs +51 -21
- package/dist/coverage.mjs +251 -0
- package/dist/dump-bing.mjs +71 -0
- package/dist/dump-writers.mjs +246 -0
- package/dist/error-handler.mjs +108 -38
- package/dist/filters.mjs +125 -0
- package/dist/hosted-site.mjs +91 -0
- package/dist/inspect-urls.mjs +76 -0
- package/dist/inspection-record.mjs +93 -0
- package/dist/local-entities.mjs +658 -0
- package/dist/mcp/errors.mjs +71 -18
- package/dist/mcp/handlers/diagnostics.mjs +15 -12
- package/dist/mcp/handlers/reports.mjs +33 -42
- package/dist/mcp/server/index.mjs +80 -38
- package/dist/mcp/types.mjs +7 -7
- package/dist/package.mjs +1 -1
- package/dist/quota-ledger.mjs +370 -0
- package/dist/render/analysis.mjs +8 -1
- package/dist/render/report.mjs +8 -1
- package/dist/request-pacer.mjs +29 -0
- package/dist/route.mjs +301 -0
- package/dist/sitemap.mjs +4 -0
- package/dist/sql-views.mjs +117 -0
- package/dist/store-sites.mjs +134 -0
- package/dist/sync-plan.mjs +129 -0
- package/dist/sync-run.mjs +96 -0
- package/dist/table-sources.mjs +57 -0
- package/dist/token-info.mjs +54 -0
- package/dist/utils.mjs +28 -19
- package/dist/window.mjs +159 -0
- package/package.json +17 -14
- package/skills/gscdump/SKILL.md +172 -37
- package/dist/commands/export.mjs +0 -76
- package/dist/native-duckdb.mjs +0 -46
package/skills/gscdump/SKILL.md
CHANGED
|
@@ -10,6 +10,8 @@ It keeps a local Parquet Store for Google rows. Every command has `--help`.
|
|
|
10
10
|
For `query`, `-s` means `--site`, `-d` means `--dimensions`, and `-f` means `--format`.
|
|
11
11
|
Use `--start` and `--end` for dates. `--site=SITE` also works.
|
|
12
12
|
Use each option once, with either its short or long spelling.
|
|
13
|
+
Put options after the subcommand name: `gscdump store stats --json`, not `gscdump store --json stats`.
|
|
14
|
+
A failed command prints one `Error:` line to stderr and exits 1.
|
|
13
15
|
Example: `gscdump query --site=SITE --start=DATE --end=DATE -d page -f json`.
|
|
14
16
|
|
|
15
17
|
## Start each task
|
|
@@ -17,9 +19,11 @@ Example: `gscdump query --site=SITE --start=DATE --end=DATE -d page -f json`.
|
|
|
17
19
|
1. Before reading traffic, run `gscdump auth status --json`. Do this even when the user says authentication works.
|
|
18
20
|
2. Keep the requested Site, dates, dimensions, and task scope. A request for pages does not need query dimensions.
|
|
19
21
|
3. Before local queries, check coverage with `gscdump store stats --site SITE --json`.
|
|
20
|
-
Use `gscdump sync --site SITE --status --json` when you need sync-state details.
|
|
22
|
+
Use `gscdump sync --site SITE --status --json` when you need coverage, gaps, or sync-state details.
|
|
23
|
+
On a fresh Store, `store stats` exits 1 and says it has no data. Continue with the bounded sync.
|
|
21
24
|
4. Read the table dimensions and watermarks. Sync only missing tables and the requested dates, once per task.
|
|
22
25
|
5. Use `sync --json`. Read its completion result before deciding what to do next. Never repeat a successful sync.
|
|
26
|
+
6. If the user asks for saved rows, check `meta.source: "local"` in the query result. A successful query can answer live when its table has no synced data.
|
|
23
27
|
|
|
24
28
|
If the task only asks about deletion, explain the scope and ask for consent.
|
|
25
29
|
You may read Store metadata with `store stats` and `sync --status`.
|
|
@@ -29,6 +33,8 @@ Call the local data directory the Store in your answer.
|
|
|
29
33
|
## Authentication mode
|
|
30
34
|
|
|
31
35
|
Check `gscdump auth status --json` before queries. Reuse the user's selected mode.
|
|
36
|
+
In local mode, `googleAuthenticated: true` means Google accepted the credentials. When it is false, `googleError` says why.
|
|
37
|
+
In cloud mode, `hostedSync` lists each gscdump.com Site with `syncStatus` and `syncProgress`. `gscdump sites` shows the same progress. The CLI does not read the hosted Store: cloud mode queries go to the live API through gscdump.com.
|
|
32
38
|
|
|
33
39
|
| Mode | Credentials | Query path |
|
|
34
40
|
| --- | --- | --- |
|
|
@@ -68,21 +74,45 @@ After cloud Bing login opens a browser, use `bing status --site s_SITE_ID` to co
|
|
|
68
74
|
Hosted Bing commands use the API's plan and preview access rules.
|
|
69
75
|
Hosted connection verification uses `bing verify --site s_SITE_ID`.
|
|
70
76
|
Google Indexing API and Site Verification commands require local mode.
|
|
71
|
-
Hosted sitemap
|
|
77
|
+
Hosted sitemap reads and `indexing urls` require hosted credentials. Their `--site` takes a Site URL, such as `example.com`.
|
|
72
78
|
|
|
73
79
|
## Data boundaries
|
|
74
80
|
|
|
75
81
|
- `sync`, `query --live`, `analyze --live`, `report --live`, `sites`,
|
|
76
82
|
`sitemaps`, and `inspect` use the selected authentication mode.
|
|
77
83
|
- Google Indexing API requests require local credentials. `indexing quota` only prints documented limits and needs no authentication.
|
|
78
|
-
- `
|
|
79
|
-
|
|
84
|
+
- `dump` and `store` read the local Store only.
|
|
85
|
+
- `query`, `analyze`, and `report` pick a source for each run. See [Routing](#routing).
|
|
80
86
|
- Google returns a 2 to 3 day data delay. Default windows end three days ago.
|
|
81
87
|
- Google omits low-volume rows. Pagination cannot recover them.
|
|
82
88
|
- URL Inspection is limited to 2,000 requests per Site per day.
|
|
83
89
|
- `indexing submit` and `indexing remove` are only for job posting and
|
|
84
90
|
livestream pages. Google rejects other content.
|
|
85
91
|
|
|
92
|
+
## Routing
|
|
93
|
+
|
|
94
|
+
Login is optional. A user logs in (Google or hosted) or syncs a local Store.
|
|
95
|
+
`query`, `analyze`, and `report` choose one source for each run. One run never
|
|
96
|
+
mixes Store rows and live rows.
|
|
97
|
+
|
|
98
|
+
| Store data for the Site | Google connected | Result |
|
|
99
|
+
|---|---|---|
|
|
100
|
+
| Covers every date the run needs | any | Answers from the Store, also while a sync runs |
|
|
101
|
+
| None for the tables the run needs | yes | Answers from the live API. stderr says so, and JSON has `meta.source: "live"` |
|
|
102
|
+
| None | no | Stops. Next command: `gscdump init` |
|
|
103
|
+
| Some dates missing | yes | Stops with the exact `gscdump sync` command, or pass `--live` |
|
|
104
|
+
| Some dates missing, a sync is running | yes | Stops with `Sync running: 41 of 90 days done.` Run again later, or pass `--live` |
|
|
105
|
+
|
|
106
|
+
- `--live` always asks Search Console. It needs Google auth.
|
|
107
|
+
- `query --sql` reads the Store only. It stops when no table it names has data.
|
|
108
|
+
- JSON output carries `meta.source`: `local` or `live`.
|
|
109
|
+
- A stop with `--format json` or `--json` prints `{ "error": { "code", "message", "nextCommand" } }` on stdout and exits 1.
|
|
110
|
+
Codes: `NOT_CONNECTED`, `STORE_RANGE_NOT_COVERED` (with `missingDates`), `SYNC_RUNNING` (with `sync.done` and `sync.total`), `NO_SYNCED_DATA`, `STORE_ONLY`, `LIVE_ONLY`.
|
|
111
|
+
Partial coverage and a running sync are normal progress. Run `nextCommand`, or tell the user to.
|
|
112
|
+
- A sync is running only while its heartbeat is recent. A killed sync does not block reads.
|
|
113
|
+
- Every Search Analytics, URL Inspection, and Indexing API call spends the shared quota ledger in the data dir.
|
|
114
|
+
When a quota is spent, the command stops at once and says when the quota resets.
|
|
115
|
+
|
|
86
116
|
## Get the binary
|
|
87
117
|
|
|
88
118
|
```sh
|
|
@@ -134,10 +164,18 @@ Cloud mode requires hosted access. Pro is free during beta, then paid after laun
|
|
|
134
164
|
|
|
135
165
|
## Site identifiers
|
|
136
166
|
|
|
137
|
-
For Google,
|
|
167
|
+
For Google, pass the Site as the user writes it: `--site example.com`.
|
|
168
|
+
`https://example.com`, `www.example.com`, and `Example.com` resolve to the same Site.
|
|
169
|
+
Do not add `sc-domain:`.
|
|
138
170
|
|
|
139
|
-
-
|
|
140
|
-
- URL
|
|
171
|
+
- The CLI checks Sites in the Store first. It asks Search Console only when the Store has no match and auth exists.
|
|
172
|
+
- A full Site URL that names a property exactly, such as `https://example.com/`, picks that property.
|
|
173
|
+
- Otherwise, if a domain property and a URL-prefix property both match, the CLI picks the one with Store data, then the domain property.
|
|
174
|
+
- If a URL-prefix property has a path, include the path: `--site example.com/blog`.
|
|
175
|
+
- If the input is a subdomain inside a domain property, the command fails. The error names the parent Site and a `--page` filter to use.
|
|
176
|
+
- If the input matches more than one Site, the command fails and lists them. Pass one of the listed Site URLs.
|
|
177
|
+
- Without `--site` and `defaultSite`, a command in a terminal shows a picker.
|
|
178
|
+
Without a terminal, the command fails with `Pass --site. Sites: ...`, unless only one Site is available.
|
|
141
179
|
|
|
142
180
|
For cloud Bing commands, use a Site ID from `gscdump bing sites`, such as `s_SITE_ID`.
|
|
143
181
|
For local Bing commands, use the full verified Site URL from `gscdump bing sites --mode local`.
|
|
@@ -146,7 +184,7 @@ Bing commands require their own explicit `--site`; the Google `defaultSite` sett
|
|
|
146
184
|
Set a default once to drop `--site` from later commands:
|
|
147
185
|
|
|
148
186
|
```sh
|
|
149
|
-
gscdump config set defaultSite
|
|
187
|
+
gscdump config set defaultSite example.com
|
|
150
188
|
```
|
|
151
189
|
|
|
152
190
|
## Output
|
|
@@ -174,17 +212,17 @@ Do not rewrite rows, estimate metrics, or add manually calculated totals.
|
|
|
174
212
|
| `gscdump query` | Rows by page, query, date, country, or device |
|
|
175
213
|
| `gscdump analyze <id>` | One Analyzer over the Store or live rows |
|
|
176
214
|
| `gscdump report <id>` | A Report that composes several Analyzers |
|
|
177
|
-
| `gscdump inspect <url
|
|
215
|
+
| `gscdump inspect <url...>` | URL Inspection with Indexing Evidence, saved to the Store |
|
|
178
216
|
| `gscdump sitemaps` | List, submit, delete, and probe sitemaps |
|
|
179
|
-
| `gscdump indexing` | Indexing API notifications and quota |
|
|
180
|
-
| `gscdump dump` | Export Store tables
|
|
217
|
+
| `gscdump indexing` | Indexing API notifications and quota; hosted URL Inspection results |
|
|
218
|
+
| `gscdump dump` | Export Store tables, inspections, sitemaps, and Bing data as Parquet, CSV, JSON, NDJSON, SQLite, or DuckDB |
|
|
181
219
|
| `gscdump store` | Store stats, compaction, garbage collection, resets |
|
|
182
|
-
| `gscdump entities` |
|
|
220
|
+
| `gscdump entities` | Read saved inspections; snapshot Indexing API metadata |
|
|
183
221
|
| `gscdump config` | Defaults such as `defaultSite`, `dataDir`, `defaultLimit` |
|
|
184
222
|
| `gscdump profile` | Separate credential and config directories |
|
|
185
223
|
| `gscdump auth` | `status`, `login`, `logout`, `refresh` |
|
|
186
224
|
| `gscdump doctor` | Health checks for auth, scopes, Store, and reachability |
|
|
187
|
-
| `gscdump init` |
|
|
225
|
+
| `gscdump init` | First-time setup. Without a terminal it never prompts: it uses BYOK env credentials or fails with the auth command |
|
|
188
226
|
| `gscdump mcp` | Start Google MCP tools with the selected authentication |
|
|
189
227
|
| `gscdump skill install` | Copy this skill into an agent skill directory |
|
|
190
228
|
| `gscdump papercut` | Report a CLI problem to gscdump.com |
|
|
@@ -192,63 +230,145 @@ Do not rewrite rows, estimate metrics, or add manually calculated totals.
|
|
|
192
230
|
`gscdump login`, `gscdump logout`, and `gscdump status` are top-level aliases of
|
|
193
231
|
the matching `auth` subcommands.
|
|
194
232
|
The MCP server does not expose Bing tools. Use `gscdump bing` commands through this skill.
|
|
233
|
+
`gscdump mcp` starts without authentication. Its Google tools then return an error with the command to run.
|
|
234
|
+
A failed Google request returns its status, Google's explanation, and the next step in the tool result.
|
|
195
235
|
|
|
196
236
|
## Sync before local analysis
|
|
197
237
|
|
|
198
238
|
```sh
|
|
199
|
-
gscdump store stats --site
|
|
200
|
-
gscdump sync --site
|
|
201
|
-
|
|
202
|
-
gscdump sync --site sc-domain:example.com --status --json
|
|
239
|
+
gscdump store stats --site example.com --json
|
|
240
|
+
gscdump sync --site example.com --status --json
|
|
241
|
+
gscdump sync --site example.com --json
|
|
203
242
|
```
|
|
204
243
|
|
|
205
|
-
-
|
|
206
|
-
|
|
207
|
-
|
|
244
|
+
- A plain sync catches up. Each table runs from its oldest synced date to the
|
|
245
|
+
latest date Google has finalized (Pacific time, about 3 days late). A table
|
|
246
|
+
with no history starts 28 days back. Newest dates come first.
|
|
247
|
+
- `--days N`, `--start`, and `--end` pick a range instead. `--full` fetches the
|
|
248
|
+
16 months Google keeps, plus 14 days Google often still serves.
|
|
249
|
+
- Sync covers every table and search type by default. Pass `--tables` and
|
|
250
|
+
`--types` to sync less. Sync skips table and type pairs Google cannot answer.
|
|
251
|
+
- Sync paces Google calls: 8 in flight and 600 per minute across all tables.
|
|
252
|
+
`--requests-per-minute N` changes the rate.
|
|
253
|
+
Sync does not retry a quota 403. The quota ledger stops the run instead.
|
|
254
|
+
- Every call goes through a quota ledger in the Store directory. If Google
|
|
255
|
+
refuses a call for quota, or the run reaches `--max-calls N`, sync stops,
|
|
256
|
+
keeps the rest `pending`, and exits 0. `status` in `sync --json` is then
|
|
257
|
+
`partial` and `stopped` says why. Run the same command later to continue.
|
|
258
|
+
Exit 1 means real failures: read `failed` dates in `sync --status --json`.
|
|
259
|
+
- Sync also saves the sitemap list, sitemap URLs, and URL Inspection results.
|
|
260
|
+
It inspects up to 50 due URLs per run: never-inspected sitemap URLs first,
|
|
261
|
+
then pages with impressions, then the oldest results. `--inspect-limit N`
|
|
262
|
+
changes that; Google allows 2,000 per Site per day. `--no-sitemaps` and
|
|
263
|
+
`--no-inspections` skip those steps.
|
|
264
|
+
- `coverage` in `sync --json` and `sync --status --json` says how much the
|
|
265
|
+
Store holds. Partial coverage is normal progress. Never report data as
|
|
266
|
+
complete unless its `kind` is `complete`.
|
|
267
|
+
- A day Google still updates stays `pending`; the next sync fetches it again.
|
|
208
268
|
- Sync skips completed dates. `--force` refreshes them. `--retry-failed`
|
|
209
|
-
reruns only failed dates.
|
|
210
|
-
- `--dry-run` prints the planned
|
|
211
|
-
-
|
|
269
|
+
reruns only failed dates. A plain sync also retries failed dates.
|
|
270
|
+
- `--dry-run` prints the planned dates and the fewest calls without calling Google.
|
|
271
|
+
- `--all-sites` syncs every verified Site, one after another.
|
|
272
|
+
- Use the user's date range. If the user names a range, pass `--start` and `--end`, not `--full`.
|
|
273
|
+
- If the user excludes rollups, pass `--no-rollups` on the sync command.
|
|
212
274
|
- Empty Store metadata is expected before the first sync. It does not prove zero traffic.
|
|
213
275
|
|
|
214
276
|
## Query rows
|
|
215
277
|
|
|
216
278
|
```sh
|
|
217
|
-
gscdump query --site
|
|
279
|
+
gscdump query --site example.com --dimensions page,query \
|
|
218
280
|
--start 2026-08-01 --end 2026-08-28 --limit 1000 --format json
|
|
219
281
|
```
|
|
220
282
|
|
|
221
283
|
- Dimension names are singular: `page`, `query`, `date`, `country`, `device`.
|
|
284
|
+
- A page breakdown uses `--tables pages` for sync and `-d page` for query.
|
|
285
|
+
`-d page,query` needs `page_queries`; syncing only `pages` does not fill that table.
|
|
222
286
|
- Filters: `--query`, `--page`, `--country`, `--device`,
|
|
223
287
|
`--search-appearance`. Prefixes: bare equals, `~` contains, `!~` not
|
|
224
288
|
contains, `re:` regex, `!re:` not regex, `!` not equals.
|
|
225
|
-
- `--
|
|
289
|
+
- `--page` takes a path or a full URL. The Store compares paths.
|
|
290
|
+
- Without dates, `query` reads the 28 days ending on the newest synced day.
|
|
291
|
+
- `--live` bypasses the Store. `--type` selects a search type. The default is `web`.
|
|
226
292
|
`--data-state` and `--aggregation-type` apply to live mode only.
|
|
227
293
|
- Metrics already include clicks, impressions, CTR, and position. There is no `--metrics` option.
|
|
228
|
-
- If Store coverage is missing, read the JSON error and its
|
|
294
|
+
- If Store coverage is missing, read the JSON error and run its `nextCommand`. It syncs only the missing dates and tables.
|
|
229
295
|
Do not switch dimensions to make a failed query succeed.
|
|
230
296
|
- `--explain` prints the request body or planned SQL without executing.
|
|
231
|
-
|
|
297
|
+
|
|
298
|
+
## SQL over the Store
|
|
299
|
+
|
|
300
|
+
```sh
|
|
301
|
+
gscdump query --schema --format json
|
|
302
|
+
gscdump query --format json --sql "SELECT search_type, SUM(clicks) AS clicks,
|
|
303
|
+
gsc_position(sum_position, impressions) AS position
|
|
304
|
+
FROM pages WHERE date >= DATE '2026-08-01' GROUP BY search_type"
|
|
305
|
+
```
|
|
306
|
+
|
|
307
|
+
- `--sql` runs DuckDB SQL over one view per Store table: `pages`, `queries`,
|
|
308
|
+
`page_queries`, `countries`, `dates`, `hourly_pages`, and the
|
|
309
|
+
`search_appearance*` tables. Join views on `site`, `search_type`, `url`, and `date`.
|
|
310
|
+
- `--schema` lists each view, its columns, its Sites, and its date range.
|
|
311
|
+
- Every view has `site` (the Site URL) and `search_type`. The Store keeps
|
|
312
|
+
every search type, so filter or group by `search_type`. A plain `SUM` adds
|
|
313
|
+
web, image, and Discover rows together.
|
|
314
|
+
- `url` holds the page path. `page` is the same value.
|
|
315
|
+
- `sum_position` is the zero-based position times impressions. Use
|
|
316
|
+
`gsc_position(sum_position, impressions)` for the average position. It adds 1
|
|
317
|
+
and weights by impressions. Never average a per-row position.
|
|
318
|
+
- The views cover every Site in the Store. `--site` and `--type` narrow them.
|
|
319
|
+
- Dates return as `YYYY-MM-DD`. Integers return as numbers.
|
|
320
|
+
- If a query names a table with no synced data, the JSON has a `warnings` list.
|
|
321
|
+
|
|
322
|
+
## Export the Store
|
|
323
|
+
|
|
324
|
+
```sh
|
|
325
|
+
gscdump dump --site example.com --format parquet --out ./export
|
|
326
|
+
gscdump dump --all-sites --format sqlite --out ./export
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
- `dump` reads only the Store. It never calls Google to fill a gap.
|
|
330
|
+
- Every exported row has `site` and `search_type`.
|
|
331
|
+
- File formats write `<site>/<search_type>/<table>.<ext>` and
|
|
332
|
+
`<site>/<dataset>.<ext>` for inspections, sitemaps, and Indexing API metadata.
|
|
333
|
+
- `csv`, `json`, and `ndjson` rows also have `position`: `sum_position / impressions + 1`.
|
|
334
|
+
- `sqlite` and `duckdb` write one file, `gscdump.sqlite` or `gscdump.duckdb`,
|
|
335
|
+
with one table per dataset for every Site and search type.
|
|
336
|
+
- `manifest.json` lists every dataset with its row count, plus the coverage that `sync --status --json` reports. Partial coverage is progress: daily sync fills the rest.
|
|
337
|
+
- `sites.json` lists each exported Site URL with its Store ID.
|
|
232
338
|
|
|
233
339
|
## Analyze and report
|
|
234
340
|
|
|
235
341
|
```sh
|
|
236
342
|
gscdump report list --json
|
|
237
|
-
gscdump report opportunities --site
|
|
238
|
-
gscdump report movers --site
|
|
343
|
+
gscdump report opportunities --site example.com --json
|
|
344
|
+
gscdump report movers --site example.com --period 28d --vs prev-period --json
|
|
239
345
|
gscdump analyze list --json
|
|
240
|
-
gscdump analyze striking-distance --site
|
|
346
|
+
gscdump analyze striking-distance --site example.com --json
|
|
241
347
|
```
|
|
242
348
|
|
|
243
349
|
- Report ids: `brand`, `growth`, `health`, `movers`, `opportunities`,
|
|
244
350
|
`pre-publish`, `risks`, `triage`.
|
|
245
|
-
- `--period` takes `7d`, `28d`, `90d`, `
|
|
246
|
-
|
|
351
|
+
- `--period` takes `7d`, `28d`, `30d`, `90d`, `180d`, `365d`, `mtd`, `qtd`,
|
|
352
|
+
`ytd`, `last-quarter`, or `custom`. `--start`/`--end` without `--period`
|
|
353
|
+
select a custom window. `--vs` takes `none`, `prev-period`, or `yoy`
|
|
354
|
+
(same weekdays 52 weeks earlier).
|
|
355
|
+
- Windows end on the newest synced day, or three days ago (Pacific time)
|
|
356
|
+
with `--live`. They never end on today.
|
|
247
357
|
- `report <id> --explain` prints the plan without credentials or data.
|
|
248
358
|
- `triage` needs `--target <page-or-query> --target-kind page|query`.
|
|
249
359
|
`pre-publish` needs `--topic`. `brand` needs `--brand-terms 'a,b'`.
|
|
250
|
-
- Analyzers take `--start` and `--end`. `movers` and `decay`
|
|
251
|
-
|
|
360
|
+
- Analyzers take `--period`, `--start` and `--end`. `movers` and `decay`
|
|
361
|
+
compare with the previous period by default. Pass `--prev-start` and
|
|
362
|
+
`--prev-end` together to override it. `--vs` belongs to `report`.
|
|
363
|
+
- `--limit` caps the rows returned. It never caps the rows read.
|
|
364
|
+
`--fetch-budget` caps each live fetch (default 25000, max 100000).
|
|
365
|
+
- A `! Partial data` warning, or `meta.coverage.kind: "truncated"` in JSON,
|
|
366
|
+
means a live fetch hit its budget. Say the result is partial, or rerun
|
|
367
|
+
with a larger `--fetch-budget`.
|
|
368
|
+
- Local `analyze` and `report` runs need every day of the current and
|
|
369
|
+
comparison windows synced. If a day is missing, failed or pending, the
|
|
370
|
+
run stops and prints the `gscdump sync --site ... --start ... --end ...
|
|
371
|
+
--tables ...` command that fills it. Run it, or pass `--live`.
|
|
252
372
|
- SQL-only Analyzers need Store rows. `--live` runs row-based Analyzers
|
|
253
373
|
against Google.
|
|
254
374
|
- Results name candidates for review. They do not prove why traffic changed.
|
|
@@ -256,15 +376,30 @@ gscdump analyze striking-distance --site sc-domain:example.com --json
|
|
|
256
376
|
## Inspect and index
|
|
257
377
|
|
|
258
378
|
```sh
|
|
259
|
-
gscdump inspect https://example.com/page --site
|
|
260
|
-
gscdump inspect
|
|
379
|
+
gscdump inspect https://example.com/page https://example.com/other --site example.com --json
|
|
380
|
+
gscdump inspect --site example.com --file urls.txt --json
|
|
261
381
|
gscdump indexing quota --json
|
|
262
382
|
```
|
|
263
383
|
|
|
264
|
-
Inspection spends Google's separate 2,000 requests per
|
|
384
|
+
Inspection spends Google's separate quota: 2,000 requests per day and 600 per minute for each property.
|
|
385
|
+
`inspect` refuses more than 2,000 URLs in one run. It saves each result to the Store.
|
|
386
|
+
On a quota error it stops and reports `remaining`. It exits 1 when any URL fails or remains.
|
|
265
387
|
`indexing quota` describes Indexing API limits. It does not report remaining URL Inspection requests.
|
|
266
388
|
Report the Indexing Evidence fields as Google returned them.
|
|
267
389
|
|
|
390
|
+
## Find URLs Google has not indexed (hosted)
|
|
391
|
+
|
|
392
|
+
```sh
|
|
393
|
+
gscdump indexing urls --site example.com --status not_indexed --json
|
|
394
|
+
gscdump indexing urls --site example.com --status not_indexed --all --format csv
|
|
395
|
+
```
|
|
396
|
+
|
|
397
|
+
- The command reads URL Inspection results that gscdump.com already saved. It spends no inspection quota.
|
|
398
|
+
- `--status` takes `indexed`, `not_indexed`, or `pending`. `--search` keeps URLs that contain the text.
|
|
399
|
+
- Each row lists the sitemaps that contain the URL.
|
|
400
|
+
- Pages hold 100 rows by default and 500 at most. Use `--offset` for the next page, or `--all` for every page.
|
|
401
|
+
- With local authentication, the command fails. Pipe `gscdump sitemaps urls <sitemap-url>` into `gscdump inspect --site <site>` instead.
|
|
402
|
+
|
|
268
403
|
## Report a papercut
|
|
269
404
|
|
|
270
405
|
If CLI behavior blocks or slows your work, report it once per distinct
|
package/dist/commands/export.mjs
DELETED
|
@@ -1,76 +0,0 @@
|
|
|
1
|
-
import { OUTPUT_ARGS, applyOutputMode, displayPath } from "../utils.mjs";
|
|
2
|
-
import { allTables } from "../local-store.mjs";
|
|
3
|
-
import { createCommandContext } from "../context.mjs";
|
|
4
|
-
import { materializeParquetTables } from "../native-duckdb.mjs";
|
|
5
|
-
import { defineCommand } from "citty";
|
|
6
|
-
import path from "node:path";
|
|
7
|
-
async function exportToDuckDB(opts) {
|
|
8
|
-
const outPath = path.resolve(opts.outPath);
|
|
9
|
-
const entries = await opts.engine.listLive({
|
|
10
|
-
userId: opts.userId,
|
|
11
|
-
siteId: opts.siteId
|
|
12
|
-
});
|
|
13
|
-
const inputs = [];
|
|
14
|
-
for (const table of allTables()) {
|
|
15
|
-
const tableEntries = entries.filter((entry) => entry.table === table);
|
|
16
|
-
if (tableEntries.length > 0) inputs.push({
|
|
17
|
-
table,
|
|
18
|
-
filePaths: tableEntries.map((entry) => path.join(opts.dataDir, entry.objectKey))
|
|
19
|
-
});
|
|
20
|
-
}
|
|
21
|
-
const tables = await materializeParquetTables(outPath, inputs, opts.force);
|
|
22
|
-
return {
|
|
23
|
-
outPath,
|
|
24
|
-
tables,
|
|
25
|
-
totalRows: tables.reduce((acc, t) => acc + t.rows, 0)
|
|
26
|
-
};
|
|
27
|
-
}
|
|
28
|
-
const exportCommand = defineCommand({
|
|
29
|
-
meta: {
|
|
30
|
-
name: "export",
|
|
31
|
-
description: "Pack live Parquet partitions into a single .duckdb file for portable distribution (browser attach, CDN serving, etc.)"
|
|
32
|
-
},
|
|
33
|
-
args: {
|
|
34
|
-
out: {
|
|
35
|
-
type: "string",
|
|
36
|
-
required: true,
|
|
37
|
-
description: "Output path for the .duckdb file"
|
|
38
|
-
},
|
|
39
|
-
site: {
|
|
40
|
-
type: "string",
|
|
41
|
-
description: "Limit export to a single site URL (omit to include all)"
|
|
42
|
-
},
|
|
43
|
-
force: {
|
|
44
|
-
type: "boolean",
|
|
45
|
-
default: false,
|
|
46
|
-
description: "Overwrite the output file if it already exists"
|
|
47
|
-
},
|
|
48
|
-
...OUTPUT_ARGS
|
|
49
|
-
},
|
|
50
|
-
async run({ args }) {
|
|
51
|
-
const { json } = applyOutputMode(args);
|
|
52
|
-
const store = (await createCommandContext({ needsStore: true })).store;
|
|
53
|
-
const siteId = args.site ? store.siteIdFor(args.site) : void 0;
|
|
54
|
-
const result = await exportToDuckDB({
|
|
55
|
-
engine: store.engine,
|
|
56
|
-
dataDir: store.dataDir,
|
|
57
|
-
userId: store.userId,
|
|
58
|
-
siteId,
|
|
59
|
-
outPath: args.out,
|
|
60
|
-
force: args.force
|
|
61
|
-
});
|
|
62
|
-
if (json) {
|
|
63
|
-
console.log(JSON.stringify(result, null, 2));
|
|
64
|
-
return;
|
|
65
|
-
}
|
|
66
|
-
if (result.tables.length === 0) {
|
|
67
|
-
console.log(`\n No data to export. Run \`gscdump sync\` first.`);
|
|
68
|
-
return;
|
|
69
|
-
}
|
|
70
|
-
for (const t of result.tables) console.log(` ${t.table.padEnd(15)} ${String(t.files).padStart(4)} parquet → ${t.table} (${t.rows.toLocaleString()} rows)`);
|
|
71
|
-
console.log(`\n Exported ${result.tables.length} table(s), ${result.totalRows.toLocaleString()} rows → ${displayPath(result.outPath)}`);
|
|
72
|
-
console.log(`\n Attach from DuckDB: \x1B[36mATTACH '${result.outPath}' AS gsc (READ_ONLY); SELECT * FROM gsc.pages LIMIT 10;\x1B[0m`);
|
|
73
|
-
console.log(` Attach in a browser: use DuckDB-WASM registerFileBuffer + \x1B[36mATTACH 'gsc.duckdb' AS gsc (READ_ONLY)\x1B[0m`);
|
|
74
|
-
}
|
|
75
|
-
});
|
|
76
|
-
export { exportCommand, exportToDuckDB };
|
package/dist/native-duckdb.mjs
DELETED
|
@@ -1,46 +0,0 @@
|
|
|
1
|
-
import { rm } from "node:fs/promises";
|
|
2
|
-
import { dateColumnsFor } from "@gscdump/engine/schema";
|
|
3
|
-
import { sqlEscape } from "@gscdump/engine/sql";
|
|
4
|
-
import { dateReplaceClause } from "@gscdump/engine/sql-fragments";
|
|
5
|
-
function parquetFileListSql(filePaths) {
|
|
6
|
-
return filePaths.map((filePath) => `'${sqlEscape(filePath)}'`).join(", ");
|
|
7
|
-
}
|
|
8
|
-
async function loadDuckDB() {
|
|
9
|
-
return import("@duckdb/node-api");
|
|
10
|
-
}
|
|
11
|
-
async function readParquetRows(filePaths, table) {
|
|
12
|
-
const { DuckDBInstance } = await loadDuckDB();
|
|
13
|
-
const instance = await DuckDBInstance.create(":memory:");
|
|
14
|
-
const conn = await instance.connect();
|
|
15
|
-
try {
|
|
16
|
-
const replace = dateReplaceClause(dateColumnsFor(table), "string");
|
|
17
|
-
return (await conn.runAndReadAll(`SELECT * ${replace} FROM read_parquet([${parquetFileListSql(filePaths)}], union_by_name=true)`)).getRowObjects();
|
|
18
|
-
} finally {
|
|
19
|
-
conn.closeSync();
|
|
20
|
-
instance.closeSync();
|
|
21
|
-
}
|
|
22
|
-
}
|
|
23
|
-
async function materializeParquetTables(outPath, tables, force = false) {
|
|
24
|
-
if (force) await rm(outPath, { force: true });
|
|
25
|
-
const { DuckDBInstance } = await loadDuckDB();
|
|
26
|
-
const instance = await DuckDBInstance.create(outPath);
|
|
27
|
-
const conn = await instance.connect();
|
|
28
|
-
const results = [];
|
|
29
|
-
try {
|
|
30
|
-
for (const input of tables) {
|
|
31
|
-
const replace = dateReplaceClause(dateColumnsFor(input.table), "date");
|
|
32
|
-
await conn.run(`CREATE OR REPLACE TABLE ${input.table} AS SELECT * ${replace} FROM read_parquet([${parquetFileListSql(input.filePaths)}], union_by_name=true)`);
|
|
33
|
-
const rows = (await conn.runAndReadAll(`SELECT count(*)::BIGINT AS n FROM ${input.table}`)).getRowObjects();
|
|
34
|
-
results.push({
|
|
35
|
-
table: input.table,
|
|
36
|
-
files: input.filePaths.length,
|
|
37
|
-
rows: Number(rows[0]?.n ?? 0)
|
|
38
|
-
});
|
|
39
|
-
}
|
|
40
|
-
} finally {
|
|
41
|
-
conn.closeSync();
|
|
42
|
-
instance.closeSync();
|
|
43
|
-
}
|
|
44
|
-
return results;
|
|
45
|
-
}
|
|
46
|
-
export { materializeParquetTables, readParquetRows };
|