tablefacts 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +14 -0
- package/CHANGELOG.md +12 -0
- package/LICENSE +21 -0
- package/README.md +313 -0
- package/bin/tablefacts.mjs +28 -0
- package/package.json +77 -0
- package/src/index.mjs +52 -0
- package/src/instagram/README.md +135 -0
- package/src/instagram/download.mjs +62 -0
- package/src/instagram/index.mjs +355 -0
- package/src/instagram/links.mjs +55 -0
- package/src/instagram/record.mjs +18 -0
- package/src/lib/edge.mjs +49 -0
- package/src/lib/env.mjs +35 -0
- package/src/lib/errors.mjs +39 -0
- package/src/lib/files.mjs +10 -0
- package/src/lib/images.mjs +20 -0
- package/src/lib/log.mjs +17 -0
- package/src/lib/photos.mjs +42 -0
- package/src/lib/playwright.mjs +13 -0
- package/src/lib/project.mjs +42 -0
- package/src/lib/text.mjs +7 -0
- package/src/lib/types.mjs +247 -0
- package/src/menu/README.md +97 -0
- package/src/menu/cluvi/config.mjs +33 -0
- package/src/menu/cluvi/extract.mjs +24 -0
- package/src/menu/cluvi/import.mjs +30 -0
- package/src/menu/cluvi/source.mjs +156 -0
- package/src/menu/index.mjs +9 -0
- package/src/menu/lib/db.mjs +119 -0
- package/src/menu/lib/import.mjs +126 -0
- package/src/menu/lib/menu.mjs +95 -0
- package/src/menu/lib/run.mjs +83 -0
- package/src/menu/raw/config.mjs +38 -0
- package/src/menu/raw/extract.mjs +72 -0
- package/src/menu/raw/import.mjs +99 -0
- package/src/menu/raw/normalize.mjs +120 -0
- package/src/menu/raw/source.mjs +116 -0
- package/src/menu/raw/vision.mjs +252 -0
- package/src/research/README.md +69 -0
- package/src/research/index.mjs +147 -0
- package/src/research/lib/google.mjs +93 -0
- package/src/research/lib/hours.mjs +109 -0
- package/src/research/lib/merge.mjs +119 -0
- package/src/research/lib/osm.mjs +49 -0
- package/src/research/lib/report.mjs +118 -0
- package/src/research/lib/social.mjs +49 -0
- package/src/research/lib/util.mjs +104 -0
- package/src/research/lib/website.mjs +285 -0
- package/src/research/research.mjs +67 -0
- package/src/tripadvisor/README.md +78 -0
- package/src/tripadvisor/index.mjs +178 -0
- package/src/tripadvisor/links.mjs +78 -0
- package/src/tripadvisor/photos.mjs +55 -0
- package/types/index.d.mts +65 -0
- package/types/instagram/download.d.mts +1 -0
- package/types/instagram/index.d.mts +13 -0
- package/types/instagram/links.d.mts +14 -0
- package/types/instagram/record.d.mts +1 -0
- package/types/lib/edge.d.mts +14 -0
- package/types/lib/env.d.mts +12 -0
- package/types/lib/errors.d.mts +25 -0
- package/types/lib/files.d.mts +1 -0
- package/types/lib/images.d.mts +6 -0
- package/types/lib/log.d.mts +5 -0
- package/types/lib/photos.d.mts +30 -0
- package/types/lib/playwright.d.mts +1677 -0
- package/types/lib/project.d.mts +32 -0
- package/types/lib/text.d.mts +4 -0
- package/types/lib/types.d.mts +668 -0
- package/types/menu/cluvi/config.d.mts +13 -0
- package/types/menu/cluvi/extract.d.mts +2 -0
- package/types/menu/cluvi/import.d.mts +35 -0
- package/types/menu/cluvi/source.d.mts +37 -0
- package/types/menu/index.d.mts +6 -0
- package/types/menu/lib/db.d.mts +23 -0
- package/types/menu/lib/import.d.mts +11 -0
- package/types/menu/lib/menu.d.mts +24 -0
- package/types/menu/lib/run.d.mts +27 -0
- package/types/menu/raw/config.d.mts +18 -0
- package/types/menu/raw/extract.d.mts +2 -0
- package/types/menu/raw/import.d.mts +64 -0
- package/types/menu/raw/normalize.d.mts +29 -0
- package/types/menu/raw/source.d.mts +19 -0
- package/types/menu/raw/vision.d.mts +13 -0
- package/types/research/index.d.mts +9 -0
- package/types/research/lib/google.d.mts +12 -0
- package/types/research/lib/hours.d.mts +22 -0
- package/types/research/lib/merge.d.mts +87 -0
- package/types/research/lib/osm.d.mts +36 -0
- package/types/research/lib/report.d.mts +6 -0
- package/types/research/lib/social.d.mts +118 -0
- package/types/research/lib/util.d.mts +45 -0
- package/types/research/lib/website.d.mts +283 -0
- package/types/research/research.d.mts +1 -0
- package/types/tripadvisor/index.d.mts +11 -0
- package/types/tripadvisor/links.d.mts +13 -0
- package/types/tripadvisor/photos.d.mts +1 -0
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
# Instagram photos
|
|
2
|
+
|
|
3
|
+
Downloads a restaurant's Instagram photos through <https://toolzu.com>, driven by Playwright in **the user's own Edge window**. Use it to source real photos for a restaurant's site when it has no originals to send.
|
|
4
|
+
|
|
5
|
+
| | |
|
|
6
|
+
| --- | --- |
|
|
7
|
+
| Photos of a whole profile | `--profile <link or @name>`, needs a toolzu account signed in to that browser |
|
|
8
|
+
| Photos of specific posts | links as arguments or `--file links.txt`, saves every image of each post |
|
|
9
|
+
| Videos, stories, highlights | not supported (videos on a profile are skipped) |
|
|
10
|
+
|
|
11
|
+
**Rights first.** Only download what the restaurant owns or has allowed you to use. The restaurant sending you the originals, or using Instagram's own "Download your information" export, is better than this tool: full size, no scraping. Mention in the hand-off where each photo came from.
|
|
12
|
+
|
|
13
|
+
## The browser: read this before running anything
|
|
14
|
+
|
|
15
|
+
Toolzu guards its forms with a Cloudflare Turnstile check. **Headless browsers, Playwright's bundled Chromium, a fresh automated Edge and Playwright's `codegen` window all failed it** in testing. An ordinary Edge window that the user started with remote debugging passes it by itself, usually in 8 to 45 seconds. So the script does not drive a browser of its own: it **attaches to that window** (`--cdp http://localhost:9222`) and opens its own tab in it. **If nothing is listening on that port, the script starts Edge itself** (detached, so it stays open and keeps the toolzu login for the next run).
|
|
16
|
+
|
|
17
|
+
Rules for agents:
|
|
18
|
+
|
|
19
|
+
- **Never try to bypass, solve or fake the Cloudflare check** (no stealth plugins, header tricks, solvers). The script only waits for it (up to 2 minutes). If a "verify you are human" box appears in the window, **the user ticks it**. Tell them when you are waiting.
|
|
20
|
+
- **Never kill Edge or its processes.** The window may be the user's, and the run holds a lock on its profile folder. If a run hangs, stop only the `node` process you started.
|
|
21
|
+
- Leave the window open between runs. Closing it ends the session (toolzu login included).
|
|
22
|
+
|
|
23
|
+
### Starting the window by hand (only if the script cannot)
|
|
24
|
+
|
|
25
|
+
The script does this for you using the profile folder `C:\ig-edge` (`--edge-dir` changes it). If it reports that Edge did not open its debugging port, start it yourself:
|
|
26
|
+
|
|
27
|
+
```powershell
|
|
28
|
+
& "C:\Program Files (x86)\Microsoft\Edge\Application\msedge.exe" --remote-debugging-port=9222 --user-data-dir=C:\ig-edge
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
- The `&` is required in PowerShell (without it: `Unexpected token 'remote-debugging-port=9222'`).
|
|
32
|
+
- **Close every other Edge window first**, or Edge ignores the debugging flag and nothing listens on 9222.
|
|
33
|
+
- `--user-data-dir` must be a folder of its own (recent Edge refuses debugging on the default profile). `C:\ig-edge` keeps the toolzu login between sessions. Do not point it at `.tablefacts/instagram-profile`, which the script uses for its own fallback browser.
|
|
34
|
+
- Check it is up: `curl http://localhost:9222/json/version` returns JSON. `ECONNREFUSED ::1:9222` means it is not running: ask the user to start it (it opens a window on their screen).
|
|
35
|
+
- First time: in that window open toolzu.com, **sign in or sign up**, then leave it. The profile tool needs the account; the single-post tool does not.
|
|
36
|
+
|
|
37
|
+
## Run
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
# one profile, first batch, listing only (writes nothing)
|
|
41
|
+
tablefacts photos instagram --cdp http://localhost:9222 --google --profile https://www.instagram.com/<user>/ --out <folder> --dry-run
|
|
42
|
+
|
|
43
|
+
# same, for real; several batches
|
|
44
|
+
tablefacts photos instagram --cdp http://localhost:9222 --google --profile @<user> --pages 3 --out <folder>
|
|
45
|
+
|
|
46
|
+
# specific posts
|
|
47
|
+
tablefacts photos instagram --cdp http://localhost:9222 --out <folder> <post-link> <post-link>
|
|
48
|
+
tablefacts photos instagram --cdp http://localhost:9222 --out <folder> --file links.txt
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
| Flag | Meaning |
|
|
52
|
+
| --- | --- |
|
|
53
|
+
| `--out <folder>` | required; created if missing. Use a scratch folder outside the site's public folder, then copy what you pick |
|
|
54
|
+
| `--cdp <url>` | attach to the user's Edge, starting it if nothing answers (the supported way) |
|
|
55
|
+
| `--edge-dir <dir>` | profile folder for that Edge, where the toolzu login lives (default `C:\ig-edge`) |
|
|
56
|
+
| `--google` | reach toolzu through a Google search first, as the user does (first link only; falls back to the direct address). Library option: `viaGoogle` |
|
|
57
|
+
| `--profile <link\|@name>` | download a whole profile's photos |
|
|
58
|
+
| `--pages <n\|all>` | with `--profile`, batches of posts to load with NEXT (default 1) |
|
|
59
|
+
| `--dry-run` | print what it would save, write nothing. Do this first |
|
|
60
|
+
| `--debug` | post mode: save the results HTML in `<out>/_debug` to fix selectors |
|
|
61
|
+
| `--browser msedge`, `--headed`, `--user-data-dir` | launch a browser instead of attaching. Fails the Cloudflare check unless a person ticks it; keep as a fallback |
|
|
62
|
+
|
|
63
|
+
Links may carry tracking queries (`?utm_source=…`, `?img_index=1`); they are stripped and `img_index` is ignored: **every link saves all the images of its post**.
|
|
64
|
+
|
|
65
|
+
Output: profile photos are `<user>-<instagram file name>.jpg`; post photos are `<shortcode>-<n>.jpg`. A re-run skips files already there. The summary prints saved, already there and failed; the exit code is non-zero if anything failed.
|
|
66
|
+
|
|
67
|
+
## From code
|
|
68
|
+
|
|
69
|
+
```js
|
|
70
|
+
import { downloadInstagram } from 'tablefacts'
|
|
71
|
+
|
|
72
|
+
const { saved, skipped, failed } = await downloadInstagram({
|
|
73
|
+
links: ['https://www.instagram.com/p/<shortcode>/'], // and/or `file`, `profile`
|
|
74
|
+
out: 'photos/instagram', // relative paths resolve against `projectDir`
|
|
75
|
+
cdp: 'http://localhost:9222',
|
|
76
|
+
viaGoogle: true, // the CLI's --google
|
|
77
|
+
projectDir: '/path/to/project',
|
|
78
|
+
log: (message, level) => console.log(level ?? 'info', message),
|
|
79
|
+
})
|
|
80
|
+
for (const { item, reason } of failed) console.error(item, reason) // the post or profile link that failed
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Give at least one of `links`, `file` or `profile`. `out`, `file` and `userDataDir` resolve against `projectDir` (default
|
|
84
|
+
`TABLEFACTS_PROJECT`, then the current folder), as does the fallback browser profile `.tablefacts/instagram-profile`.
|
|
85
|
+
`log(message, level)` gets `'info'`, `'warn'` (skipped links, retries) or `'error'`; without it nothing is printed.
|
|
86
|
+
Bad arguments throw a `TablefactsError` with code `EUSAGE` and `err.option` (`'out'`, `'links'`, `'profile'`); the CLI
|
|
87
|
+
prints them with flag names and exits `2`. A missing Playwright is `EDEPENDENCY`. Post failures do not throw: they are
|
|
88
|
+
in `failed`, and the CLI exits `1` when there are any.
|
|
89
|
+
|
|
90
|
+
## What the script does (so you can fix it when toolzu changes)
|
|
91
|
+
|
|
92
|
+
1. Optionally Google `toolzu instagram profile` (or `…photo downloader`) and click the result.
|
|
93
|
+
2. Paste the link into `#instagramdownloaderform-search` and **click Download straight away**, like a person. If toolzu flags the box invalid because the check was not finished, wait for the check and click again (up to 3 times).
|
|
94
|
+
3. Results land in `#ajax-results` as `.download-card` elements. A photo has `.fa-image`, a video `.fa-play-circle`; the card's button is `a[download]`. Only photo cards are used.
|
|
95
|
+
4. **Both modes click each card's own Download button** and save what the browser receives (a real download, or an image opened in a new tab, which the script fetches with the browser's cookies). The click is a click event sent straight to the button: a real Playwright click scrolls each card into view and waits for the lazy-loading images to stop shifting, which made later batches about 3x slower. A real click is only the fallback. The `NEXT` button (`#viewer-next`) adds another batch to the list; each batch is saved before NEXT is pressed.
|
|
96
|
+
5. It waits for the cards to appear, never for `networkidle` (toolzu's ads keep the network busy for tens of seconds). Waits 2.5 seconds between links.
|
|
97
|
+
|
|
98
|
+
Everything toolzu-specific is `findImages`, `downloadProfile`, `profileImages`, `submitForm` and `waitForCheck` in `download.mjs`. Link parsing, saving, retries and the summary do not depend on toolzu.
|
|
99
|
+
|
|
100
|
+
## Verified, and not
|
|
101
|
+
|
|
102
|
+
Verified on the Zelavi profile and one carousel post, signed in to toolzu:
|
|
103
|
+
- **Profile, `--pages 3`:** Google to toolzu, Download clicked right after pasting, then 27 + 32 + 48 = 107 photos saved with no failures in 1 minute 2 seconds (about 0.4 s per photo in every batch). All valid JPEGs at full resolution (about 1080 px wide). NEXT adds to the list and did not need another Cloudflare tick. The 4 videos on the first page were skipped.
|
|
104
|
+
- **Post mode:** a 3-image carousel saved as `<shortcode>-1.jpg` to `-3.jpg`, all valid JPEGs; a re-run skipped all 3; a non-Instagram link was reported and skipped.
|
|
105
|
+
- **Starting Edge:** the script started it when port 9222 was closed and reused it afterwards.
|
|
106
|
+
|
|
107
|
+
**Not verified, so check before relying on them:**
|
|
108
|
+
- A profile with hundreds of posts: `--pages all` has not been run, and toolzu may limit or slow later batches.
|
|
109
|
+
- Private accounts, reels in post mode, and a toolzu account that is not signed in.
|
|
110
|
+
- `record.mjs` (attach to the window and open Playwright's Inspector to record a flow): the user reported recording did not work. Ask them to describe or screenshot the steps instead.
|
|
111
|
+
|
|
112
|
+
## Troubleshooting
|
|
113
|
+
|
|
114
|
+
| Message | Cause and fix |
|
|
115
|
+
| --- | --- |
|
|
116
|
+
| `Edge did not open its debugging port` | another Edge is using the `C:\ig-edge` profile without debugging, or Edge is not in its usual folder; close that window or start Edge by hand (see above) |
|
|
117
|
+
| `connect ECONNREFUSED ::1:9222` | `--cdp` points at a non-local address, or Edge died mid-run; run again |
|
|
118
|
+
| `the Cloudflare check was not passed` | the user must tick the box in the window, or the window is not a normal Edge |
|
|
119
|
+
| `not signed in to toolzu` | sign in in the debugging window; the profile tool may refuse otherwise |
|
|
120
|
+
| `toolzu did not accept the link` | link is not a public profile or post, or toolzu is blocking the request |
|
|
121
|
+
| `the tool returned no photos` / `no images` | markup changed or the account is private; run with `--debug` (post mode) and compare with the selectors above |
|
|
122
|
+
| `clicking Download started nothing` / `Download navigated instead of saving` | the card button now behaves differently; check by hand what happens when a person clicks it |
|
|
123
|
+
| `launchPersistentContext … closed` | another run still holds `.tablefacts/instagram-profile`; stop the `node` process you started |
|
|
124
|
+
|
|
125
|
+
## After the download
|
|
126
|
+
|
|
127
|
+
- Most of a restaurant's feed is event posters and guest photos with text. Look at every image before using it; posters with text rarely belong on the site.
|
|
128
|
+
- Copy the chosen files into your site's public folder, and record real pixel `width`/`height` and an `alt` text for each.
|
|
129
|
+
- In the hand-off list the profile, how many photos were taken, and whether the restaurant confirmed they may be used.
|
|
130
|
+
|
|
131
|
+
## Shell notes for agents on Windows
|
|
132
|
+
|
|
133
|
+
- Run the commands in PowerShell. In Git Bash, a long heredoc containing quotes can break and write nothing: put scripts in a file instead.
|
|
134
|
+
- `pkill` does not exist in Git Bash; use `Get-CimInstance Win32_Process` in PowerShell and match the command line narrowly (a loose pattern also matches your own shell).
|
|
135
|
+
- Playwright's ESM import needs `file:///C:/...` URLs for absolute paths on Windows.
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
// Downloads the photos of Instagram posts through https://toolzu.com/downloader/instagram/photo/
|
|
2
|
+
// Usage: tablefacts photos instagram --out <folder> [--cdp <url>] [--profile <link|@name>] [--file links.txt] [link...]
|
|
3
|
+
// The work is in index.mjs (downloadInstagram); this file is only the command line.
|
|
4
|
+
import { TablefactsError, cliMessage, exitCodeFor } from '../lib/errors.mjs'
|
|
5
|
+
import { consoleLog } from '../lib/log.mjs'
|
|
6
|
+
import { downloadInstagram } from './index.mjs'
|
|
7
|
+
import { parseArgs } from './links.mjs'
|
|
8
|
+
|
|
9
|
+
const HELP = `Download Instagram post photos through toolzu.com.
|
|
10
|
+
|
|
11
|
+
tablefacts photos instagram --out <folder> [options] [link...]
|
|
12
|
+
|
|
13
|
+
--out <folder> where the images are saved (required, created if missing)
|
|
14
|
+
--file <path> text file with one link per line (# comments and blank lines ignored)
|
|
15
|
+
--cdp <url> attach to Edge on this debugging address, e.g. http://localhost:9222. If nothing is
|
|
16
|
+
listening there the script starts Edge itself (the most reliable way past
|
|
17
|
+
toolzu's Cloudflare check)
|
|
18
|
+
--edge-dir <dir> profile folder of that Edge, where the toolzu login is kept (default C:\ig-edge)
|
|
19
|
+
--google reach toolzu through a Google search, like a person would
|
|
20
|
+
--browser <name> use an installed browser: msedge or chrome (default: bundled Chromium)
|
|
21
|
+
--user-data-dir <dir> browser profile kept between runs (default: .tablefacts/instagram-profile)
|
|
22
|
+
--profile <link|@name> download the photos of a whole Instagram profile (videos are skipped);
|
|
23
|
+
needs a toolzu account signed in to the browser
|
|
24
|
+
--pages <n|all> with --profile, how many batches of posts to load with NEXT (default 1)
|
|
25
|
+
--headed show the browser (to watch, or to pass a challenge by hand)
|
|
26
|
+
--dry-run print the image URLs found, save nothing
|
|
27
|
+
--debug save the page HTML after each submit into <folder>/_debug
|
|
28
|
+
--help this text
|
|
29
|
+
|
|
30
|
+
Every link saves all the images of its post; ?img_index=N in a link is ignored.
|
|
31
|
+
|
|
32
|
+
Only download photos the restaurant owns or has allowed you to use.`
|
|
33
|
+
|
|
34
|
+
const FLAGS = {
|
|
35
|
+
out: '--out <folder>',
|
|
36
|
+
file: '--file',
|
|
37
|
+
profile: '--profile',
|
|
38
|
+
links: 'the links',
|
|
39
|
+
cdp: '--cdp',
|
|
40
|
+
debug: '--debug',
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
let summary
|
|
44
|
+
try {
|
|
45
|
+
const opts = parseArgs(process.argv.slice(2))
|
|
46
|
+
if (opts.help) {
|
|
47
|
+
console.log(HELP)
|
|
48
|
+
process.exit(0)
|
|
49
|
+
}
|
|
50
|
+
if (!opts.out) {
|
|
51
|
+
console.error('Missing --out <folder>.\n\n' + HELP)
|
|
52
|
+
process.exit(2)
|
|
53
|
+
}
|
|
54
|
+
summary = await downloadInstagram({ ...opts, log: consoleLog })
|
|
55
|
+
} catch (err) {
|
|
56
|
+
if (!(err instanceof TablefactsError)) throw err
|
|
57
|
+
console.error(cliMessage(err, FLAGS))
|
|
58
|
+
process.exit(exitCodeFor(err))
|
|
59
|
+
}
|
|
60
|
+
console.log(`\nSaved ${summary.saved}, already there ${summary.skipped}, failed ${summary.failed.length}.`)
|
|
61
|
+
for (const f of summary.failed) console.log(` ${f.item}: ${cliMessage({ message: f.reason }, FLAGS)}`)
|
|
62
|
+
process.exit(summary.failed.length ? 1 : 0)
|
|
@@ -0,0 +1,355 @@
|
|
|
1
|
+
// Library entry for downloading Instagram post photos through https://toolzu.com/downloader/instagram/photo/
|
|
2
|
+
// Silent unless a `log` function is given; importing this file loads nothing heavy (playwright is loaded on use).
|
|
3
|
+
import { readFile, writeFile } from 'node:fs/promises'
|
|
4
|
+
import { join } from 'node:path'
|
|
5
|
+
import { ensureEdge } from '../lib/edge.mjs'
|
|
6
|
+
import { TablefactsError, optionError } from '../lib/errors.mjs'
|
|
7
|
+
import { assertImageResponse, writeImage } from '../lib/images.mjs'
|
|
8
|
+
import { normalizeLog } from '../lib/log.mjs'
|
|
9
|
+
import { DELAY_MS, newSummary, prepareOut, storeFile } from '../lib/photos.mjs'
|
|
10
|
+
import { loadPlaywright } from '../lib/playwright.mjs'
|
|
11
|
+
import { resolveIn, workDirIn } from '../lib/project.mjs'
|
|
12
|
+
import { normalize, parseProfile } from './links.mjs'
|
|
13
|
+
|
|
14
|
+
const TOOL = 'https://toolzu.com/downloader/instagram/photo/'
|
|
15
|
+
const PROFILE_TOOL = 'https://toolzu.com/downloader/instagram/profile/'
|
|
16
|
+
const IMAGE_HOSTS = /(cdninstagram\.com|fbcdn\.net)$/
|
|
17
|
+
|
|
18
|
+
async function readLinks(links, file) {
|
|
19
|
+
const raw = [...links]
|
|
20
|
+
if (file) {
|
|
21
|
+
const text = await readFile(file, 'utf8')
|
|
22
|
+
for (const line of text.split(/\r?\n/)) {
|
|
23
|
+
const t = line.replace(/#.*/, '').trim()
|
|
24
|
+
if (t) raw.push(t)
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
const seen = new Map()
|
|
28
|
+
const bad = []
|
|
29
|
+
for (const r of raw) {
|
|
30
|
+
const n = normalize(r)
|
|
31
|
+
if (!n) bad.push(r)
|
|
32
|
+
else if (!seen.has(n.shortcode)) seen.set(n.shortcode, n)
|
|
33
|
+
}
|
|
34
|
+
return { posts: [...seen.values()], bad }
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
// Runs `attempt` once more if it fails the first time, telling `onRetry` why.
|
|
38
|
+
async function retry(attempt, onRetry) {
|
|
39
|
+
try {
|
|
40
|
+
return await attempt()
|
|
41
|
+
} catch (err) {
|
|
42
|
+
onRetry(err)
|
|
43
|
+
return attempt()
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
// Toolzu guards its forms with a Cloudflare Turnstile that fills a hidden field when passed.
|
|
48
|
+
// Headless fails it. In a real browser (cdp) it passes alone, sometimes after 40 seconds;
|
|
49
|
+
// if it shows a box, tick it. We only wait for it, never try to solve it.
|
|
50
|
+
async function waitForCheck(page, log, visible) {
|
|
51
|
+
if (visible) log(' waiting for the Cloudflare check: tick it in the browser window if it asks (2 min)')
|
|
52
|
+
await page
|
|
53
|
+
.waitForFunction(() => [...document.querySelectorAll('input[name*=turnstile]')].some((i) => i.value), null, { timeout: 120000 })
|
|
54
|
+
.catch(() => {
|
|
55
|
+
throw new TablefactsError('the Cloudflare check was not passed (use `cdp` with your own browser, or solve it by hand)', 'EFAILED')
|
|
56
|
+
})
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// Clicks Download right after the link is pasted, as a person does. If toolzu answers that the
|
|
60
|
+
// check was not finished (it marks the box invalid), wait for the check and click again.
|
|
61
|
+
async function submitForm(page, path, log, visible) {
|
|
62
|
+
for (let attempt = 0; attempt < 3; attempt++) {
|
|
63
|
+
const reply = page
|
|
64
|
+
.waitForResponse((r) => r.url().includes(path) && r.request().method() === 'POST', { timeout: attempt ? 60000 : 10000 })
|
|
65
|
+
.catch(() => null)
|
|
66
|
+
await page.click('#downloader-form button[type=submit]')
|
|
67
|
+
const answered = await reply
|
|
68
|
+
await page.waitForTimeout(500)
|
|
69
|
+
if (answered && !(await page.locator('#instagramdownloaderform-search.is-invalid').count())) return
|
|
70
|
+
await waitForCheck(page, log, visible)
|
|
71
|
+
}
|
|
72
|
+
throw new TablefactsError('toolzu did not accept the link', 'EFAILED')
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// Opens the tool from a Google search: type the query, click the toolzu result.
|
|
76
|
+
// Falls back to the direct address if Google asks for its own check or shows no result.
|
|
77
|
+
async function openFromGoogle(page, query, resultHref, log) {
|
|
78
|
+
try {
|
|
79
|
+
await page.goto('https://www.google.com/', { waitUntil: 'domcontentloaded' })
|
|
80
|
+
await page.getByRole('button', { name: /accept all|aceptar todo/i }).click({ timeout: 3000 }).catch(() => {})
|
|
81
|
+
const box = page.locator('textarea[name=q], input[name=q]').first()
|
|
82
|
+
await box.click({ timeout: 10000 })
|
|
83
|
+
await box.pressSequentially(query, { delay: 90 })
|
|
84
|
+
await page.keyboard.press('Enter')
|
|
85
|
+
await page.locator(`a[href*="${resultHref}"]`).first().click({ timeout: 20000 })
|
|
86
|
+
await page.waitForSelector('#instagramdownloaderform-search', { timeout: 20000 })
|
|
87
|
+
return true
|
|
88
|
+
} catch {
|
|
89
|
+
log(' Google did not lead to toolzu; opening it directly', 'warn')
|
|
90
|
+
return false
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// Submits one post link to the tool and returns the image URLs it offers.
|
|
95
|
+
async function findImages(page, post, debugDir, { log, visible, google }) {
|
|
96
|
+
const opened = google && (await openFromGoogle(page, 'toolzu instagram photo downloader', 'toolzu.com/downloader/instagram/photo', log))
|
|
97
|
+
if (!opened) await page.goto(TOOL, { waitUntil: 'domcontentloaded' })
|
|
98
|
+
await page.getByRole('button', { name: 'Got it' }).click({ timeout: 2000 }).catch(() => {})
|
|
99
|
+
await page.fill('#instagramdownloaderform-search', post.url)
|
|
100
|
+
await submitForm(page, '/downloader/instagram/photo', log, visible)
|
|
101
|
+
// The results are injected after the response; the cards are what we wait for.
|
|
102
|
+
await page.waitForSelector('#ajax-results .download-card', { timeout: 30000 }).catch(() => {})
|
|
103
|
+
if (debugDir) await writeFile(join(debugDir, `${post.shortcode}.html`), await page.content())
|
|
104
|
+
|
|
105
|
+
// Each result has a download button: <a download href="…cdninstagram.com/…">. Read those first
|
|
106
|
+
// and only fall back to every link and image on the page if the markup changes.
|
|
107
|
+
// `index` is the button's position among the page's Download buttons, so it can be clicked.
|
|
108
|
+
const { urls, buttons } = await page.evaluate(() => {
|
|
109
|
+
const buttons = [...document.querySelectorAll('#ajax-results a[download]')].map((a) => a.href)
|
|
110
|
+
if (buttons.length) return { urls: buttons, buttons: true }
|
|
111
|
+
return { urls: [...document.querySelectorAll('#ajax-results a[href], #ajax-results img[src]')].map((el) => el.href ?? el.src), buttons: false }
|
|
112
|
+
})
|
|
113
|
+
const images = []
|
|
114
|
+
for (const [index, url] of urls.entries()) {
|
|
115
|
+
let host
|
|
116
|
+
try {
|
|
117
|
+
host = new URL(url).hostname
|
|
118
|
+
} catch {
|
|
119
|
+
continue
|
|
120
|
+
}
|
|
121
|
+
if (IMAGE_HOSTS.test(host) && !images.some((i) => i.url === url)) images.push({ url, index: buttons ? index : null })
|
|
122
|
+
}
|
|
123
|
+
return images
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
// The photo cards currently on the profile results page; videos (play icon) are left out.
|
|
127
|
+
function profileImages(page) {
|
|
128
|
+
return page.evaluate(() =>
|
|
129
|
+
[...document.querySelectorAll('#ajax-results .download-card')]
|
|
130
|
+
.filter((card) => card.querySelector('.fa-image'))
|
|
131
|
+
.map((card) => card.querySelector('a[download]')?.href)
|
|
132
|
+
.filter(Boolean),
|
|
133
|
+
)
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// Loads a profile through the profile tool and calls `handle` with the photo cards of each
|
|
137
|
+
// batch, following NEXT for `maxPages` batches. Returns how many photos it saw.
|
|
138
|
+
async function downloadProfile(page, profile, maxPages, handle, { log, visible, google }) {
|
|
139
|
+
const opened = google && (await openFromGoogle(page, 'toolzu instagram profile', 'toolzu.com/downloader/instagram/profile', log))
|
|
140
|
+
if (!opened) await page.goto(PROFILE_TOOL, { waitUntil: 'domcontentloaded' })
|
|
141
|
+
await page.getByRole('button', { name: 'Got it' }).click({ timeout: 2000 }).catch(() => {})
|
|
142
|
+
if (!(await page.getByText('Logout').count())) log(' not signed in to toolzu: the profile tool may ask you to sign up', 'warn')
|
|
143
|
+
await page.fill('#instagramdownloaderform-search', profile.url)
|
|
144
|
+
await submitForm(page, '/downloader/instagram/profile', log, visible)
|
|
145
|
+
await page.waitForSelector('#ajax-results .download-card', { timeout: 60000 }).catch(() => {})
|
|
146
|
+
|
|
147
|
+
const found = new Set()
|
|
148
|
+
// Hands over the cards not seen yet, with their position among the photo cards on the page.
|
|
149
|
+
const take = async () => {
|
|
150
|
+
const fresh = (await profileImages(page)).map((url, index) => ({ url, index })).filter((c) => !found.has(c.url))
|
|
151
|
+
for (const c of fresh) found.add(c.url)
|
|
152
|
+
if (fresh.length) await handle(fresh)
|
|
153
|
+
return fresh.length
|
|
154
|
+
}
|
|
155
|
+
await take()
|
|
156
|
+
for (let n = 1; n < maxPages; n++) {
|
|
157
|
+
const next = page.locator('#viewer-next')
|
|
158
|
+
if (!(await next.count()) || !(await next.isVisible())) break
|
|
159
|
+
log(` loading batch ${n + 1} (${found.size} photos so far)`)
|
|
160
|
+
const before = await page.locator('#ajax-results .download-card').count()
|
|
161
|
+
const firstHref = await page.locator('#ajax-results a[download]').first().getAttribute('href')
|
|
162
|
+
await next.click()
|
|
163
|
+
// The next batch is in once there are more cards than before, or the first one is a different
|
|
164
|
+
// photo (in case NEXT replaces the list instead of adding to it).
|
|
165
|
+
await page
|
|
166
|
+
.waitForFunction(
|
|
167
|
+
({ n, first }) =>
|
|
168
|
+
document.querySelectorAll('#ajax-results .download-card').length > n ||
|
|
169
|
+
document.querySelector('#ajax-results a[download]')?.getAttribute('href') !== first,
|
|
170
|
+
{ n: before, first: firstHref },
|
|
171
|
+
{ timeout: 90000 },
|
|
172
|
+
)
|
|
173
|
+
.catch(() => log(' NEXT brought no new photos', 'warn'))
|
|
174
|
+
const added = await take()
|
|
175
|
+
if (!added && (await page.locator('#ajax-results .download-card').count()) <= before) break
|
|
176
|
+
await page.waitForTimeout(DELAY_MS)
|
|
177
|
+
}
|
|
178
|
+
return found.size
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
// Clicks a card's Download button and saves what the browser receives. Depending on the link it
|
|
182
|
+
// arrives as a real download or opens the image in a new tab; both are handled.
|
|
183
|
+
async function saveByClick(page, link, file) {
|
|
184
|
+
const context = page.context()
|
|
185
|
+
const before = page.url()
|
|
186
|
+
// First a click event sent straight to the button: nothing scrolls and nothing waits for the
|
|
187
|
+
// layout to settle, which on a long list of lazy-loading cards is what makes a real click slow.
|
|
188
|
+
// If that starts nothing, a real click (scroll, wait, click) is the fallback.
|
|
189
|
+
let got = null
|
|
190
|
+
for (const real of [false, true]) {
|
|
191
|
+
const timeout = real ? 20000 : 8000
|
|
192
|
+
const started = Promise.race([
|
|
193
|
+
page.waitForEvent('download', { timeout }).then((download) => ({ download })),
|
|
194
|
+
context.waitForEvent('page', { timeout }).then((tab) => ({ tab })),
|
|
195
|
+
])
|
|
196
|
+
await (real ? link.click() : link.dispatchEvent('click'))
|
|
197
|
+
got = await started.catch(() => null)
|
|
198
|
+
if (got) break
|
|
199
|
+
}
|
|
200
|
+
if (!got) throw new TablefactsError('clicking Download started nothing', 'EFAILED')
|
|
201
|
+
if (got.download) return got.download.saveAs(file)
|
|
202
|
+
await got.tab.waitForURL((u) => u.toString() !== 'about:blank', { timeout: 15000 }).catch(() => {})
|
|
203
|
+
const url = got.tab.url()
|
|
204
|
+
await got.tab.close()
|
|
205
|
+
if (page.url() !== before) throw new TablefactsError('the results page was left; Download navigated instead of saving', 'EFAILED')
|
|
206
|
+
const res = await context.request.get(url)
|
|
207
|
+
const bytes = await res.body()
|
|
208
|
+
assertImageResponse({ ok: res.ok(), status: res.status(), contentType: res.headers()['content-type'], bytes })
|
|
209
|
+
await writeImage(file, bytes)
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
async function saveUrl(url, file) {
|
|
213
|
+
const res = await fetch(url)
|
|
214
|
+
const bytes = Buffer.from(await res.arrayBuffer())
|
|
215
|
+
assertImageResponse({ ok: res.ok, status: res.status, contentType: res.headers.get('content-type'), bytes })
|
|
216
|
+
await writeImage(file, bytes)
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* Downloads the photos of Instagram posts (and optionally a whole profile) through toolzu.com.
|
|
221
|
+
* At least one of `links`, `file` or `profile` is required. Resolves to a summary whose `failed`
|
|
222
|
+
* lists `{ item, reason }` and whose `found` is present only in a dry run. Throws a TablefactsError
|
|
223
|
+
* on invalid arguments (code 'EUSAGE', with `option`), a missing playwright ('EDEPENDENCY') or a
|
|
224
|
+
* browser that cannot be reached.
|
|
225
|
+
* `out`, `file` and `userDataDir` resolve against `projectDir` (default: TABLEFACTS_PROJECT or the
|
|
226
|
+
* current folder). `viaGoogle` reaches toolzu through a Google search. `log(message, level)` gets
|
|
227
|
+
* level 'info', 'warn' or 'error'.
|
|
228
|
+
* @param {import('../lib/types.mjs').InstagramOptions} options
|
|
229
|
+
* @returns {Promise<import('../lib/types.mjs').PhotoSummary>}
|
|
230
|
+
*/
|
|
231
|
+
export async function downloadInstagram({
|
|
232
|
+
links = [],
|
|
233
|
+
file,
|
|
234
|
+
out,
|
|
235
|
+
profile,
|
|
236
|
+
pages = 1,
|
|
237
|
+
cdp,
|
|
238
|
+
edgeDir,
|
|
239
|
+
viaGoogle = false,
|
|
240
|
+
browser,
|
|
241
|
+
userDataDir,
|
|
242
|
+
projectDir,
|
|
243
|
+
headed = false,
|
|
244
|
+
dryRun = false,
|
|
245
|
+
debug = false,
|
|
246
|
+
log: logOption,
|
|
247
|
+
} = {}) {
|
|
248
|
+
const log = normalizeLog(logOption)
|
|
249
|
+
if (!out) throw optionError('out', '`out` is required')
|
|
250
|
+
const { posts, bad } = await readLinks(links, file && resolveIn(projectDir, file))
|
|
251
|
+
for (const b of bad) log(`Skipping, not an Instagram post link: ${b}`, 'warn')
|
|
252
|
+
const profileInfo = profile ? parseProfile(profile) : null
|
|
253
|
+
if (profile && !profileInfo) throw optionError('profile', `\`profile\` is not an Instagram profile: ${profile}`)
|
|
254
|
+
if (!posts.length && !profileInfo) throw optionError('links', 'no valid Instagram post links in `links`/`file`/`profile`')
|
|
255
|
+
|
|
256
|
+
const { outDir, debugDir } = await prepareOut({ out, dryRun, debug, projectDir })
|
|
257
|
+
const summary = newSummary({ dryRun })
|
|
258
|
+
const visible = headed || Boolean(cdp) || viaGoogle
|
|
259
|
+
// Through Google once, for the first page opened only; later ones open the tool directly.
|
|
260
|
+
let usedGoogle = false
|
|
261
|
+
const takeGoogle = () => viaGoogle && !usedGoogle && (usedGoogle = true)
|
|
262
|
+
|
|
263
|
+
const { chromium } = await loadPlaywright()
|
|
264
|
+
let context
|
|
265
|
+
let close
|
|
266
|
+
let page
|
|
267
|
+
try {
|
|
268
|
+
if (cdp) {
|
|
269
|
+
// Your own browser: our tab is opened and closed here, the browser itself is left running.
|
|
270
|
+
await ensureEdge(cdp, edgeDir)
|
|
271
|
+
const connection = await chromium.connectOverCDP(cdp)
|
|
272
|
+
context = connection.contexts()[0]
|
|
273
|
+
close = async () => {
|
|
274
|
+
try {
|
|
275
|
+
await page?.close()
|
|
276
|
+
} finally {
|
|
277
|
+
await connection.close()
|
|
278
|
+
}
|
|
279
|
+
}
|
|
280
|
+
page = await context.newPage()
|
|
281
|
+
} else {
|
|
282
|
+
// A persistent profile keeps cookies between runs, which helps with the Cloudflare check.
|
|
283
|
+
context = await chromium.launchPersistentContext(resolveIn(projectDir, userDataDir ?? workDirIn(projectDir, 'instagram-profile')), {
|
|
284
|
+
channel: browser,
|
|
285
|
+
headless: !headed,
|
|
286
|
+
args: ['--disable-blink-features=AutomationControlled'],
|
|
287
|
+
})
|
|
288
|
+
page = context.pages()[0] ?? (await context.newPage())
|
|
289
|
+
close = () => context.close()
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
// Saves one image unless the file is already there (or only reports it in a dry run).
|
|
293
|
+
// `link` is the locator of the card's own Download button, or null to fetch the URL directly.
|
|
294
|
+
const store = (name, url, link) =>
|
|
295
|
+
storeFile({
|
|
296
|
+
outDir,
|
|
297
|
+
name,
|
|
298
|
+
url,
|
|
299
|
+
dryRun,
|
|
300
|
+
summary,
|
|
301
|
+
log,
|
|
302
|
+
save: ({ file }) => (link ? saveByClick(page, link, file) : saveUrl(url, file)),
|
|
303
|
+
})
|
|
304
|
+
|
|
305
|
+
if (profileInfo) {
|
|
306
|
+
log(`Profile ${profileInfo.url}`)
|
|
307
|
+
try {
|
|
308
|
+
const maxPages = pages === 'all' ? Infinity : Math.max(1, Number(pages) || 1)
|
|
309
|
+
const handle = async (cards) => {
|
|
310
|
+
log(` ${cards.length} photos`)
|
|
311
|
+
for (const { url, index } of cards) {
|
|
312
|
+
// The file name from Instagram is unique per photo, so a re-run skips what is saved.
|
|
313
|
+
const stem = new URL(url).pathname.split('/').pop().replace(/\.\w+$/, '').replace(/_n$/, '')
|
|
314
|
+
const name = `${profileInfo.user}-${stem}.jpg`
|
|
315
|
+
const before = summary.saved
|
|
316
|
+
// The card's own Download button, as a person clicks it.
|
|
317
|
+
await store(name, url, page.locator('#ajax-results .download-card:has(.fa-image) a[download]').nth(index))
|
|
318
|
+
if (summary.saved > before) await page.waitForTimeout(150)
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
const seen = await downloadProfile(page, profileInfo, maxPages, handle, { log, visible, google: takeGoogle() })
|
|
322
|
+
if (!seen) throw new TablefactsError('the tool returned no photos', 'EFAILED')
|
|
323
|
+
} catch (err) {
|
|
324
|
+
summary.failed.push({ item: profileInfo.url, reason: err.message })
|
|
325
|
+
log(` failed: ${err.message}`, 'error')
|
|
326
|
+
}
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
for (const [i, post] of posts.entries()) {
|
|
330
|
+
log(`[${i + 1}/${posts.length}] ${post.url}`)
|
|
331
|
+
try {
|
|
332
|
+
const images = await retry(
|
|
333
|
+
async () => {
|
|
334
|
+
const found = await findImages(page, post, debugDir, { log, visible, google: takeGoogle() })
|
|
335
|
+
if (!found.length) throw new TablefactsError('the tool returned no images', 'EFAILED')
|
|
336
|
+
return found
|
|
337
|
+
},
|
|
338
|
+
(err) => log(` retrying (${err.message})`, 'warn'),
|
|
339
|
+
)
|
|
340
|
+
for (const [n, { url, index }] of images.entries()) {
|
|
341
|
+
// The card's own Download button when we know which one it is, as a person clicks it.
|
|
342
|
+
const link = index === null ? null : page.locator('#ajax-results a[download]').nth(index)
|
|
343
|
+
await store(`${post.shortcode}-${n + 1}.jpg`, url, link)
|
|
344
|
+
}
|
|
345
|
+
} catch (err) {
|
|
346
|
+
summary.failed.push({ item: post.url, reason: err.message })
|
|
347
|
+
log(` failed: ${err.message}`, 'error')
|
|
348
|
+
}
|
|
349
|
+
if (i < posts.length - 1) await page.waitForTimeout(DELAY_MS)
|
|
350
|
+
}
|
|
351
|
+
} finally {
|
|
352
|
+
await close?.().catch(() => {})
|
|
353
|
+
}
|
|
354
|
+
return summary
|
|
355
|
+
}
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
// The pure parts of download.mjs: reading the command line and the links it is given.
|
|
2
|
+
// Split out so they can be tested without starting a browser.
|
|
3
|
+
|
|
4
|
+
import { usageError } from '../lib/errors.mjs'
|
|
5
|
+
|
|
6
|
+
export function parseArgs(argv) {
|
|
7
|
+
const opts = { links: [], headed: false, dryRun: false, debug: false }
|
|
8
|
+
for (let i = 0; i < argv.length; i++) {
|
|
9
|
+
const a = argv[i]
|
|
10
|
+
if (a === '--out') opts.out = argv[++i]
|
|
11
|
+
else if (a === '--cdp') opts.cdp = argv[++i]
|
|
12
|
+
else if (a === '--edge-dir') opts.edgeDir = argv[++i]
|
|
13
|
+
else if (a === '--google') opts.viaGoogle = true
|
|
14
|
+
else if (a === '--file') opts.file = argv[++i]
|
|
15
|
+
else if (a === '--browser') opts.browser = argv[++i]
|
|
16
|
+
else if (a === '--profile') opts.profile = argv[++i]
|
|
17
|
+
else if (a === '--user-data-dir') opts.userDataDir = argv[++i]
|
|
18
|
+
else if (a === '--pages') opts.pages = argv[++i]
|
|
19
|
+
else if (a === '--headed') opts.headed = true
|
|
20
|
+
else if (a === '--dry-run') opts.dryRun = true
|
|
21
|
+
else if (a === '--debug') opts.debug = true
|
|
22
|
+
else if (a === '--help' || a === '-h') opts.help = true
|
|
23
|
+
else if (a.startsWith('--')) throw usageError(`Unknown option ${a}`)
|
|
24
|
+
else opts.links.push(a)
|
|
25
|
+
}
|
|
26
|
+
return opts
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
// Keeps only post and reel links and drops the tracking query, so duplicates collapse.
|
|
30
|
+
export function normalize(raw) {
|
|
31
|
+
let url
|
|
32
|
+
try {
|
|
33
|
+
url = new URL(raw.trim())
|
|
34
|
+
} catch {
|
|
35
|
+
return null
|
|
36
|
+
}
|
|
37
|
+
if (!/(^|\.)instagram\.com$/.test(url.hostname)) return null
|
|
38
|
+
const m = url.pathname.match(/\/(p|reel|reels|tv)\/([\w-]+)/)
|
|
39
|
+
if (!m) return null
|
|
40
|
+
return { shortcode: m[2], url: `https://www.instagram.com/${m[1] === 'reels' ? 'reel' : m[1]}/${m[2]}/` }
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// '@name', 'name' or a profile link -> { user, url }
|
|
44
|
+
export function parseProfile(raw) {
|
|
45
|
+
let user = raw.trim()
|
|
46
|
+
if (/^https?:/i.test(user)) {
|
|
47
|
+
try {
|
|
48
|
+
user = new URL(user).pathname.split('/').filter(Boolean)[0] ?? ''
|
|
49
|
+
} catch {
|
|
50
|
+
user = ''
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
user = user.replace(/^@/, '')
|
|
54
|
+
return /^[\w.]+$/.test(user) ? { user, url: `https://www.instagram.com/${user}/` } : null
|
|
55
|
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
// Records what you do on a page, in your own browser, as Playwright code.
|
|
2
|
+
// Start Edge with --remote-debugging-port=9222 first, then:
|
|
3
|
+
// node src/instagram/record.mjs [url]
|
|
4
|
+
// In the Inspector window press Record, do the steps in the browser tab, copy the code.
|
|
5
|
+
import { loadPlaywright } from '../lib/playwright.mjs'
|
|
6
|
+
|
|
7
|
+
try {
|
|
8
|
+
const { chromium } = await loadPlaywright()
|
|
9
|
+
const url = process.argv[2] ?? 'https://www.google.com/'
|
|
10
|
+
const browser = await chromium.connectOverCDP('http://localhost:9222')
|
|
11
|
+
const page = await browser.contexts()[0].newPage()
|
|
12
|
+
await page.goto(url)
|
|
13
|
+
await page.pause()
|
|
14
|
+
await browser.close()
|
|
15
|
+
} catch (err) {
|
|
16
|
+
console.error(err.message)
|
|
17
|
+
process.exitCode = 1
|
|
18
|
+
}
|