pi-quiver 4.1.1 → 4.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +8 -0
- package/README.md +30 -3
- package/dist/bin/pi-quiver.js +604 -0
- package/extensions/fetch.ts +7 -571
- package/lib/fetch-core.ts +616 -0
- package/package.json +9 -2
package/CHANGELOG.md
CHANGED
|
@@ -8,6 +8,14 @@ Published to npm as `pi-quiver` (`pi install npm:pi-quiver`). Pushing a
|
|
|
8
8
|
via OIDC trusted publishing. The release helper at
|
|
9
9
|
`.agents/skills/release/scripts/release.sh` cuts the tag; CI publishes.
|
|
10
10
|
|
|
11
|
+
## v4.3.0 - 2026-08-25
|
|
12
|
+
|
|
13
|
+
- fetch: GitHub Actions job URLs (`.../actions/runs/<runId>/job/<jobId>`, singular `/job/` as GitHub's UI produces) now route through `gh run view --job <jobId> --repo <slug>`; plural `/jobs/<id>` paths still fall back to plain HTTP. Both run and job fetches now make a best-effort second `gh run view ... --log-failed` call and append its output under a `## Failed step logs` heading; when nothing failed, the run is still in progress, or logs have expired, the call yields no section and behavior is unchanged (summary-only).
|
|
14
|
+
|
|
15
|
+
## v4.2.0 - 2026-08-20
|
|
16
|
+
|
|
17
|
+
- fetch: data plane extracted to `lib/fetch-core.ts`; new `pi-quiver fetch` CLI (esbuild-built `dist/` bin) with full parameter parity; Claude Code skill + plugin marketplace (`quiver:fetch` via `npx -y pi-quiver@latest`). pi tool behavior unchanged.
|
|
18
|
+
|
|
11
19
|
## v4.1.1 - 2026-08-17
|
|
12
20
|
|
|
13
21
|
- **`session-name`: fix silent auto-naming failure on GitHub Copilot business/enterprise accounts.** Those credentials pin requests to an account-specific endpoint (`auth.baseUrl` from `getApiKeyAndHeaders`); the naming call ignored it and hit the catalog's individual endpoint, failing with `421 Misdirected Request` and leaving sessions unnamed. The naming request now mirrors pi's own request path and prefers the credential's endpoint.
|
package/README.md
CHANGED
|
@@ -62,7 +62,7 @@ A 300 KB changelog page never touches your context window - you get a preview an
|
|
|
62
62
|
|
|
63
63
|
| Extension | Tool | What it does |
|
|
64
64
|
| --- | --- | --- |
|
|
65
|
-
| `extensions/fetch.ts` | `fetch` | Retrieve URLs over HTTP(S). HTML -> Markdown (Readability extraction, Turndown conversion). Binary saved untouched to a temp file. GitHub issue/PR/repo/actions-run URLs auto-route through `gh` (falls back to HTTP). Same size gate as `doc_to_md`. |
|
|
65
|
+
| `extensions/fetch.ts` | `fetch` | Retrieve URLs over HTTP(S). HTML -> Markdown (Readability extraction, Turndown conversion). Binary saved untouched to a temp file. GitHub issue/PR/repo/actions-run/actions-job URLs auto-route through `gh` (falls back to HTTP); failed runs/jobs include failed-step logs (best-effort, summary-only otherwise). Same size gate as `doc_to_md`. Behavior lives in `lib/fetch-core.ts`; also exposed as the `pi-quiver fetch` CLI (see [Claude Code support](#claude-code-support)). |
|
|
66
66
|
| `extensions/doc_to_md.ts` | `doc_to_md` | Convert a local PDF/DOCX/PPTX to Markdown. High-fidelity via `pymupdf4llm` (run through `uv`); degraded pure-JS fallback (`unpdf`) when `uv`/Python is unavailable or conversion times out. DOCX/PPTX convert via LibreOffice first. |
|
|
67
67
|
| `extensions/session-name.ts` | `/session-name` | Manual + opt-in automatic session naming, naming rules and deny list, long-session revisits, and Ghostty tab rename. OFF by default. |
|
|
68
68
|
| `extensions/sword-header.ts` | `/builtin-header` | Themed ASCII startup header replacing pi's default logo. OFF by default. |
|
|
@@ -76,7 +76,7 @@ Full routing rules, size-gate mechanics, and config: [doc/fetch.md](doc/fetch.md
|
|
|
76
76
|
| Concept | Meaning |
|
|
77
77
|
| --- | --- |
|
|
78
78
|
| Size gate | Text/Markdown/JSON output over 32 KB or 1000 lines spills to a temp file with a 60-line preview instead of inlining. |
|
|
79
|
-
| Content routing | HTML -> Markdown, binary -> untouched file, GitHub URLs -> `gh` CLI, everything else -> the size gate. |
|
|
79
|
+
| Content routing | HTML -> Markdown, binary -> untouched file, GitHub URLs -> `gh` CLI (failed runs/jobs get failed-step logs appended), everything else -> the size gate. |
|
|
80
80
|
| Graceful degradation | Optional binaries (`gh`, `uv`, LibreOffice) are never hard install-time deps; each has a defined, documented fallback or failure mode. |
|
|
81
81
|
| Opt-in extensions | `session-name`, `sword-header`, `fast-mode`, and `provider-stall-watchdog` do nothing until explicitly enabled in `settings.json`. |
|
|
82
82
|
| Provider stall recovery | The watchdog detects a missing first stream event and missing parsed semantic progress, not network liveness. The pre-first-event tier covers every mode and origin; the mid-stream tier is TUI-only. |
|
|
@@ -129,7 +129,7 @@ The npm package's bundled JS deps install automatically on `pi install`. A few *
|
|
|
129
129
|
|
|
130
130
|
| Prerequisite | Needed by | If absent |
|
|
131
131
|
| --- | --- | --- |
|
|
132
|
-
| `gh` (GitHub CLI, installed + `gh auth login`) | `fetch` GitHub issue/PR/repo/actions-run routing | Falls back to an HTTP fetch of the rendered page (private repos hit a login wall). |
|
|
132
|
+
| `gh` (GitHub CLI, installed + `gh auth login`) | `fetch` GitHub issue/PR/repo/actions-run/actions-job routing | Falls back to an HTTP fetch of the rendered page (private repos hit a login wall). |
|
|
133
133
|
| `uv` (+ managed Python 3.14, fetched on first use) | `doc_to_md` high-fidelity PDF conversion | Degrades to the pure-JS `unpdf` fallback (no faithful tables/headings). |
|
|
134
134
|
| LibreOffice (`soffice` on `PATH`) | `doc_to_md` DOCX/PPTX conversion | Office inputs error (no JS fallback for office->PDF); PDFs unaffected. |
|
|
135
135
|
|
|
@@ -205,6 +205,29 @@ Operational notes:
|
|
|
205
205
|
- **A watchdog abort that the provider ignores escalates after a fixed 10s.** Any post-abort stream event re-arms that deadline (bytes prove only that the connection was alive at that instant), so a stream that emits a straggler and then wedges still escalates 10s after its last event. This reduces the hang; it cannot force the provider to stop, and undici's timeouts remain the final backstop.
|
|
206
206
|
- **Headless runs report on stderr.** In `print`/`json` mode pi binds a no-op UI, so watchdog notices go out via `console.warn`. Nothing is ever written to stdout, which `json` mode uses for its protocol. In TUI and RPC the notices render as main-window notifications, not the bottom status line.
|
|
207
207
|
|
|
208
|
+
## Claude Code support
|
|
209
|
+
|
|
210
|
+
`fetch`'s core (`lib/fetch-core.ts`) is also published as a CLI, so Claude Code can use the same routing, size gate, and spill behavior as pi's native tool - without pi ever seeing Claude-only files.
|
|
211
|
+
|
|
212
|
+
**Exposed:** the `quiver` plugin, served from this repo's `.claude-plugin/marketplace.json`, with one skill: `fetch` (invoked as `quiver:fetch` / `/quiver:fetch`). The skill runs `npx -y pi-quiver@latest fetch <url> [flags]` via Bash - full parameter parity with the pi tool (`--method`, `--header`, `--body`, `--raw`, `--timeout-ms`), same GitHub `gh` routing (including failed-step logs on failed runs/jobs), same size gate, same binary-to-temp-file handling. See [doc/fetch.md](doc/fetch.md#claude-code-cli-pi-quiver-fetch) for exit codes and flags.
|
|
213
|
+
|
|
214
|
+
**Not exposed:** pi extensions, `doc_to_md`, and everything else in this package - the marketplace allowlists only `./skills/fetch`, and the npm tarball never ships `skills/` or `.claude-plugin/` (pi's own `files` allowlist excludes them, and pi's explicit `pi.extensions` manifest makes them invisible to pi's convention-directory auto-discovery either way).
|
|
215
|
+
|
|
216
|
+
Add the marketplace and enable the plugin in `.claude/settings.json`:
|
|
217
|
+
|
|
218
|
+
```json
|
|
219
|
+
{
|
|
220
|
+
"extraKnownMarketplaces": {
|
|
221
|
+
"pi-quiver": { "source": { "source": "github", "repo": "jjuraszek/pi-quiver" } }
|
|
222
|
+
},
|
|
223
|
+
"enabledPlugins": { "quiver@pi-quiver": true }
|
|
224
|
+
}
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
Activates on folder trust.
|
|
228
|
+
|
|
229
|
+
**Release sequencing:** the skill goes live only with (or after) the npm release that ships the `pi-quiver` bin - until that tag is on npm, `npx -y pi-quiver@latest fetch` resolves a bin-less package and fails.
|
|
230
|
+
|
|
208
231
|
## Development
|
|
209
232
|
|
|
210
233
|
Deps are peers (`@earendil-works/*`, `@sinclair/typebox`) plus the bundled
|
|
@@ -222,6 +245,10 @@ Both run in CI on ubuntu + windows (`.github/workflows/test.yml`).
|
|
|
222
245
|
|
|
223
246
|
pi-quiver is how ground truth gets into an agent's context - real pages, PDFs, docs, cleanly and safely. The other three then coordinate work over it ([pi-cohort](https://github.com/jjuraszek/pi-cohort)), prune it once it's stale ([pi-condense](https://github.com/jjuraszek/pi-condense)), and govern the process end to end ([pi-gauntlet](https://github.com/jjuraszek/pi-gauntlet)).
|
|
224
247
|
|
|
248
|
+
## Contributing
|
|
249
|
+
|
|
250
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md) - issues follow a Context / Problem / Idea / Acceptance Criteria template; PRs run the [pi-gauntlet](https://github.com/jjuraszek/pi-gauntlet) workflow (one-liners exempt from ceremony, never from keeping docs truthful).
|
|
251
|
+
|
|
225
252
|
## Support
|
|
226
253
|
|
|
227
254
|
If this saves you time, consider [buying me a coffee](https://buymeacoffee.com/jjurasszek).
|
|
@@ -0,0 +1,604 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
// bin/pi-quiver.ts
|
|
4
|
+
import { realpathSync } from "node:fs";
|
|
5
|
+
import { fileURLToPath } from "node:url";
|
|
6
|
+
import { resolve } from "node:path";
|
|
7
|
+
|
|
8
|
+
// lib/fetch-core.ts
|
|
9
|
+
import { mkdirSync, writeFileSync, createWriteStream } from "node:fs";
|
|
10
|
+
import { rm } from "node:fs/promises";
|
|
11
|
+
import { tmpdir } from "node:os";
|
|
12
|
+
import { join } from "node:path";
|
|
13
|
+
import { createHash } from "node:crypto";
|
|
14
|
+
import { execFile } from "node:child_process";
|
|
15
|
+
import { promisify } from "node:util";
|
|
16
|
+
import { JSDOM } from "jsdom";
|
|
17
|
+
import { Readability } from "@mozilla/readability";
|
|
18
|
+
import TurndownService from "turndown";
|
|
19
|
+
import { gfm } from "turndown-plugin-gfm";
|
|
20
|
+
function formatSize(bytes) {
|
|
21
|
+
if (bytes < 1024) {
|
|
22
|
+
return `${bytes}B`;
|
|
23
|
+
} else if (bytes < 1024 * 1024) {
|
|
24
|
+
return `${(bytes / 1024).toFixed(1)}KB`;
|
|
25
|
+
} else {
|
|
26
|
+
return `${(bytes / (1024 * 1024)).toFixed(1)}MB`;
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
var RESERVED_OWNERS = /* @__PURE__ */ new Set([
|
|
30
|
+
"orgs",
|
|
31
|
+
"users",
|
|
32
|
+
"sponsors",
|
|
33
|
+
"topics",
|
|
34
|
+
"marketplace",
|
|
35
|
+
"apps",
|
|
36
|
+
"collections",
|
|
37
|
+
"stars",
|
|
38
|
+
"settings",
|
|
39
|
+
"notifications",
|
|
40
|
+
"codespaces",
|
|
41
|
+
"features",
|
|
42
|
+
"trending",
|
|
43
|
+
"security",
|
|
44
|
+
"customer-stories"
|
|
45
|
+
]);
|
|
46
|
+
var GH_NAME = /^[A-Za-z0-9._-]+$/;
|
|
47
|
+
function classifyGitHubTarget(url) {
|
|
48
|
+
const host = url.hostname.toLowerCase();
|
|
49
|
+
if (host !== "github.com" && host !== "www.github.com") return null;
|
|
50
|
+
const segs = url.pathname.split("/").filter((s) => s.length > 0);
|
|
51
|
+
if (segs.length < 2) return null;
|
|
52
|
+
const [owner, repo] = segs;
|
|
53
|
+
if (!GH_NAME.test(owner) || !GH_NAME.test(repo)) return null;
|
|
54
|
+
if (RESERVED_OWNERS.has(owner.toLowerCase())) return null;
|
|
55
|
+
if (segs.length === 4 && segs[2] === "issues" && /^\d+$/.test(segs[3])) {
|
|
56
|
+
return { kind: "issue", url: `https://github.com/${owner}/${repo}/issues/${segs[3]}` };
|
|
57
|
+
}
|
|
58
|
+
if (segs.length === 4 && segs[2] === "pull" && /^\d+$/.test(segs[3])) {
|
|
59
|
+
return { kind: "pr", url: `https://github.com/${owner}/${repo}/pull/${segs[3]}` };
|
|
60
|
+
}
|
|
61
|
+
if (segs.length === 5 && segs[2] === "actions" && segs[3] === "runs" && /^\d+$/.test(segs[4])) {
|
|
62
|
+
return { kind: "run", slug: `${owner}/${repo}`, runId: segs[4], url: `https://github.com/${owner}/${repo}/actions/runs/${segs[4]}` };
|
|
63
|
+
}
|
|
64
|
+
if (segs.length === 7 && segs[2] === "actions" && segs[3] === "runs" && /^\d+$/.test(segs[4]) && segs[5] === "job" && /^\d+$/.test(segs[6])) {
|
|
65
|
+
return { kind: "job", slug: `${owner}/${repo}`, jobId: segs[6], url: `https://github.com/${owner}/${repo}/actions/runs/${segs[4]}/job/${segs[6]}` };
|
|
66
|
+
}
|
|
67
|
+
if (segs.length === 2) {
|
|
68
|
+
return { kind: "repo", slug: `${owner}/${repo}` };
|
|
69
|
+
}
|
|
70
|
+
return null;
|
|
71
|
+
}
|
|
72
|
+
function buildGhArgs(target) {
|
|
73
|
+
if (target.kind === "issue") return ["issue", "view", target.url, "--comments"];
|
|
74
|
+
if (target.kind === "pr") return ["pr", "view", target.url, "--comments"];
|
|
75
|
+
if (target.kind === "run") return ["run", "view", target.runId, "--repo", target.slug];
|
|
76
|
+
if (target.kind === "job") return ["run", "view", "--job", target.jobId, "--repo", target.slug];
|
|
77
|
+
return ["repo", "view", target.slug];
|
|
78
|
+
}
|
|
79
|
+
function buildGhLogArgs(target) {
|
|
80
|
+
if (target.kind === "run") return ["run", "view", target.runId, "--log-failed", "--repo", target.slug];
|
|
81
|
+
if (target.kind === "job") return ["run", "view", "--job", target.jobId, "--log-failed", "--repo", target.slug];
|
|
82
|
+
return null;
|
|
83
|
+
}
|
|
84
|
+
var GH_MAX_BUFFER = 1e7;
|
|
85
|
+
var execFileAsync = promisify(execFile);
|
|
86
|
+
var runGh = async (args, timeoutMs, signal) => {
|
|
87
|
+
try {
|
|
88
|
+
const { stdout } = await execFileAsync("gh", args, {
|
|
89
|
+
timeout: timeoutMs,
|
|
90
|
+
signal,
|
|
91
|
+
maxBuffer: GH_MAX_BUFFER,
|
|
92
|
+
encoding: "utf8"
|
|
93
|
+
});
|
|
94
|
+
if (!stdout.trim()) return { ok: false };
|
|
95
|
+
return { ok: true, stdout };
|
|
96
|
+
} catch {
|
|
97
|
+
return { ok: false };
|
|
98
|
+
}
|
|
99
|
+
};
|
|
100
|
+
function planGhRouting(params, url) {
|
|
101
|
+
if (params.raw) return null;
|
|
102
|
+
if ((params.method ?? "GET") !== "GET") return null;
|
|
103
|
+
if (params.body) return null;
|
|
104
|
+
if (params.headers && Object.keys(params.headers).length > 0) return null;
|
|
105
|
+
return classifyGitHubTarget(url);
|
|
106
|
+
}
|
|
107
|
+
function ghCommandLabel(target) {
|
|
108
|
+
if (target.kind === "issue") return "issue view --comments";
|
|
109
|
+
if (target.kind === "pr") return "pr view --comments";
|
|
110
|
+
if (target.kind === "run") return "run view";
|
|
111
|
+
if (target.kind === "job") return "run view --job";
|
|
112
|
+
return "repo view";
|
|
113
|
+
}
|
|
114
|
+
function ghSourceLine(target, ref) {
|
|
115
|
+
if (target.kind === "issue") return `gh issue view ${ref} --comments`;
|
|
116
|
+
if (target.kind === "pr") return `gh pr view ${ref} --comments`;
|
|
117
|
+
if (target.kind === "run") return `gh run view ${target.runId} --repo ${target.slug}`;
|
|
118
|
+
if (target.kind === "job") return `gh run view --job ${target.jobId} --repo ${target.slug}`;
|
|
119
|
+
return `gh repo view ${ref}`;
|
|
120
|
+
}
|
|
121
|
+
function renderGhResult(target, stdout, failedLogs) {
|
|
122
|
+
const body = failedLogs !== void 0 ? `${stdout.trimEnd()}
|
|
123
|
+
|
|
124
|
+
## Failed step logs
|
|
125
|
+
|
|
126
|
+
${failedLogs.trimEnd()}` : stdout.trimEnd();
|
|
127
|
+
const ref = target.kind === "repo" ? target.slug : target.url;
|
|
128
|
+
const { spill, bytes, lines } = applyGate(body);
|
|
129
|
+
const baseDetails = {
|
|
130
|
+
url: ref,
|
|
131
|
+
bytes,
|
|
132
|
+
lines,
|
|
133
|
+
category: "markdown",
|
|
134
|
+
via: "gh",
|
|
135
|
+
ghCommand: ghCommandLabel(target)
|
|
136
|
+
};
|
|
137
|
+
const source = failedLogs !== void 0 ? `Source: ${ghSourceLine(target, ref)}
|
|
138
|
+
Source: gh ${buildGhLogArgs(target).join(" ")}` : `Source: ${ghSourceLine(target, ref)}`;
|
|
139
|
+
if (!spill) {
|
|
140
|
+
return {
|
|
141
|
+
output: [source, "", body].join("\n"),
|
|
142
|
+
details: { ...baseDetails, spilled: false }
|
|
143
|
+
};
|
|
144
|
+
}
|
|
145
|
+
const spillUrl = target.kind === "repo" ? `https://github.com/${target.slug}` : target.url;
|
|
146
|
+
const file = spillToFile(spillUrl, body, "md");
|
|
147
|
+
return {
|
|
148
|
+
output: [
|
|
149
|
+
source,
|
|
150
|
+
`Body: ${formatSize(bytes)} across ${lines} lines \u2014 written to file (too large to inline)`,
|
|
151
|
+
`Saved-To: ${file}`,
|
|
152
|
+
"",
|
|
153
|
+
"Read slices of this file with the read tool (offset/limit) or grep it; do not read the whole file unless you must. Markdown is grep-able by heading (^#).",
|
|
154
|
+
"",
|
|
155
|
+
`----- preview (first ${PREVIEW_LINES} lines) -----`,
|
|
156
|
+
buildPreview(body)
|
|
157
|
+
].join("\n"),
|
|
158
|
+
details: { ...baseDetails, spilled: true, file }
|
|
159
|
+
};
|
|
160
|
+
}
|
|
161
|
+
async function executeGhRouting(params, url, signal, runner = runGh) {
|
|
162
|
+
const target = planGhRouting(params, url);
|
|
163
|
+
if (!target) return null;
|
|
164
|
+
const timeoutMs = params.timeoutMs ?? DEFAULT_TIMEOUT_MS;
|
|
165
|
+
signal?.throwIfAborted();
|
|
166
|
+
const gh = await runner(buildGhArgs(target), timeoutMs, signal);
|
|
167
|
+
signal?.throwIfAborted();
|
|
168
|
+
if (!gh.ok) return null;
|
|
169
|
+
const logArgs = buildGhLogArgs(target);
|
|
170
|
+
let failedLogs;
|
|
171
|
+
if (logArgs) {
|
|
172
|
+
const logs = await runner(logArgs, timeoutMs, signal);
|
|
173
|
+
signal?.throwIfAborted();
|
|
174
|
+
if (logs.ok && logs.stdout !== "") failedLogs = logs.stdout;
|
|
175
|
+
}
|
|
176
|
+
return renderGhResult(target, gh.stdout, failedLogs);
|
|
177
|
+
}
|
|
178
|
+
var PARSABLE_MAX_BYTES = 1e6;
|
|
179
|
+
var BINARY_MAX_BYTES = 5e7;
|
|
180
|
+
var SNIFF_MAX_BYTES = 64e3;
|
|
181
|
+
var DEFAULT_TIMEOUT_MS = 2e4;
|
|
182
|
+
var INLINE_MAX_BYTES = 32e3;
|
|
183
|
+
var INLINE_MAX_LINES = 1e3;
|
|
184
|
+
var PREVIEW_LINES = 60;
|
|
185
|
+
var PREVIEW_MAX_BYTES = 4e3;
|
|
186
|
+
var FIREFOX_UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 14.7; rv:135.0) Gecko/20100101 Firefox/135.0";
|
|
187
|
+
var DEFAULT_ACCEPT = "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8";
|
|
188
|
+
function parseCharset(contentType) {
|
|
189
|
+
const m = /charset\s*=\s*"?([^";\s]+)"?/i.exec(contentType);
|
|
190
|
+
return (m?.[1] ?? "utf-8").trim().toLowerCase();
|
|
191
|
+
}
|
|
192
|
+
function decodeBuffer(buf, charset) {
|
|
193
|
+
try {
|
|
194
|
+
return new TextDecoder(charset, { fatal: false }).decode(buf);
|
|
195
|
+
} catch {
|
|
196
|
+
return new TextDecoder("utf-8", { fatal: false }).decode(buf);
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
function buildPreview(body) {
|
|
200
|
+
let preview = body.split("\n").slice(0, PREVIEW_LINES).join("\n");
|
|
201
|
+
if (preview.length > PREVIEW_MAX_BYTES) {
|
|
202
|
+
preview = `${preview.slice(0, PREVIEW_MAX_BYTES)}
|
|
203
|
+
\u2026[preview truncated]`;
|
|
204
|
+
}
|
|
205
|
+
return preview;
|
|
206
|
+
}
|
|
207
|
+
var turndownService = new TurndownService({
|
|
208
|
+
headingStyle: "atx",
|
|
209
|
+
codeBlockStyle: "fenced",
|
|
210
|
+
bulletListMarker: "-"
|
|
211
|
+
});
|
|
212
|
+
turndownService.use(gfm);
|
|
213
|
+
function mimeType(contentType) {
|
|
214
|
+
return contentType.split(";")[0].trim().toLowerCase();
|
|
215
|
+
}
|
|
216
|
+
var TEXT_ALLOWLIST = [
|
|
217
|
+
/^text\//,
|
|
218
|
+
/^application\/(json|xml|xhtml\+xml|javascript)$/,
|
|
219
|
+
/\+json$/,
|
|
220
|
+
/\+xml$/
|
|
221
|
+
];
|
|
222
|
+
var KNOWN_BINARY = [
|
|
223
|
+
/^audio\//,
|
|
224
|
+
/^video\//,
|
|
225
|
+
/^font\//,
|
|
226
|
+
/^application\/(pdf|zip|gzip|x-tar|x-7z-compressed|x-rar-compressed|wasm)$/
|
|
227
|
+
];
|
|
228
|
+
function categorize(contentType, sniff, raw) {
|
|
229
|
+
const mime = mimeType(contentType);
|
|
230
|
+
if (/^image\//.test(mime)) return "binary";
|
|
231
|
+
const isText = TEXT_ALLOWLIST.some((re) => re.test(mime));
|
|
232
|
+
const isBinary = KNOWN_BINARY.some((re) => re.test(mime));
|
|
233
|
+
if (!isText && !isBinary) return sniff.includes(0) ? "binary" : "text";
|
|
234
|
+
if (isBinary && !isText) return "binary";
|
|
235
|
+
if (sniff.includes(0)) return "binary";
|
|
236
|
+
if (raw) return "text";
|
|
237
|
+
if (mime === "text/html" || mime === "application/xhtml+xml") return "markdown";
|
|
238
|
+
if (mime === "application/json" || /\+json$/.test(mime)) return "json";
|
|
239
|
+
return "text";
|
|
240
|
+
}
|
|
241
|
+
function htmlToMarkdown(html, url) {
|
|
242
|
+
let doc;
|
|
243
|
+
try {
|
|
244
|
+
doc = new JSDOM(html, { url }).window.document;
|
|
245
|
+
} catch {
|
|
246
|
+
return null;
|
|
247
|
+
}
|
|
248
|
+
let article = null;
|
|
249
|
+
try {
|
|
250
|
+
article = new Readability(doc).parse();
|
|
251
|
+
} catch {
|
|
252
|
+
return null;
|
|
253
|
+
}
|
|
254
|
+
if (!article?.content) return null;
|
|
255
|
+
let md;
|
|
256
|
+
try {
|
|
257
|
+
md = turndownService.turndown(article.content).trim();
|
|
258
|
+
} catch {
|
|
259
|
+
return null;
|
|
260
|
+
}
|
|
261
|
+
if (!md) return null;
|
|
262
|
+
if (article.title) md = `# ${article.title}
|
|
263
|
+
|
|
264
|
+
${md}`;
|
|
265
|
+
return md;
|
|
266
|
+
}
|
|
267
|
+
function prettyJson(text) {
|
|
268
|
+
try {
|
|
269
|
+
return JSON.stringify(JSON.parse(text), null, 2);
|
|
270
|
+
} catch {
|
|
271
|
+
return text;
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
function applyGate(body) {
|
|
275
|
+
const bytes = Buffer.byteLength(body, "utf8");
|
|
276
|
+
const lines = body.length ? body.split("\n").length : 0;
|
|
277
|
+
const spill = body.length > 0 && (bytes > INLINE_MAX_BYTES || lines > INLINE_MAX_LINES);
|
|
278
|
+
return { spill, bytes, lines };
|
|
279
|
+
}
|
|
280
|
+
function tempFilePath(url, ext) {
|
|
281
|
+
const dir = join(tmpdir(), "pi-fetch");
|
|
282
|
+
mkdirSync(dir, { recursive: true });
|
|
283
|
+
let host = "page";
|
|
284
|
+
try {
|
|
285
|
+
host = new URL(url).hostname.replace(/[^a-z0-9.-]/gi, "_") || "page";
|
|
286
|
+
} catch {
|
|
287
|
+
}
|
|
288
|
+
const hash = createHash("sha1").update(url).digest("hex").slice(0, 8);
|
|
289
|
+
const stamp = (/* @__PURE__ */ new Date()).toISOString().replace(/[:.]/g, "-");
|
|
290
|
+
return join(dir, `${stamp}-${host}-${hash}.${ext}`);
|
|
291
|
+
}
|
|
292
|
+
function spillToFile(url, body, ext) {
|
|
293
|
+
const file = tempFilePath(url, ext);
|
|
294
|
+
writeFileSync(file, body, "utf8");
|
|
295
|
+
return file;
|
|
296
|
+
}
|
|
297
|
+
function textExtension(category, contentType) {
|
|
298
|
+
if (category === "markdown") return "md";
|
|
299
|
+
if (category === "json") return "json";
|
|
300
|
+
return mimeType(contentType).includes("xml") ? "xml" : "txt";
|
|
301
|
+
}
|
|
302
|
+
var BINARY_EXT = {
|
|
303
|
+
"application/pdf": "pdf",
|
|
304
|
+
"application/zip": "zip",
|
|
305
|
+
"application/vnd.openxmlformats-officedocument.wordprocessingml.document": "docx",
|
|
306
|
+
"application/vnd.openxmlformats-officedocument.presentationml.presentation": "pptx",
|
|
307
|
+
"application/gzip": "gz",
|
|
308
|
+
"image/png": "png",
|
|
309
|
+
"image/jpeg": "jpg",
|
|
310
|
+
"image/gif": "gif",
|
|
311
|
+
"image/webp": "webp",
|
|
312
|
+
"image/svg+xml": "svg"
|
|
313
|
+
};
|
|
314
|
+
function binaryExtension(contentType) {
|
|
315
|
+
const mime = mimeType(contentType);
|
|
316
|
+
if (BINARY_EXT[mime]) return BINARY_EXT[mime];
|
|
317
|
+
const sub = (mime.split("/")[1] ?? "").replace(/^x-/, "").replace(/[^a-z0-9]+/g, "").slice(0, 8);
|
|
318
|
+
return sub || "bin";
|
|
319
|
+
}
|
|
320
|
+
function writeChunk(stream, b) {
|
|
321
|
+
return new Promise((resolve2, reject) => {
|
|
322
|
+
stream.write(b, (err) => err ? reject(err) : resolve2());
|
|
323
|
+
});
|
|
324
|
+
}
|
|
325
|
+
async function pumpToFile(stream, reader, prefix, exhausted) {
|
|
326
|
+
let bytes = 0;
|
|
327
|
+
let truncated = false;
|
|
328
|
+
let head = prefix;
|
|
329
|
+
if (head.length > BINARY_MAX_BYTES) {
|
|
330
|
+
head = head.subarray(0, BINARY_MAX_BYTES);
|
|
331
|
+
truncated = true;
|
|
332
|
+
}
|
|
333
|
+
await writeChunk(stream, head);
|
|
334
|
+
bytes += head.length;
|
|
335
|
+
while (!exhausted && !truncated) {
|
|
336
|
+
const { done, value } = await reader.read();
|
|
337
|
+
if (done) break;
|
|
338
|
+
let chunk = Buffer.from(value);
|
|
339
|
+
if (bytes + chunk.length > BINARY_MAX_BYTES) {
|
|
340
|
+
chunk = chunk.subarray(0, BINARY_MAX_BYTES - bytes);
|
|
341
|
+
truncated = true;
|
|
342
|
+
}
|
|
343
|
+
await writeChunk(stream, chunk);
|
|
344
|
+
bytes += chunk.length;
|
|
345
|
+
}
|
|
346
|
+
await new Promise((resolve2, reject) => stream.end((err) => err ? reject(err) : resolve2()));
|
|
347
|
+
return { bytes, truncated };
|
|
348
|
+
}
|
|
349
|
+
async function collectBody(res, contentType, raw) {
|
|
350
|
+
const reader = res.body.getReader();
|
|
351
|
+
const prefixParts = [];
|
|
352
|
+
let prefixLen = 0;
|
|
353
|
+
let exhausted = false;
|
|
354
|
+
while (prefixLen < SNIFF_MAX_BYTES) {
|
|
355
|
+
const { done, value } = await reader.read();
|
|
356
|
+
if (done) {
|
|
357
|
+
exhausted = true;
|
|
358
|
+
break;
|
|
359
|
+
}
|
|
360
|
+
const chunk = Buffer.from(value);
|
|
361
|
+
prefixParts.push(chunk);
|
|
362
|
+
prefixLen += chunk.length;
|
|
363
|
+
}
|
|
364
|
+
const prefix = Buffer.concat(prefixParts);
|
|
365
|
+
const category = categorize(contentType, prefix.subarray(0, SNIFF_MAX_BYTES), raw);
|
|
366
|
+
if (category === "binary") {
|
|
367
|
+
const file = tempFilePath(res.url, binaryExtension(contentType));
|
|
368
|
+
const stream = createWriteStream(file);
|
|
369
|
+
try {
|
|
370
|
+
const { bytes: bytes2, truncated: truncated2 } = await pumpToFile(stream, reader, prefix, exhausted);
|
|
371
|
+
if (truncated2) await reader.cancel().catch(() => {
|
|
372
|
+
});
|
|
373
|
+
return { category, file, bytes: bytes2, truncated: truncated2 };
|
|
374
|
+
} catch (err) {
|
|
375
|
+
stream.destroy();
|
|
376
|
+
await rm(file, { force: true });
|
|
377
|
+
await reader.cancel().catch(() => {
|
|
378
|
+
});
|
|
379
|
+
throw err;
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
const parts = [prefix];
|
|
383
|
+
let bytes = prefix.length;
|
|
384
|
+
let streamDone = exhausted;
|
|
385
|
+
while (!streamDone && bytes < PARSABLE_MAX_BYTES) {
|
|
386
|
+
const { done, value } = await reader.read();
|
|
387
|
+
if (done) {
|
|
388
|
+
streamDone = true;
|
|
389
|
+
break;
|
|
390
|
+
}
|
|
391
|
+
const chunk = Buffer.from(value);
|
|
392
|
+
parts.push(chunk);
|
|
393
|
+
bytes += chunk.length;
|
|
394
|
+
}
|
|
395
|
+
let buffer = Buffer.concat(parts);
|
|
396
|
+
if (buffer.length > PARSABLE_MAX_BYTES) {
|
|
397
|
+
buffer = buffer.subarray(0, PARSABLE_MAX_BYTES);
|
|
398
|
+
}
|
|
399
|
+
const truncated = !streamDone;
|
|
400
|
+
if (truncated) await reader.cancel().catch(() => {
|
|
401
|
+
});
|
|
402
|
+
return { category, buffer, bytes: buffer.length, truncated };
|
|
403
|
+
}
|
|
404
|
+
async function fetchUrl(opts) {
|
|
405
|
+
const url = new URL(opts.url);
|
|
406
|
+
if (url.protocol !== "http:" && url.protocol !== "https:") {
|
|
407
|
+
throw new Error(`Unsupported protocol: ${url.protocol}`);
|
|
408
|
+
}
|
|
409
|
+
const ghResult = await executeGhRouting(opts, url, opts.signal);
|
|
410
|
+
if (ghResult) return ghResult;
|
|
411
|
+
const headers = new Headers(opts.headers ?? {});
|
|
412
|
+
if (!headers.has("user-agent")) headers.set("user-agent", FIREFOX_UA);
|
|
413
|
+
if (!headers.has("accept")) headers.set("accept", DEFAULT_ACCEPT);
|
|
414
|
+
if (!headers.has("accept-language"))
|
|
415
|
+
headers.set("accept-language", "en-US,en;q=0.5");
|
|
416
|
+
const controller = new AbortController();
|
|
417
|
+
const onAbort = () => controller.abort();
|
|
418
|
+
opts.signal?.addEventListener("abort", onAbort);
|
|
419
|
+
const timer = setTimeout(
|
|
420
|
+
() => controller.abort(new Error("fetch timeout")),
|
|
421
|
+
opts.timeoutMs ?? DEFAULT_TIMEOUT_MS
|
|
422
|
+
);
|
|
423
|
+
try {
|
|
424
|
+
const res = await fetch(url, {
|
|
425
|
+
method: opts.method ?? "GET",
|
|
426
|
+
headers,
|
|
427
|
+
body: opts.body,
|
|
428
|
+
signal: controller.signal,
|
|
429
|
+
redirect: "follow"
|
|
430
|
+
});
|
|
431
|
+
const ct = res.headers.get("content-type") ?? "";
|
|
432
|
+
const charset = parseCharset(ct);
|
|
433
|
+
const header = [
|
|
434
|
+
`HTTP ${res.status} ${res.statusText}`,
|
|
435
|
+
`Content-Type: ${ct}`,
|
|
436
|
+
`Charset: ${charset}`
|
|
437
|
+
];
|
|
438
|
+
if (!res.body || (opts.method ?? "GET") === "HEAD") {
|
|
439
|
+
return {
|
|
440
|
+
output: [...header, "Length: 0 (no body)"].join("\n"),
|
|
441
|
+
details: { url: res.url, status: res.status, contentType: ct, charset, bytes: 0 }
|
|
442
|
+
};
|
|
443
|
+
}
|
|
444
|
+
const collected = await collectBody(res, ct, opts.raw ?? false);
|
|
445
|
+
const baseDetails = {
|
|
446
|
+
url: res.url,
|
|
447
|
+
status: res.status,
|
|
448
|
+
contentType: ct,
|
|
449
|
+
charset,
|
|
450
|
+
bytes: collected.bytes,
|
|
451
|
+
truncated: collected.truncated,
|
|
452
|
+
category: collected.category
|
|
453
|
+
};
|
|
454
|
+
if (collected.category === "binary") {
|
|
455
|
+
const note = collected.truncated ? " (truncated to 50MB)" : "";
|
|
456
|
+
return {
|
|
457
|
+
output: [
|
|
458
|
+
...header,
|
|
459
|
+
`Body: ${formatSize(collected.bytes)}${note} binary (${mimeType(ct) || "unknown"}) \u2014 saved untouched for processing`,
|
|
460
|
+
`Saved-To: ${collected.file}`,
|
|
461
|
+
"",
|
|
462
|
+
"Binary content is not decoded. Use the appropriate tool to process the file at the path above."
|
|
463
|
+
].join("\n"),
|
|
464
|
+
details: { ...baseDetails, spilled: true, file: collected.file }
|
|
465
|
+
};
|
|
466
|
+
}
|
|
467
|
+
const decoded = decodeBuffer(collected.buffer, charset);
|
|
468
|
+
let body;
|
|
469
|
+
let effectiveCategory = collected.category;
|
|
470
|
+
if (collected.category === "markdown") {
|
|
471
|
+
const md = htmlToMarkdown(decoded, res.url);
|
|
472
|
+
if (md !== null) {
|
|
473
|
+
body = md;
|
|
474
|
+
} else {
|
|
475
|
+
body = decoded;
|
|
476
|
+
effectiveCategory = "text";
|
|
477
|
+
}
|
|
478
|
+
} else if (collected.category === "json") {
|
|
479
|
+
body = prettyJson(decoded);
|
|
480
|
+
} else {
|
|
481
|
+
body = decoded;
|
|
482
|
+
}
|
|
483
|
+
baseDetails.category = effectiveCategory;
|
|
484
|
+
const truncNote = collected.truncated ? "\n[Note: source truncated at 1MB \u2014 content may be partial]" : "";
|
|
485
|
+
const lengthLine = `Length: ${collected.bytes}${collected.truncated ? " (truncated to 1MB)" : ""}`;
|
|
486
|
+
const { spill, bytes: bodyBytes, lines: lineCount } = applyGate(body);
|
|
487
|
+
baseDetails.lines = lineCount;
|
|
488
|
+
if (!spill) {
|
|
489
|
+
return {
|
|
490
|
+
output: [...header, lengthLine, "", body + truncNote].join("\n"),
|
|
491
|
+
details: { ...baseDetails, spilled: false }
|
|
492
|
+
};
|
|
493
|
+
}
|
|
494
|
+
const ext = textExtension(effectiveCategory, ct);
|
|
495
|
+
const file = spillToFile(res.url, body, ext);
|
|
496
|
+
const grepHint = effectiveCategory === "markdown" ? "Read slices of this file with the read tool (offset/limit) or grep it; do not read the whole file unless you must. Markdown is grep-able by heading (^#)." : "Read slices of this file with the read tool (offset/limit) or grep it; do not read the whole file unless you must.";
|
|
497
|
+
return {
|
|
498
|
+
output: [
|
|
499
|
+
...header,
|
|
500
|
+
lengthLine,
|
|
501
|
+
`Body: ${formatSize(bodyBytes)} across ${lineCount} lines \u2014 written to file (too large to inline)`,
|
|
502
|
+
`Saved-To: ${file}`,
|
|
503
|
+
...collected.truncated ? ["[Note: source truncated at 1MB \u2014 content may be partial]"] : [],
|
|
504
|
+
"",
|
|
505
|
+
grepHint,
|
|
506
|
+
"",
|
|
507
|
+
`----- preview (first ${PREVIEW_LINES} lines) -----`,
|
|
508
|
+
buildPreview(body)
|
|
509
|
+
].join("\n"),
|
|
510
|
+
details: { ...baseDetails, spilled: true, file }
|
|
511
|
+
};
|
|
512
|
+
} finally {
|
|
513
|
+
clearTimeout(timer);
|
|
514
|
+
opts.signal?.removeEventListener("abort", onAbort);
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
// bin/pi-quiver.ts
|
|
519
|
+
var USAGE = 'Usage: pi-quiver fetch <url> [--method GET|HEAD|POST] [--header "K: V"]... [--body <str>] [--raw] [--timeout-ms <n>]';
|
|
520
|
+
function parseCliArgs(argv) {
|
|
521
|
+
if (argv[0] !== "fetch") return { ok: false, error: `unknown command: ${argv[0] ?? "(none)"}` };
|
|
522
|
+
const rest = argv.slice(1);
|
|
523
|
+
let url;
|
|
524
|
+
let method;
|
|
525
|
+
let headers;
|
|
526
|
+
let body;
|
|
527
|
+
let raw;
|
|
528
|
+
let timeoutMs;
|
|
529
|
+
for (let i = 0; i < rest.length; i++) {
|
|
530
|
+
const arg = rest[i];
|
|
531
|
+
if (arg === "--raw") {
|
|
532
|
+
raw = true;
|
|
533
|
+
continue;
|
|
534
|
+
}
|
|
535
|
+
if (arg === "--method" || arg === "--header" || arg === "--body" || arg === "--timeout-ms") {
|
|
536
|
+
const value = rest[++i];
|
|
537
|
+
if (value === void 0) return { ok: false, error: `${arg} requires a value` };
|
|
538
|
+
if (arg === "--method") {
|
|
539
|
+
if (value !== "GET" && value !== "HEAD" && value !== "POST") {
|
|
540
|
+
return { ok: false, error: `invalid --method: ${value}` };
|
|
541
|
+
}
|
|
542
|
+
method = value;
|
|
543
|
+
} else if (arg === "--header") {
|
|
544
|
+
const sep = value.indexOf(": ");
|
|
545
|
+
if (sep <= 0) return { ok: false, error: `malformed --header (expected "Key: Value"): ${value}` };
|
|
546
|
+
headers ??= {};
|
|
547
|
+
headers[value.slice(0, sep)] = value.slice(sep + 2);
|
|
548
|
+
} else if (arg === "--body") {
|
|
549
|
+
body = value;
|
|
550
|
+
} else {
|
|
551
|
+
const n = Number(value);
|
|
552
|
+
if (!Number.isFinite(n) || n <= 0) return { ok: false, error: `invalid --timeout-ms: ${value}` };
|
|
553
|
+
timeoutMs = n;
|
|
554
|
+
}
|
|
555
|
+
continue;
|
|
556
|
+
}
|
|
557
|
+
if (arg.startsWith("--")) return { ok: false, error: `unknown flag: ${arg}` };
|
|
558
|
+
if (url !== void 0) return { ok: false, error: `unexpected argument: ${arg}` };
|
|
559
|
+
url = arg;
|
|
560
|
+
}
|
|
561
|
+
if (!url) return { ok: false, error: "missing <url>" };
|
|
562
|
+
const opts = { url };
|
|
563
|
+
if (method !== void 0) opts.method = method;
|
|
564
|
+
if (headers !== void 0) opts.headers = headers;
|
|
565
|
+
if (body !== void 0) opts.body = body;
|
|
566
|
+
if (raw !== void 0) opts.raw = raw;
|
|
567
|
+
if (timeoutMs !== void 0) opts.timeoutMs = timeoutMs;
|
|
568
|
+
return { ok: true, opts };
|
|
569
|
+
}
|
|
570
|
+
async function main() {
|
|
571
|
+
const parsed = parseCliArgs(process.argv.slice(2));
|
|
572
|
+
if (!parsed.ok) {
|
|
573
|
+
process.stderr.write(`${parsed.error}
|
|
574
|
+
${USAGE}
|
|
575
|
+
`);
|
|
576
|
+
return 2;
|
|
577
|
+
}
|
|
578
|
+
try {
|
|
579
|
+
const result = await fetchUrl(parsed.opts);
|
|
580
|
+
process.stdout.write(`${result.output}
|
|
581
|
+
`);
|
|
582
|
+
return 0;
|
|
583
|
+
} catch (err) {
|
|
584
|
+
process.stderr.write(`fetch failed: ${err instanceof Error ? err.message : String(err)}
|
|
585
|
+
`);
|
|
586
|
+
return 1;
|
|
587
|
+
}
|
|
588
|
+
}
|
|
589
|
+
function isMainEntry() {
|
|
590
|
+
if (!process.argv[1]) return false;
|
|
591
|
+
try {
|
|
592
|
+
return realpathSync(fileURLToPath(import.meta.url)) === realpathSync(resolve(process.argv[1]));
|
|
593
|
+
} catch {
|
|
594
|
+
return false;
|
|
595
|
+
}
|
|
596
|
+
}
|
|
597
|
+
if (isMainEntry()) {
|
|
598
|
+
main().then((code) => {
|
|
599
|
+
process.exitCode = code;
|
|
600
|
+
});
|
|
601
|
+
}
|
|
602
|
+
export {
|
|
603
|
+
parseCliArgs
|
|
604
|
+
};
|