@haystackeditor/cli 0.17.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +208 -26
- package/dist/commands/case-batch-contract.js +2 -2
- package/dist/commands/case-batch.js +7 -17
- package/dist/commands/crawl-contract.js +3 -0
- package/dist/commands/db-profile-contract.js +419 -0
- package/dist/commands/db-profile-database.js +289 -0
- package/dist/commands/db-profile-distinct.js +97 -0
- package/dist/commands/db-profile-owners.js +94 -0
- package/dist/commands/db-profile-privacy.js +450 -0
- package/dist/commands/db-profile-show.js +218 -0
- package/dist/commands/db-profile-sql-literals.js +140 -0
- package/dist/commands/db-profile-sql.js +362 -0
- package/dist/commands/db-profile-upload.js +88 -0
- package/dist/commands/db-profile.js +1020 -0
- package/dist/commands/fleet-policy-contract.js +28 -0
- package/dist/commands/fleet-policy.js +305 -0
- package/dist/commands/precompute-delivery-worker.js +17 -3
- package/dist/commands/precompute-delivery.js +16 -6
- package/dist/commands/verify-hosted.js +20 -0
- package/dist/commands/verify-precompute.js +140 -95
- package/dist/commands/verify.js +523 -52
- package/dist/index.js +173 -21
- package/dist/schema.js +1 -0
- package/dist/triage/astra.js +5 -2
- package/dist/triage/runner.js +1 -1
- package/dist/utils/verify-base.js +31 -0
- package/package.json +4 -1
- package/schemas/verify.v1.json +217 -0
- package/dist/commands/combination-search-hook-contract.js +0 -1
- package/dist/commands/combination-search-hook.js +0 -139
package/README.md
CHANGED
|
@@ -80,41 +80,83 @@ haystack triage <ref> --json # poll findings later / after --no-wait
|
|
|
80
80
|
`dismiss`, `mark_reviewed`, `undismiss`, `request_review`, `trigger_review`,
|
|
81
81
|
and `schema`.
|
|
82
82
|
|
|
83
|
-
**`haystack verify`**
|
|
84
|
-
|
|
83
|
+
**`haystack verify`** shows the product blast radius of your current change:
|
|
84
|
+
where it shows up in the running app, and what broke. Inside a git checkout
|
|
85
|
+
(nothing needs to be committed or pushed), run:
|
|
85
86
|
|
|
86
87
|
```bash
|
|
87
88
|
haystack verify
|
|
89
|
+
haystack verify --no-wait
|
|
88
90
|
haystack verify --json
|
|
89
91
|
```
|
|
90
92
|
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
origin/<default branch>`,
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
93
|
+
It captures the checkout, committed and uncommitted changes together, exactly
|
|
94
|
+
as the stop hook does: the repository from `origin`, the base at `git
|
|
95
|
+
merge-base HEAD origin/<default branch>`, and the working tree as a
|
|
96
|
+
deterministic snapshot commit, so the same tree always names the same capture.
|
|
97
|
+
It then finds your crawl of that exact capture: the same code and the same
|
|
98
|
+
change title (HEAD's subject, which tells the crawl's judge what the change is;
|
|
99
|
+
committing or rewording keeps the code but changes the title). When there is
|
|
100
|
+
none, or the one it finds was cancelled, stopped before finishing or is being
|
|
101
|
+
cancelled, it submits the capture through the same request `haystack verify
|
|
102
|
+
precompute` sends (the service starts the crawl, or runs the stopped one again
|
|
103
|
+
once it has shut down), says so in one line, and follows the crawl the service
|
|
104
|
+
acknowledged; it never shows a crawl of another revision or another title.
|
|
105
|
+
|
|
106
|
+
A crawl builds the app with and without your change, starts from the changed
|
|
107
|
+
code, reaches it in the running app, explores outward with both builds side by
|
|
108
|
+
side, and double-checks and judges every difference. `haystack verify` waits
|
|
109
|
+
until the crawl finishes, printing each step as it happens; there is no time
|
|
110
|
+
limit, and Ctrl-C stops waiting, never the crawl. It then prints:
|
|
111
|
+
|
|
112
|
+
- a headline: how many bugs it found, none, or why the crawl could not finish;
|
|
113
|
+
- **Where your change shows up in the app**: every changed spot as
|
|
114
|
+
`file:line`, how far the crawl got with it (never reached, loaded but never
|
|
115
|
+
ran, shown on screen, ran, ran when its control was pressed, a difference
|
|
116
|
+
seen, a difference seen and confirmed, judged a bug), and the steps and page
|
|
117
|
+
that reached it;
|
|
118
|
+
- **What the crawl found**: bugs first, each with a one-sentence reason and the
|
|
119
|
+
steps to see it;
|
|
120
|
+
- **What the crawl never ran**: changed files no path ran, and how many of the
|
|
121
|
+
functions your change affects ran.
|
|
122
|
+
|
|
123
|
+
`--no-wait` prints the crawl's current state and returns. `--interval
|
|
124
|
+
<seconds>` sets the polling interval (default 5), `--account <login>` picks a
|
|
125
|
+
saved account, and `--repo owner/repo` names the repository when `origin` does
|
|
126
|
+
not. `--json` prints one document, `{ "schema_version", "crawl" }`, where
|
|
127
|
+
`crawl` is the crawl as the service returns it (or `null` when no crawl of the
|
|
128
|
+
capture could be started); `haystack schema verify` prints its schema. Exit codes
|
|
129
|
+
follow `haystack case-batch status`: 0 when the crawl finished (the bugs it
|
|
130
|
+
found are in the output) or is still running under `--no-wait`; 2 when it ended
|
|
131
|
+
without finishing (stopped early, or cancelled because a newer stop in the
|
|
132
|
+
repository replaced it) or the machines it used could not be proven shut down;
|
|
133
|
+
1 when the command failed or no crawl could be started.
|
|
134
|
+
|
|
135
|
+
**The stop hook.** `haystack hooks install-session --cli claude` installs a
|
|
136
|
+
Claude Code Stop hook that runs `haystack verify precompute --hook` whenever an
|
|
137
|
+
agent turn stops. It captures the checkout within five seconds and hands the
|
|
138
|
+
request to a detached sender; the service then starts the change's analysis
|
|
139
|
+
and, for a repository set up for crawling, its crawl in the background, so
|
|
140
|
+
`haystack verify` usually finds the crawl already running or finished. A newer stop in the same repository replaces an
|
|
141
|
+
older crawl that has not finished. `haystack verify precompute` (without
|
|
142
|
+
`--hook`) submits the same capture in the foreground and prints one line for
|
|
143
|
+
the crawl it started.
|
|
144
|
+
|
|
145
|
+
The hosted fleet run of exact pushed commits stays available:
|
|
109
146
|
|
|
110
147
|
```bash
|
|
148
|
+
haystack verify hosted start owner/repo --base <sha> --head <sha> --json
|
|
111
149
|
haystack verify hosted status cv_<48-lowercase-hex-characters> --wait --json
|
|
112
150
|
haystack verify history owner/repo --limit 20
|
|
113
151
|
```
|
|
114
152
|
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
153
|
+
Intent is optional there: `--intent-file <path>` accepts JSON with exactly
|
|
154
|
+
`problem`, `goal`, and `intended_outcomes`. It waits up to 35 minutes by
|
|
155
|
+
default; pass `--no-wait` to return after the fleet run is queued. Repeating the
|
|
156
|
+
same repository, commits, and intent reuses the same run; choose a new
|
|
157
|
+
`--idempotency-key` when a fresh execution is intentional. A wait that expires
|
|
158
|
+
prints `timed_out: true` and `next_command`, then exits 2. Use the returned
|
|
159
|
+
account-bound `next_command` verbatim on machines with multiple saved accounts.
|
|
118
160
|
|
|
119
161
|
Retained fleet cases remain available through their exact case IDs:
|
|
120
162
|
|
|
@@ -230,8 +272,8 @@ haystack submit --no-wait # Don't wait for analysis results
|
|
|
230
272
|
```
|
|
231
273
|
|
|
232
274
|
**Pre-PR triage**: two checkers run in parallel, each one structured call to
|
|
233
|
-
`gpt-6-astra` over the OpenAI Responses API at reasoning effort `
|
|
234
|
-
|
|
275
|
+
`gpt-6-astra` over the OpenAI Responses API at reasoning effort `xhigh`.
|
|
276
|
+
The key comes from `OPENAI_API_KEY`
|
|
235
277
|
if set, else from the SSM SecureString `/haystack/secrets/shared/prod/OPENAI_API_KEY`
|
|
236
278
|
(us-west-2) through the default AWS credential chain (e.g. `AWS_PROFILE=prod`).
|
|
237
279
|
With neither, each checker reports `no OpenAI key (...)` and submit continues.
|
|
@@ -462,7 +504,8 @@ haystack hooks install --force # Overwrite existing hooks
|
|
|
462
504
|
# Status
|
|
463
505
|
haystack hooks status # Check installation status
|
|
464
506
|
|
|
465
|
-
# Session hooks (triage on CLI start
|
|
507
|
+
# Session hooks (triage on CLI start; Claude Code's Stop hook starts a crawl,
|
|
508
|
+
# see `haystack verify`)
|
|
466
509
|
haystack hooks install-session # Auto-detect CLIs
|
|
467
510
|
haystack hooks install-session --cli claude # Claude Code only
|
|
468
511
|
haystack hooks install-session --cli all # All detected CLIs
|
|
@@ -520,6 +563,145 @@ behind an `UNKNOWN` that is not an arrival or wall outcome, and behind an
|
|
|
520
563
|
is `null` when the outcome has none. A case in `--cases` may carry `repeatIndex` (1-16), which
|
|
521
564
|
must be part of its `contentDigest`, to run as a repeat next to its original.
|
|
522
565
|
|
|
566
|
+
### `haystack db profile`
|
|
567
|
+
|
|
568
|
+
Describe your production Postgres database so Haystack can build a full-size
|
|
569
|
+
stand-in for testing, without any personal data leaving your environment. You
|
|
570
|
+
run the profiler yourself, inside your own network, against your own database;
|
|
571
|
+
it writes a JSON file you read before you send it.
|
|
572
|
+
|
|
573
|
+
```bash
|
|
574
|
+
# The connection string is read from the variable you name, never from a flag.
|
|
575
|
+
export DATABASE_URL='postgres://profiler_readonly@db.internal:5432/app'
|
|
576
|
+
haystack db profile --url-env DATABASE_URL --owner-table public.users --out profile.json
|
|
577
|
+
|
|
578
|
+
# Read exactly what would be sent, in plain English.
|
|
579
|
+
haystack db profile show profile.json
|
|
580
|
+
|
|
581
|
+
# Send it for one repository.
|
|
582
|
+
haystack db profile upload profile.json --repo acme/app
|
|
583
|
+
```
|
|
584
|
+
|
|
585
|
+
**Which database it is used for.** Uploaded without `--dependency`, the profile
|
|
586
|
+
belongs to the repository, and Haystack builds the stand-in from it for the
|
|
587
|
+
repository's one Postgres database, whatever id onboarding gave it. When the
|
|
588
|
+
repository has several (or none), `haystack verify hosted start` says the
|
|
589
|
+
stand-in was not used, and why, naming the ids; upload again with `--dependency <id>` to pick
|
|
590
|
+
one. A profile uploaded with `--dependency <id>` is used for that database; when
|
|
591
|
+
both apply, the newer upload is used. `haystack verify hosted start` prints one
|
|
592
|
+
line saying which profile the stand-in was built from, or why it was not used, for every
|
|
593
|
+
repository that has uploaded a profile.
|
|
594
|
+
|
|
595
|
+
**Connection strings.** Any `postgres://` or `postgresql://` URL that
|
|
596
|
+
node-postgres accepts works, including a Unix socket
|
|
597
|
+
(`postgresql://profiler_readonly@/app?host=/var/run/postgresql`).
|
|
598
|
+
|
|
599
|
+
**What it does to your database.** It only reads. Every query is a single
|
|
600
|
+
`SELECT` in its own `READ ONLY` transaction (the session default is read only
|
|
601
|
+
too, and the transaction's read-only state is checked before each query), with
|
|
602
|
+
a statement timeout and a 5-second lock timeout so it never queues behind a
|
|
603
|
+
migration. Each transaction is rolled back. All of them read one `REPEATABLE
|
|
604
|
+
READ` snapshot, so every number describes the same moment even while your
|
|
605
|
+
application writes; one extra connection holds that snapshot open for the run
|
|
606
|
+
(on a primary, rows deleted during the run are cleaned up by vacuum only after
|
|
607
|
+
it ends). A role with `SELECT` on the
|
|
608
|
+
application tables is enough, and is what we recommend; tables with row-level
|
|
609
|
+
security active for that role, or that the role cannot read, stop the run with
|
|
610
|
+
their names. It works on Postgres 12 and later (tested on 16).
|
|
611
|
+
|
|
612
|
+
**Run it on a read replica.** It works on a streaming-replication hot standby
|
|
613
|
+
(tested on Postgres 16): nothing it runs writes, takes a write lock or creates a
|
|
614
|
+
temporary table, and each transaction names its isolation level (`REPEATABLE
|
|
615
|
+
READ`), so a replica whose default is `serializable` still answers. A
|
|
616
|
+
replica does not count dead rows, so a profile taken there marks each table's
|
|
617
|
+
dead rows as unavailable and says why; everything else is the same as on the
|
|
618
|
+
primary. A replica may cancel a long query when replaying the primary's changes
|
|
619
|
+
cannot wait for it (`canceling statement due to conflict with recovery`), and
|
|
620
|
+
the run's snapshot is held for its whole length; the run then stops and names
|
|
621
|
+
the fix: turn on `hot_standby_feedback` on the replica, or raise its
|
|
622
|
+
`max_standby_streaming_delay` above `--statement-timeout`.
|
|
623
|
+
|
|
624
|
+
**What leaves your database.** The database computes every count and pattern;
|
|
625
|
+
no row ever reaches the profiler. A column's actual values are written only when
|
|
626
|
+
all three hold:
|
|
627
|
+
|
|
628
|
+
1. the column is a category: at most `--category-max-distinct` distinct values,
|
|
629
|
+
counted exactly over the whole table even when the table is otherwise
|
|
630
|
+
sampled (never estimated, so a column with many rare values cannot pass);
|
|
631
|
+
2. the column is not personal. Personal means names, emails, phones, addresses,
|
|
632
|
+
free text (notes, comments, descriptions, messages), secrets, tokens and
|
|
633
|
+
passwords, IP and network addresses, dates of birth and government IDs. It is
|
|
634
|
+
decided from the column's name, from its type (`inet`, `cidr`, `macaddr`), and
|
|
635
|
+
from value shapes the database measures (the share of values shaped like an
|
|
636
|
+
email, phone number, IP address, secret, SSN, card number or "First Last"
|
|
637
|
+
name, and the number of words per value). When in doubt a column is personal,
|
|
638
|
+
and the profile records why;
|
|
639
|
+
3. each value is shared by at least `--category-min-owners` owners. Owners are the
|
|
640
|
+
distinct rows of `--owner-table` (your users or organizations) reached by
|
|
641
|
+
following foreign keys from the column's table, and only owners that exist
|
|
642
|
+
there count (a key left dangling under a `NOT VALID` foreign key owns
|
|
643
|
+
nothing). A table with no foreign-key path to the owner table keeps no
|
|
644
|
+
values, shapes, JSON keys, percentiles or shares of true. Without
|
|
645
|
+
`--owner-table`, each row counts as an owner and the profile says so; a value
|
|
646
|
+
one heavy user repeats in thousands of rows then passes, so name an owner
|
|
647
|
+
table whenever you have one.
|
|
648
|
+
|
|
649
|
+
Values your schema declares (enum labels, `CHECK (col IN (...))` lists) need no
|
|
650
|
+
data and are kept with their share of rows. Everything else leaves only as
|
|
651
|
+
numbers: row counts (and whether each was counted exactly or is the planner's
|
|
652
|
+
estimate), how each table was read (in full, or the sample percentage and
|
|
653
|
+
seed), table/index/TOAST sizes, dead-row share (or why it is unavailable), null
|
|
654
|
+
share, distinct count, average width, the share of `true` in each boolean
|
|
655
|
+
column, and the 5th/25th/50th/75th/95th percentiles of lengths and numbers
|
|
656
|
+
(never the minimum or maximum; percentiles and the share of `true` are withheld
|
|
657
|
+
for a column with fewer than `--category-min-owners` owners, and number
|
|
658
|
+
percentiles are withheld for personal columns, because a percentile of birth
|
|
659
|
+
dates is a birth date). Character shapes (`Aa Aa`, `+9 (9) 9-9`) and JSON key
|
|
660
|
+
paths leave only when shared by at least `--category-min-owners` owners, and a
|
|
661
|
+
JSON key path never leaves when any key on it looks like an email, phone number,
|
|
662
|
+
IP address, SSN, card number or secret (`15551234567.verified` stays home). Odd
|
|
663
|
+
rows (wrong type, oversize, bad encoding, rare nulls, rare shapes, missing JSON
|
|
664
|
+
keys) are counted and described without their values, even for a single row.
|
|
665
|
+
Index definitions are sent with every literal replaced by `?`, since a partial
|
|
666
|
+
index can name a value (`WHERE email <> 'ceo@example.com'` is sent as `WHERE
|
|
667
|
+
(email <> ?::text)`); the column names, operators and types stay. Settings are
|
|
668
|
+
an allowlist of planner and storage settings; connection, authentication, SSL
|
|
669
|
+
and file settings are never read.
|
|
670
|
+
|
|
671
|
+
`show` prints every string the file carries. `upload` refuses a file with any
|
|
672
|
+
field `show` does not print, a file whose percentiles include anything but
|
|
673
|
+
p5 to p95, and a file with a literal left in an index definition. Haystack's
|
|
674
|
+
server checks the file again and also refuses one made with a looser rule than
|
|
675
|
+
at most 50 distinct values and at least 50 owners, a kept value its row share
|
|
676
|
+
shows fewer than `--category-min-owners` rows hold, and any kept value shaped
|
|
677
|
+
like an email, phone number, IP address, SSN, card number, token or password
|
|
678
|
+
hash, whatever its column's type, or that does not fit its column's declared
|
|
679
|
+
type (an integer column keeps only integers, a boolean column `true` or
|
|
680
|
+
`false`, a date or time column dates or times; a column of any other type than
|
|
681
|
+
text, numbers, booleans, dates and times keeps no data values, and enum labels
|
|
682
|
+
leave as schema values). The profiler never writes such a value itself:
|
|
683
|
+
however many owners share it (a date like `2031-07-19` has a phone number's
|
|
684
|
+
shape, and so does any number of seven or more digits), it is withheld like a
|
|
685
|
+
value too few owners share. An accepted file is stored once, under its
|
|
686
|
+
SHA-256, in Haystack's results storage.
|
|
687
|
+
|
|
688
|
+
Knobs and what turning them does:
|
|
689
|
+
|
|
690
|
+
| Flag | Default | Turning it up | Turning it down |
|
|
691
|
+
|------|---------|---------------|-----------------|
|
|
692
|
+
| `--owner-table <schema.table>` | none (rows are owners) | n/a | n/a; naming one makes the owner threshold count people or organizations instead of rows, which is stricter and what we recommend. |
|
|
693
|
+
| `--schema <name>` (repeatable) | every non-system schema | n/a | Profiles fewer schemas; foreign-key paths to the owner table only run through profiled tables. |
|
|
694
|
+
| `--category-max-distinct <n>` | 50 (1 to 1000) | More columns count as categories and can keep values (each still needs enough owners); `upload` accepts at most 50. | Fewer columns keep values; more leave only as numbers and shapes. |
|
|
695
|
+
| `--category-min-owners <n>` | 50 (5 to 1,000,000) | Stricter: fewer values, shapes, JSON keys and percentiles pass. | More rare values pass; each one is shared by fewer people. Not below 5; `upload` requires at least 50. |
|
|
696
|
+
| `--scan-max-mb <mb>` | 256 | Bigger tables are read in full: exact counts and more rare-but-common-enough values, at the cost of more reading on your database. | Tables above this are read through `TABLESAMPLE SYSTEM` of about this size (same pages for every query). Owner counts in a sample are lower than the truth, so fewer values pass (never more). Row counts of sampled tables come from the planner's statistics, so a sampled table must have been analyzed; the profile records the sample percentage and seed and marks the row count as an estimate. Distinct counts of a sampled table are estimated from the sample (a single-column primary or unique key counts one per row, and a validated single-column foreign key counts the parent rows it references), except a column that could be a category: its distinct values are counted exactly, which reads that table in full once for all such columns (the run says which tables). |
|
|
697
|
+
| `--concurrency <n>` | 2 (1 to 8) | Faster, more simultaneous load on your database. | Gentler; 1 runs one query at a time. Either way one more connection holds the run's snapshot. |
|
|
698
|
+
| `--statement-timeout <seconds>` | 120 | Allows longer queries on big tables. | Caps each query sooner; a query that hits it stops the run and names the table. |
|
|
699
|
+
|
|
700
|
+
Every failure is one line naming what failed. Nothing is written unless the
|
|
701
|
+
whole profile succeeds. On a primary, dead-row shares come from
|
|
702
|
+
`pg_stat_user_tables`; a table with rows but no counts there (statistics reset)
|
|
703
|
+
stops the run and asks for `ANALYZE` on that table.
|
|
704
|
+
|
|
523
705
|
### `haystack policy`
|
|
524
706
|
|
|
525
707
|
Manage review policies (`.haystack/review-policy.md`):
|
|
@@ -162,8 +162,8 @@ export function integer(value, what, minimum, maximum) {
|
|
|
162
162
|
}
|
|
163
163
|
return value;
|
|
164
164
|
}
|
|
165
|
-
/** Bounded JSON,
|
|
166
|
-
*
|
|
165
|
+
/** Bounded JSON: finite numbers, no prototype-polluting keys, depth 64, 8 MiB
|
|
166
|
+
* serialized. */
|
|
167
167
|
export function boundedCaseBatchJson(value, what, maximumBytes = CASE_BATCH_MAX_INPUT_BYTES) {
|
|
168
168
|
const visit = (item, depth) => {
|
|
169
169
|
if (depth > 64)
|
|
@@ -19,19 +19,8 @@ import { resolveAuthContext } from '../utils/auth.js';
|
|
|
19
19
|
import { classifyHttpError, HaystackApiError, haystackApiUrl } from '../utils/haystack-api.js';
|
|
20
20
|
import { boundedCaseBatchJson, buildCaseBatchRequest, CASE_BATCH_MAX_CASES, CASE_BATCH_MAX_CONCURRENT_CASES, CASE_BATCH_MAX_INPUT_BYTES, CASE_BATCH_MAX_CASE_WALL_MS, CASE_BATCH_MAX_TOTAL_BUDGET_MS, CASE_BATCH_MIN_CASE_WALL_MS, CASE_BATCH_MIN_TOTAL_BUDGET_MS, CaseBatchRequestValidationError, CaseBatchResponseError, isJsonObject, isTerminalCaseBatchStatus, parseCaseBatchCommit, parseCaseBatchIdempotencyKey, parseCaseBatchLimits, parseCaseBatchRepository, parseCaseBatchSnapshot, parseCaseBatchSource, parseCaseBatchWorld, parseProductCases, } from './case-batch-contract.js';
|
|
21
21
|
/** The CLI gateway mount. The same worker handlers are `/v1/case-batches` and
|
|
22
|
-
* `/api/cloud-verifier/case-batches
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
* BLOCKED ON THE SERVER LANE. No handler serves this path in this repository
|
|
26
|
-
* yet, and the auth worker does not proxy it: `isSearchProxyPath` in
|
|
27
|
-
* `infra/auth-worker/index.js` covers `/api/agent/cloud-verifier/searches`
|
|
28
|
-
* only, and `agent/cloudflare/src/combination-search-route.ts` has no
|
|
29
|
-
* case-batch sibling. Until Lane A of `docs/case-batch-coordinator.md` lands
|
|
30
|
-
* `POST/GET /v1/case-batches`, `GET/DELETE /v1/case-batches/:runId`,
|
|
31
|
-
* `GET /v1/case-batches/:runId/bundle[/path]` and the matching auth-worker
|
|
32
|
-
* proxy entry, every command in this file reaches a 404 in production. That
|
|
33
|
-
* lane must merge first; this file is written against the frozen contract
|
|
34
|
-
* ahead of it, which is what the implementation plan assigns to Lane D.
|
|
22
|
+
* `/api/cloud-verifier/case-batches` (`agent/cloudflare/src/case-batch-route.ts`);
|
|
23
|
+
* CLI callers use the agent gateway, which `infra/auth-worker/index.js` proxies.
|
|
35
24
|
*
|
|
36
25
|
* Three things the server side must hold for this file to work as written:
|
|
37
26
|
* the 8 MiB request bound, `repository`-scoped reads with `limit` and
|
|
@@ -40,8 +29,8 @@ import { boundedCaseBatchJson, buildCaseBatchRequest, CASE_BATCH_MAX_CASES, CASE
|
|
|
40
29
|
* is the one artifact with no parent digest to check it against). */
|
|
41
30
|
const GATEWAY = '/api/agent/cloud-verifier/case-batches';
|
|
42
31
|
const RUN_ID = /^cv_[0-9a-f]{48}$/;
|
|
43
|
-
/** Page size for the bounded case pagination; the
|
|
44
|
-
* `limit` at 500
|
|
32
|
+
/** Page size for the bounded case pagination; the batch read route caps
|
|
33
|
+
* `limit` at 500. */
|
|
45
34
|
const CASE_PAGE_LIMIT = 500;
|
|
46
35
|
/** A hard stop on pagination independent of what the server reports, so a
|
|
47
36
|
* cursor that never terminates cannot spin the CLI forever. */
|
|
@@ -102,8 +91,9 @@ function requireRunId(runId) {
|
|
|
102
91
|
return runId;
|
|
103
92
|
}
|
|
104
93
|
/** Every outbound request. The header set is built here and nowhere else, so
|
|
105
|
-
* the credential surface of this command is one line: the login bearer.
|
|
106
|
-
|
|
94
|
+
* the credential surface of this command is one line: the login bearer.
|
|
95
|
+
* `haystack verify` reads crawls through this same client. */
|
|
96
|
+
export async function gatewayFetch(path, token, init = { method: 'GET' }) {
|
|
107
97
|
const headers = new Headers();
|
|
108
98
|
headers.set('Authorization', `Bearer ${token}`);
|
|
109
99
|
headers.set('Accept', init.accept ?? 'application/json');
|