@octocrawl/sdk 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +17 -0
  3. package/dist/index.cjs +1278 -0
  4. package/dist/index.js +1240 -0
  5. package/dist/types/cjs/client.d.ts +362 -0
  6. package/dist/types/cjs/contracts/access.d.ts +166 -0
  7. package/dist/types/cjs/contracts/actions.d.ts +191 -0
  8. package/dist/types/cjs/contracts/api.d.ts +891 -0
  9. package/dist/types/cjs/contracts/benchmark.d.ts +116 -0
  10. package/dist/types/cjs/contracts/checkpoint.d.ts +164 -0
  11. package/dist/types/cjs/contracts/compliance.d.ts +412 -0
  12. package/dist/types/cjs/contracts/crawl.d.ts +302 -0
  13. package/dist/types/cjs/contracts/delivery.d.ts +136 -0
  14. package/dist/types/cjs/contracts/evidenceRecord.d.ts +192 -0
  15. package/dist/types/cjs/contracts/execution.d.ts +197 -0
  16. package/dist/types/cjs/contracts/extractor.d.ts +379 -0
  17. package/dist/types/cjs/contracts/file.d.ts +117 -0
  18. package/dist/types/cjs/contracts/firecrawl.d.ts +258 -0
  19. package/dist/types/cjs/contracts/groundTruth.d.ts +77 -0
  20. package/dist/types/cjs/contracts/identityBundle.d.ts +70 -0
  21. package/dist/types/cjs/contracts/index.d.ts +30 -0
  22. package/dist/types/cjs/contracts/map.d.ts +180 -0
  23. package/dist/types/cjs/contracts/monitor.d.ts +217 -0
  24. package/dist/types/cjs/contracts/monitorConfig.d.ts +9 -0
  25. package/dist/types/cjs/contracts/policy.d.ts +93 -0
  26. package/dist/types/cjs/contracts/proxy.d.ts +52 -0
  27. package/dist/types/cjs/contracts/recipe.d.ts +74 -0
  28. package/dist/types/cjs/contracts/regexSafety.d.ts +31 -0
  29. package/dist/types/cjs/contracts/result.d.ts +503 -0
  30. package/dist/types/cjs/contracts/session.d.ts +51 -0
  31. package/dist/types/cjs/contracts/ssrf.d.ts +16 -0
  32. package/dist/types/cjs/contracts/status.d.ts +27 -0
  33. package/dist/types/cjs/contracts/structured.d.ts +185 -0
  34. package/dist/types/cjs/contracts/tableMarkdown.d.ts +58 -0
  35. package/dist/types/cjs/contracts/tokens.d.ts +47 -0
  36. package/dist/types/cjs/index.d.ts +9 -0
  37. package/dist/types/cjs/package.json +1 -0
  38. package/dist/types/cjs/version.d.ts +8 -0
  39. package/dist/types/cjs/watcher.d.ts +151 -0
  40. package/dist/types/esm/client.d.ts +362 -0
  41. package/dist/types/esm/contracts/access.d.ts +166 -0
  42. package/dist/types/esm/contracts/actions.d.ts +191 -0
  43. package/dist/types/esm/contracts/api.d.ts +891 -0
  44. package/dist/types/esm/contracts/benchmark.d.ts +116 -0
  45. package/dist/types/esm/contracts/checkpoint.d.ts +164 -0
  46. package/dist/types/esm/contracts/compliance.d.ts +412 -0
  47. package/dist/types/esm/contracts/crawl.d.ts +302 -0
  48. package/dist/types/esm/contracts/delivery.d.ts +136 -0
  49. package/dist/types/esm/contracts/evidenceRecord.d.ts +192 -0
  50. package/dist/types/esm/contracts/execution.d.ts +197 -0
  51. package/dist/types/esm/contracts/extractor.d.ts +379 -0
  52. package/dist/types/esm/contracts/file.d.ts +117 -0
  53. package/dist/types/esm/contracts/firecrawl.d.ts +258 -0
  54. package/dist/types/esm/contracts/groundTruth.d.ts +77 -0
  55. package/dist/types/esm/contracts/identityBundle.d.ts +70 -0
  56. package/dist/types/esm/contracts/index.d.ts +30 -0
  57. package/dist/types/esm/contracts/map.d.ts +180 -0
  58. package/dist/types/esm/contracts/monitor.d.ts +217 -0
  59. package/dist/types/esm/contracts/monitorConfig.d.ts +9 -0
  60. package/dist/types/esm/contracts/policy.d.ts +93 -0
  61. package/dist/types/esm/contracts/proxy.d.ts +52 -0
  62. package/dist/types/esm/contracts/recipe.d.ts +74 -0
  63. package/dist/types/esm/contracts/regexSafety.d.ts +31 -0
  64. package/dist/types/esm/contracts/result.d.ts +503 -0
  65. package/dist/types/esm/contracts/session.d.ts +51 -0
  66. package/dist/types/esm/contracts/ssrf.d.ts +16 -0
  67. package/dist/types/esm/contracts/status.d.ts +27 -0
  68. package/dist/types/esm/contracts/structured.d.ts +185 -0
  69. package/dist/types/esm/contracts/tableMarkdown.d.ts +58 -0
  70. package/dist/types/esm/contracts/tokens.d.ts +47 -0
  71. package/dist/types/esm/index.d.ts +9 -0
  72. package/dist/types/esm/version.d.ts +8 -0
  73. package/dist/types/esm/watcher.d.ts +151 -0
  74. package/package.json +40 -0
@@ -0,0 +1,116 @@
1
+ import type { FetchResult } from './result.js';
2
+ import type { CheckResult, GroundTruth, SuiteMeta } from './groundTruth.js';
3
+ import type { Lane } from './status.js';
4
+ /** A crawler under test. Adapters wrap our runtime and every baseline/competitor. */
5
+ export interface Subject {
6
+ /** Stable identifier used in reports, e.g. 'bare-http', 'playwright', 'w2l'. */
7
+ id: string;
8
+ displayName: string;
9
+ /** Version of the underlying tool, recorded per run. */
10
+ version: string;
11
+ /**
12
+ * Whether results from this subject are cloud-hosted. Cloud subjects are
13
+ * reported in a separate column and never merged with self-hosted ones.
14
+ */
15
+ hosting: 'self_hosted' | 'cloud';
16
+ }
17
+ /** Reproducibility stamp. Recorded on every run; results without it are not comparable. */
18
+ export interface RunEnvironment {
19
+ gitCommit: string | null;
20
+ gitDirty: boolean;
21
+ nodeVersion: string;
22
+ platform: string;
23
+ arch: string;
24
+ cpuModel: string;
25
+ cpuCount: number;
26
+ totalMemoryBytes: number;
27
+ /** ISO timestamp of run start. */
28
+ startedAt: string;
29
+ /** Resolved versions of dependencies that affect extraction output. */
30
+ dependencyVersions: Readonly<Record<string, string>>;
31
+ }
32
+ export interface CaseOutcome {
33
+ caseId: string;
34
+ subjectId: string;
35
+ /** Terminal status the subject produced, verbatim. */
36
+ result: FetchResult;
37
+ /** Whether the terminal status matched the ground truth expectation. */
38
+ statusMatched: boolean;
39
+ /** Whether the settled lane matched expectation. Null when the subject has no lane concept. */
40
+ laneMatched: boolean | null;
41
+ /**
42
+ * Whether the named gate matched expectation. Null when the case carries no
43
+ * `expectedBlockReason` annotation — "not graded", never a free pass.
44
+ */
45
+ blockReasonMatched: boolean | null;
46
+ /**
47
+ * The five false-success checks. Every check appears exactly once, with
48
+ * outcome 'unknown' where evidence is unavailable. Never silently omitted.
49
+ */
50
+ checks: readonly CheckResult[];
51
+ /** Checks that were actually evaluated (outcome !== 'unknown'). */
52
+ evaluatedChecks: readonly string[];
53
+ /** True iff status is contentful and at least one check failed. */
54
+ isFalseSuccess: boolean;
55
+ budgetRespected: boolean;
56
+ }
57
+ export interface SuiteScore {
58
+ suite: SuiteMeta;
59
+ subjectId: string;
60
+ caseCount: number;
61
+ /** Cases whose terminal status matched the ground truth. */
62
+ statusMatchCount: number;
63
+ /**
64
+ * Of the cases annotated with an `expectedBlockReason`, how many the subject
65
+ * named correctly, and how many carried the annotation at all. Reported as a
66
+ * pair so "named 1 of 4" never reads like "named 1, nothing else asked".
67
+ */
68
+ blockReasonMatchCount: number;
69
+ blockReasonGradedCount: number;
70
+ /** Cases that returned content (success | partial). */
71
+ contentfulCount: number;
72
+ falseSuccessCount: number;
73
+ /**
74
+ * falseSuccessCount / contentfulCount, or null when nothing was contentful.
75
+ * Deliberately not blended into any composite score.
76
+ */
77
+ falseSuccessRate: number | null;
78
+ /** Per-check tallies, so an unknown-heavy canary run is visible rather than hidden. */
79
+ checkTallies: Readonly<Record<string, {
80
+ pass: number;
81
+ fail: number;
82
+ unknown: number;
83
+ }>>;
84
+ laneDistribution: Readonly<Partial<Record<Lane, number>>>;
85
+ medianWallMs: number;
86
+ p95WallMs: number;
87
+ medianContentTokens: number | null;
88
+ budgetViolations: number;
89
+ /** Ground-truth quality split by L0 identity, L1 HTTP, and L2 browser cases. */
90
+ qualityByTier: Readonly<Record<'L0' | 'L1' | 'L2', TierScore>>;
91
+ /** Contentful outcomes whose reported execution cost is fully known. */
92
+ verifiedCompletionRate: number | null;
93
+ knownCostPerContentfulPageUsd: number | null;
94
+ escalationCount: number;
95
+ }
96
+ export interface TierScore {
97
+ caseCount: number;
98
+ statusMatchCount: number;
99
+ contentfulCount: number;
100
+ falseSuccessCount: number;
101
+ falseSuccessRate: number | null;
102
+ p95WallMs: number;
103
+ knownCostPerContentfulPageUsd: number | null;
104
+ }
105
+ export interface BenchmarkRun {
106
+ runId: string;
107
+ environment: RunEnvironment;
108
+ suite: SuiteMeta;
109
+ subjects: readonly Subject[];
110
+ /** Lanes this run was permitted to exercise. Public canary runs restrict to Tier 0+1a. */
111
+ lanesUnderTest: readonly Lane[];
112
+ cases: readonly GroundTruth[];
113
+ outcomes: readonly CaseOutcome[];
114
+ scores: readonly SuiteScore[];
115
+ }
116
+ //# sourceMappingURL=benchmark.d.ts.map
@@ -0,0 +1,164 @@
1
+ /**
2
+ * Checkpoint contract: `task → attempt → step` at URL granularity.
3
+ *
4
+ * Persistence lives behind TaskStore (ADR 0003). This file is types only —
5
+ * no I/O, no SQL. Callers generate UUIDs; the store never autoincrements.
6
+ *
7
+ * Granularity is the page. A partial parse inside a page is not a step; the
8
+ * whole URL is retried. Block-level checkpoint is out of Phase 1.
9
+ */
10
+ import type { PageOptions, RequestAttribution, RobotsUrlOverride, WebhookEvent } from './api.js';
11
+ import type { CrawlMode } from './compliance.js';
12
+ import type { WebhookPayloadFormat } from './delivery.js';
13
+ import type { CrawlDiscovery, SitemapMode } from './crawl.js';
14
+ import type { FetchResult, LadderRunAudit } from './result.js';
15
+ import type { ScrapeFormat } from './structured.js';
16
+ import type { BudgetKind, Lane, ResultStatus } from './status.js';
17
+ export declare const TASK_STATUS: readonly ["pending", "running", "paused", "completed", "failed", "cancelled"];
18
+ export type TaskStatus = (typeof TASK_STATUS)[number];
19
+ export declare const ATTEMPT_STATUS: readonly ["running", "completed", "failed", "cancelled", "interrupted"];
20
+ export type AttemptStatus = (typeof ATTEMPT_STATUS)[number];
21
+ export declare const STEP_STATUS: readonly ["pending", "running", "success", "partial", "empty_verified", "blocked", "failed", "cancelled", "budget_exceeded", "duplicate"];
22
+ export type StepStatus = (typeof STEP_STATUS)[number];
23
+ /**
24
+ * Hard caps for one crawl task. Null means "this dimension is not bounded".
25
+ * Spent meters live on Attempt, not here — the spec is the ceiling.
26
+ */
27
+ export interface CrawlBudget {
28
+ maxPages: number | null;
29
+ maxWallMs: number | null;
30
+ maxCostUsd: number | null;
31
+ maxTokens: number | null;
32
+ }
33
+ export declare const DEFAULT_CRAWL_BUDGET: CrawlBudget;
34
+ /**
35
+ * A job's webhook as its task stores it: the receiver, the events taken,
36
+ * the metadata echoed in every payload, the signing secret's name and the
37
+ * destination (`job:<taskId>`) in the control database. Custom headers live
38
+ * in that database alone, never here.
39
+ */
40
+ export interface StoredJobWebhook {
41
+ url: string;
42
+ events: readonly WebhookEvent[];
43
+ metadata: Readonly<Record<string, string>>;
44
+ secretEnv?: string;
45
+ destinationId: string;
46
+ /** `firecrawl` for a job started through the `/fc` shim, whose receiver gets Firecrawl's payload shape; absent means W2L's envelope. */
47
+ payloadFormat?: WebhookPayloadFormat;
48
+ }
49
+ /** One crawl job. The SQLite file sits next to `taskDir`. */
50
+ export interface Task {
51
+ id: string;
52
+ seedUrl: string;
53
+ taskDir: string;
54
+ mode: CrawlMode;
55
+ status: TaskStatus;
56
+ budget: CrawlBudget;
57
+ /**
58
+ * Present only for an explicit URL-array batch. Stored with the checkpoint,
59
+ * page options and recorded robots overrides included, plus the batch's own
60
+ * cap on pages in flight (`maxConcurrency`, absent when the request set
61
+ * none) and the entries `ignoreInvalidURLs` skipped at submission
62
+ * (`invalidURLs`, present exactly when that option was on). An append
63
+ * (`appendToId`) extends `urls`, `robotsOverrides` and `invalidURLs` in
64
+ * place, pushing to the end in order: the orchestrator seeds the tail past
65
+ * what it has seeded, by index, and never a URL twice.
66
+ */
67
+ batch?: {
68
+ urls: readonly string[];
69
+ formats: readonly ScrapeFormat[];
70
+ includeLinks: boolean;
71
+ robotsOverrides?: readonly RobotsUrlOverride[];
72
+ maxConcurrency?: number;
73
+ invalidURLs?: readonly string[];
74
+ webhook?: StoredJobWebhook;
75
+ } & PageOptions;
76
+ /**
77
+ * Every crawl option but the page budget (`budget`), stored when the crawl
78
+ * starts so a resumed crawl runs with the options it was started with.
79
+ */
80
+ crawl?: {
81
+ formats?: readonly ScrapeFormat[];
82
+ includeLinks?: boolean;
83
+ includePaths?: readonly string[];
84
+ excludePaths?: readonly string[];
85
+ /** Link hops from the seed; null is unbounded. Absent on a task stored before depth was kept. */
86
+ maxDepth?: number | null;
87
+ /** Hosts links may lead to, beside the seed's host, its apex/www twin and where the seed redirected. */
88
+ allowlistedDomains?: readonly string[];
89
+ /** A resume reuses the pages this task already fetched instead of fetching them again. */
90
+ useCached?: boolean;
91
+ /**
92
+ * The URL-scope options the crawl was started with (see CrawlStartRequest).
93
+ * Absent on a task stored before they were kept: such a task resumes with
94
+ * the rule it was started under, whole host (`crawlEntireDomain` true) and
95
+ * exact canonical URLs (`deduplicateSimilarURLs` false), the rest false.
96
+ */
97
+ regexOnFullURL?: boolean;
98
+ ignoreQueryParameters?: boolean;
99
+ deduplicateSimilarURLs?: boolean;
100
+ crawlEntireDomain?: boolean;
101
+ allowSubdomains?: boolean;
102
+ allowExternalLinks?: boolean;
103
+ /** How the crawl uses the site's sitemap. Absent on a task stored before it was kept: such a task resumes as `skip`. */
104
+ sitemap?: SitemapMode;
105
+ /** The crawl's own cap on pages fetched at once; null takes the service's worker count. */
106
+ maxConcurrency?: number | null;
107
+ /** The crawl's webhook, when the request set one. */
108
+ webhook?: StoredJobWebhook;
109
+ } & PageOptions;
110
+ /** Who started the task (`origin`, `integration`), stored with it and reported as `attribution` on its status; absent when the request named neither. */
111
+ attribution?: RequestAttribution;
112
+ createdAt: string;
113
+ updatedAt: string;
114
+ }
115
+ /**
116
+ * One execution of a task. A resume after crash opens a new attempt against
117
+ * the same task so history is not overwritten (PHASE1 / PRODUCT_PLAN_V2 §4.3).
118
+ */
119
+ export interface Attempt {
120
+ id: string;
121
+ taskId: string;
122
+ status: AttemptStatus;
123
+ startedAt: string;
124
+ endedAt: string | null;
125
+ pagesFetched: number;
126
+ wallMs: number;
127
+ costUsd: number | null;
128
+ costUnknown?: boolean;
129
+ contentTokens: number;
130
+ contentTokensUnknown?: boolean;
131
+ /** Which budget dimension stopped this attempt, if any. */
132
+ budgetExceeded: BudgetKind | null;
133
+ /** Set when this attempt resumes a previously interrupted attempt. */
134
+ recoveredFromAttemptId?: string | null;
135
+ /** What this attempt's pages offered the frontier and what became of it, written after every page of a crawl; absent for a batch and for an attempt stored before it was kept. */
136
+ discovery?: CrawlDiscovery | null;
137
+ }
138
+ /**
139
+ * One URL inside one attempt. The atomic checkpoint unit.
140
+ *
141
+ * `canonicalUrl` is the dedupe key the frontier will use later; this slice
142
+ * stores whatever the caller supplies and does not normalize.
143
+ * `contentHash` is the body hash used on resume to decide refetch vs cache.
144
+ * `result` is the page FetchResult when one exists; null while pending/running.
145
+ */
146
+ export interface StepRecord {
147
+ id: string;
148
+ taskId: string;
149
+ attemptId: string;
150
+ url: string;
151
+ canonicalUrl: string;
152
+ depth: number;
153
+ status: StepStatus;
154
+ lane: Lane | null;
155
+ contentHash: string | null;
156
+ cached: boolean;
157
+ result: FetchResult | null;
158
+ audit?: LadderRunAudit;
159
+ createdAt: string;
160
+ updatedAt: string;
161
+ }
162
+ /** ResultStatus and StepStatus share the terminal page outcomes. */
163
+ export declare function stepStatusFromResult(status: ResultStatus): StepStatus;
164
+ //# sourceMappingURL=checkpoint.d.ts.map