@oxygen-agent/cli 1.696.2 → 1.717.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/help.js +2 -0
- package/dist/http-client.js +11 -1
- package/dist/index.js +524 -86
- package/dist/transcript.js +2 -2
- package/node_modules/@oxygen/recipe-sdk/dist/index.d.ts +4 -0
- package/node_modules/@oxygen/shared/dist/billing.d.ts +1 -1
- package/node_modules/@oxygen/shared/dist/billing.js +7 -6
- package/node_modules/@oxygen/shared/dist/dnc-identities.d.ts +53 -0
- package/node_modules/@oxygen/shared/dist/dnc-identities.js +175 -0
- package/node_modules/@oxygen/shared/dist/file-import.d.ts +36 -1
- package/node_modules/@oxygen/shared/dist/file-import.js +80 -1
- package/node_modules/@oxygen/shared/dist/image-sniff.d.ts +39 -0
- package/node_modules/@oxygen/shared/dist/image-sniff.js +75 -0
- package/node_modules/@oxygen/shared/dist/import-limits.d.ts +28 -0
- package/node_modules/@oxygen/shared/dist/import-limits.js +30 -0
- package/node_modules/@oxygen/shared/dist/index.d.ts +2 -0
- package/node_modules/@oxygen/shared/dist/index.js +2 -0
- package/node_modules/@oxygen/shared/dist/mailbox-import.d.ts +10 -4
- package/node_modules/@oxygen/shared/dist/mailbox-import.js +18 -10
- package/node_modules/@oxygen/shared/dist/member-columns.d.ts +64 -0
- package/node_modules/@oxygen/shared/dist/member-columns.js +111 -0
- package/node_modules/@oxygen/shared/dist/microsoft-consent-url.js +1 -1
- package/node_modules/@oxygen/shared/dist/object-storage.d.ts +65 -0
- package/node_modules/@oxygen/shared/dist/object-storage.js +100 -0
- package/node_modules/@oxygen/shared/dist/plan-limits.d.ts +2 -1
- package/node_modules/@oxygen/shared/dist/plan-limits.js +12 -2
- package/node_modules/@oxygen/shared/dist/spend-safety.d.ts +9 -0
- package/node_modules/@oxygen/shared/dist/spend-safety.js +10 -0
- package/node_modules/@oxygen/shared/dist/version.d.ts +1 -1
- package/node_modules/@oxygen/shared/dist/version.js +1 -1
- package/node_modules/@oxygen/shared/package.json +20 -0
- package/node_modules/@oxygen/workflows/dist/graph/lint.js +79 -11
- package/node_modules/@oxygen/workflows/dist/graph/manifest-schema.d.ts +679 -0
- package/node_modules/@oxygen/workflows/dist/graph/manifest-schema.js +80 -1
- package/node_modules/@oxygen/workflows/dist/graph/remap.js +8 -1
- package/node_modules/@oxygen/workflows/dist/graph/types.d.ts +72 -3
- package/node_modules/@oxygen/workflows/dist/graph/types.js +12 -0
- package/node_modules/@oxygen/workflows/dist/index.d.ts +16 -0
- package/node_modules/@oxygen/workflows/dist/index.js +45 -3
- package/node_modules/@oxygen/workflows/dist/portable.d.ts +43 -0
- package/node_modules/@oxygen/workflows/dist/portable.js +319 -0
- package/node_modules/@oxygen/workflows/dist/usage-estimate.js +18 -4
- package/package.json +1 -1
package/dist/transcript.js
CHANGED
|
@@ -116,8 +116,8 @@ function countEvents(buf) {
|
|
|
116
116
|
}
|
|
117
117
|
// Redact obvious credentials from transcript text before it leaves the machine.
|
|
118
118
|
// Agent JSONL can embed API keys, auth headers, and tokens captured from tool
|
|
119
|
-
// calls / shell output; scrub the common shapes
|
|
120
|
-
//
|
|
119
|
+
// calls / shell output; scrub the common shapes before an explicitly requested
|
|
120
|
+
// attachment reaches the support API. Best-effort (not a guarantee) and deliberately
|
|
121
121
|
// conservative — it targets credential shapes, not general prose, so the
|
|
122
122
|
// transcript stays useful for debugging.
|
|
123
123
|
const SECRET_PATTERNS = [
|
|
@@ -5,6 +5,8 @@ export type RecipeRuntime = "durable";
|
|
|
5
5
|
export type RecipeToolRunOptions = {
|
|
6
6
|
key?: string;
|
|
7
7
|
optional?: boolean;
|
|
8
|
+
/** Pin this call to one provider account for every replay/scheduled run. */
|
|
9
|
+
connectionId?: string;
|
|
8
10
|
};
|
|
9
11
|
export type RecipeToolApi = {
|
|
10
12
|
run: <T = unknown>(toolId: string, payload?: Record<string, unknown>, options?: RecipeToolRunOptions) => Promise<T>;
|
|
@@ -145,6 +147,8 @@ export type RecipeWaitResult = {
|
|
|
145
147
|
};
|
|
146
148
|
export type RecipeContext = {
|
|
147
149
|
input: unknown;
|
|
150
|
+
/** Operator-managed settings from the recipe manifest, separate from the trigger payload. */
|
|
151
|
+
configuration: Record<string, unknown>;
|
|
148
152
|
mode: WorkflowMode;
|
|
149
153
|
organizationId: string;
|
|
150
154
|
tools: RecipeToolApi;
|
|
@@ -125,7 +125,7 @@ export declare const BASE_PRICING_PLANS: {
|
|
|
125
125
|
readonly automationOverageCentsPerMillion: null;
|
|
126
126
|
readonly automationOverageEnabledDefault: false;
|
|
127
127
|
readonly byokEnabled: false;
|
|
128
|
-
readonly description: "No active plan. Start a 7-day
|
|
128
|
+
readonly description: "No active plan. Start a 7-day trial on a self-serve plan to use OXYGEN — existing credit balances remain spendable.";
|
|
129
129
|
readonly ctaLabel: "Start trial";
|
|
130
130
|
readonly features: readonly ["Existing credits stay spendable", "Read access to your workspace"];
|
|
131
131
|
};
|
|
@@ -145,10 +145,11 @@ export function resolveCreditTopupPack(id) {
|
|
|
145
145
|
return null;
|
|
146
146
|
return CREDIT_TOPUP_PACKS.find((pack) => pack.id === id) ?? null;
|
|
147
147
|
}
|
|
148
|
-
// Card-required free trial: every new signup
|
|
149
|
-
//
|
|
150
|
-
// guided first outcome — NOT the
|
|
151
|
-
// credit exhaustion or when the trial clock elapses,
|
|
148
|
+
// Card-required free trial: every new signup chooses a self-serve Stripe plan,
|
|
149
|
+
// gets TRIAL_CREDIT_GRANT credits (a value-demonstration budget sized to the
|
|
150
|
+
// guided first outcome — NOT the selected plan's full allowance), and converts
|
|
151
|
+
// to that paid plan on credit exhaustion or when the trial clock elapses,
|
|
152
|
+
// whichever comes first. Starter remains the signup default.
|
|
152
153
|
export const TRIAL_PERIOD_DAYS = 7;
|
|
153
154
|
export const TRIAL_CREDIT_GRANT = 20_000;
|
|
154
155
|
// Onboarding walkthrough (2026-07-21): completing the in-product walkthrough
|
|
@@ -160,7 +161,7 @@ export const WALKTHROUGH_COMPLETION_BONUS_CREDITS = 1_000;
|
|
|
160
161
|
export const WALKTHROUGH_SCRAPE_MAX_CREDITS = 2_000;
|
|
161
162
|
export const BASE_PRICING_PLANS = {
|
|
162
163
|
// Fully retired as a distribution surface (2026-07-22; entry is the
|
|
163
|
-
// card-required
|
|
164
|
+
// card-required self-serve trial). The tier survives only as the unentitled
|
|
164
165
|
// fallback for churned / never-subscribed orgs: zero monthly credits means
|
|
165
166
|
// ensureCurrentPlanCreditGrant short-circuits — no monthly minting AND no
|
|
166
167
|
// rollover-cap claw-back of leftover balances, which stay spendable.
|
|
@@ -175,7 +176,7 @@ export const BASE_PRICING_PLANS = {
|
|
|
175
176
|
automationOverageCentsPerMillion: null,
|
|
176
177
|
automationOverageEnabledDefault: false,
|
|
177
178
|
byokEnabled: false,
|
|
178
|
-
description: "No active plan. Start a 7-day
|
|
179
|
+
description: "No active plan. Start a 7-day trial on a self-serve plan to use OXYGEN — existing credit balances remain spendable.",
|
|
179
180
|
ctaLabel: "Start trial",
|
|
180
181
|
features: [
|
|
181
182
|
"Existing credits stay spendable",
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Source-neutral identities accepted by the unified DNC write contract.
|
|
3
|
+
*
|
|
4
|
+
* The existing mixed-text suppression importer intentionally remains separate:
|
|
5
|
+
* a bare domain there means an email-domain block. Typed `company_domain` and
|
|
6
|
+
* `linkedin_company` entries are organization-wide and stop every Sequence
|
|
7
|
+
* channel. Keeping the two contracts distinct preserves backwards compatibility
|
|
8
|
+
* while letting HubSpot, Tables, CSVs, webhooks, and future CRM adapters share one
|
|
9
|
+
* explicit ingestion boundary.
|
|
10
|
+
*/
|
|
11
|
+
export declare const DNC_IDENTITY_KINDS: readonly ["email", "phone", "linkedin_person", "company_domain", "linkedin_company"];
|
|
12
|
+
export type DncIdentityKind = (typeof DNC_IDENTITY_KINDS)[number];
|
|
13
|
+
export type DncIdentityInput = {
|
|
14
|
+
kind: DncIdentityKind;
|
|
15
|
+
value: string;
|
|
16
|
+
reason?: string;
|
|
17
|
+
source?: string;
|
|
18
|
+
detail?: string;
|
|
19
|
+
metadata?: Record<string, unknown>;
|
|
20
|
+
};
|
|
21
|
+
export type CompanyDncIdentity = {
|
|
22
|
+
kind: "domain" | "linkedin_company";
|
|
23
|
+
value: string;
|
|
24
|
+
key: string;
|
|
25
|
+
};
|
|
26
|
+
/** Canonical company hostname, without scheme, path, port, www, or trailing dot. */
|
|
27
|
+
export declare function normalizeCompanyDncDomain(value: unknown): string | null;
|
|
28
|
+
/** Canonical public LinkedIn organization URL (`/company/` or `/school/`). */
|
|
29
|
+
export declare function normalizeLinkedinCompanyUrl(value: unknown): string | null;
|
|
30
|
+
/**
|
|
31
|
+
* Final shape normalization shared by all typed DNC adapters. Tenant DB writers
|
|
32
|
+
* still apply their ledger-specific validators before persistence.
|
|
33
|
+
*/
|
|
34
|
+
export declare function normalizeDncIdentityValue(kind: DncIdentityKind, value: unknown): string | null;
|
|
35
|
+
export declare function companyDncIdentityKey(kind: CompanyDncIdentity["kind"], value: string): string;
|
|
36
|
+
/**
|
|
37
|
+
* Every explicit identity that can represent one LinkedIn person at the DNC
|
|
38
|
+
* boundary. Provider ids remain opaque/case-sensitive; public profile URLs are
|
|
39
|
+
* canonicalized to the same stable key used by typed importers. The helper is
|
|
40
|
+
* source-neutral so enrollment, planning, dispatch, and final provider
|
|
41
|
+
* boundaries cannot drift onto different subsets of identity fields.
|
|
42
|
+
*/
|
|
43
|
+
export declare function linkedinPersonDncKeys(input: {
|
|
44
|
+
leadProviderId?: unknown;
|
|
45
|
+
leadProfileUrl?: unknown;
|
|
46
|
+
rowValues?: Record<string, unknown> | null;
|
|
47
|
+
}): string[];
|
|
48
|
+
/**
|
|
49
|
+
* Extract only explicit company identities from an enrollment row. We never
|
|
50
|
+
* derive a company-wide block from a person's email domain: one person's opt-out
|
|
51
|
+
* must not silently suppress their entire employer.
|
|
52
|
+
*/
|
|
53
|
+
export declare function companyDncIdentitiesFromRowValues(rowValues: Record<string, unknown> | null | undefined): CompanyDncIdentity[];
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
import { normalizeLinkedinProfileUrl } from "./linkedin-url.js";
|
|
2
|
+
import { LINKEDIN_MEMBER_URN_PATTERN } from "./suppression-entries.js";
|
|
3
|
+
/**
|
|
4
|
+
* Source-neutral identities accepted by the unified DNC write contract.
|
|
5
|
+
*
|
|
6
|
+
* The existing mixed-text suppression importer intentionally remains separate:
|
|
7
|
+
* a bare domain there means an email-domain block. Typed `company_domain` and
|
|
8
|
+
* `linkedin_company` entries are organization-wide and stop every Sequence
|
|
9
|
+
* channel. Keeping the two contracts distinct preserves backwards compatibility
|
|
10
|
+
* while letting HubSpot, Tables, CSVs, webhooks, and future CRM adapters share one
|
|
11
|
+
* explicit ingestion boundary.
|
|
12
|
+
*/
|
|
13
|
+
export const DNC_IDENTITY_KINDS = [
|
|
14
|
+
"email",
|
|
15
|
+
"phone",
|
|
16
|
+
"linkedin_person",
|
|
17
|
+
"company_domain",
|
|
18
|
+
"linkedin_company",
|
|
19
|
+
];
|
|
20
|
+
const E164_PATTERN = /^\+[1-9]\d{1,14}$/;
|
|
21
|
+
const DOMAIN_PATTERN = /^(?=.{1,253}$)(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/;
|
|
22
|
+
const COMPANY_DOMAIN_ROW_KEYS = [
|
|
23
|
+
"company_domain",
|
|
24
|
+
"companyDomain",
|
|
25
|
+
"domain",
|
|
26
|
+
"website_domain",
|
|
27
|
+
"websiteDomain",
|
|
28
|
+
];
|
|
29
|
+
const LINKEDIN_COMPANY_ROW_KEYS = [
|
|
30
|
+
"company_linkedin_url",
|
|
31
|
+
"companyLinkedinUrl",
|
|
32
|
+
"linkedin_company_url",
|
|
33
|
+
"linkedinCompanyUrl",
|
|
34
|
+
"linkedin_company_page",
|
|
35
|
+
"linkedinCompanyPage",
|
|
36
|
+
];
|
|
37
|
+
const LINKEDIN_PERSON_ROW_KEYS = [
|
|
38
|
+
"linkedin_url",
|
|
39
|
+
"linkedinUrl",
|
|
40
|
+
"person_linkedin_url",
|
|
41
|
+
"personLinkedinUrl",
|
|
42
|
+
"lead_profile_url",
|
|
43
|
+
"leadProfileUrl",
|
|
44
|
+
];
|
|
45
|
+
function readNonEmptyString(value) {
|
|
46
|
+
if (typeof value !== "string")
|
|
47
|
+
return null;
|
|
48
|
+
const trimmed = value.trim();
|
|
49
|
+
return trimmed ? trimmed : null;
|
|
50
|
+
}
|
|
51
|
+
/** Canonical company hostname, without scheme, path, port, www, or trailing dot. */
|
|
52
|
+
export function normalizeCompanyDncDomain(value) {
|
|
53
|
+
const raw = readNonEmptyString(value)?.toLowerCase();
|
|
54
|
+
if (!raw || raw.includes("@") || /\s/.test(raw))
|
|
55
|
+
return null;
|
|
56
|
+
try {
|
|
57
|
+
const parsed = new URL(raw.includes("://") ? raw : `https://${raw}`);
|
|
58
|
+
if (parsed.username || parsed.password)
|
|
59
|
+
return null;
|
|
60
|
+
const hostname = parsed.hostname.replace(/^www\./, "").replace(/\.$/, "");
|
|
61
|
+
if (!DOMAIN_PATTERN.test(hostname))
|
|
62
|
+
return null;
|
|
63
|
+
return hostname;
|
|
64
|
+
}
|
|
65
|
+
catch {
|
|
66
|
+
return null;
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
/** Canonical public LinkedIn organization URL (`/company/` or `/school/`). */
|
|
70
|
+
export function normalizeLinkedinCompanyUrl(value) {
|
|
71
|
+
const raw = readNonEmptyString(value);
|
|
72
|
+
if (!raw || /\s/.test(raw))
|
|
73
|
+
return null;
|
|
74
|
+
try {
|
|
75
|
+
const parsed = new URL(raw.includes("://") ? raw : `https://${raw}`);
|
|
76
|
+
const hostname = parsed.hostname.toLowerCase();
|
|
77
|
+
if (hostname !== "linkedin.com" && !hostname.endsWith(".linkedin.com"))
|
|
78
|
+
return null;
|
|
79
|
+
const match = parsed.pathname.match(/^\/(company|school)\/([^/?#]+)\/?$/i);
|
|
80
|
+
if (!match?.[1] || !match[2])
|
|
81
|
+
return null;
|
|
82
|
+
return `https://www.linkedin.com/${match[1].toLowerCase()}/${match[2].toLowerCase()}`;
|
|
83
|
+
}
|
|
84
|
+
catch {
|
|
85
|
+
return null;
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* Final shape normalization shared by all typed DNC adapters. Tenant DB writers
|
|
90
|
+
* still apply their ledger-specific validators before persistence.
|
|
91
|
+
*/
|
|
92
|
+
export function normalizeDncIdentityValue(kind, value) {
|
|
93
|
+
const raw = readNonEmptyString(value);
|
|
94
|
+
if (!raw)
|
|
95
|
+
return null;
|
|
96
|
+
switch (kind) {
|
|
97
|
+
case "email": {
|
|
98
|
+
const email = raw.toLowerCase();
|
|
99
|
+
return email.includes("@") && !/\s/.test(email) ? email : null;
|
|
100
|
+
}
|
|
101
|
+
case "phone":
|
|
102
|
+
return E164_PATTERN.test(raw) ? raw : null;
|
|
103
|
+
case "linkedin_person": {
|
|
104
|
+
const profile = normalizeLinkedinProfileUrl(raw);
|
|
105
|
+
if (profile)
|
|
106
|
+
return profile.normalized;
|
|
107
|
+
return LINKEDIN_MEMBER_URN_PATTERN.test(raw) ? raw : null;
|
|
108
|
+
}
|
|
109
|
+
case "company_domain":
|
|
110
|
+
return normalizeCompanyDncDomain(raw);
|
|
111
|
+
case "linkedin_company":
|
|
112
|
+
return normalizeLinkedinCompanyUrl(raw);
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
export function companyDncIdentityKey(kind, value) {
|
|
116
|
+
return `${kind}:${value}`;
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* Every explicit identity that can represent one LinkedIn person at the DNC
|
|
120
|
+
* boundary. Provider ids remain opaque/case-sensitive; public profile URLs are
|
|
121
|
+
* canonicalized to the same stable key used by typed importers. The helper is
|
|
122
|
+
* source-neutral so enrollment, planning, dispatch, and final provider
|
|
123
|
+
* boundaries cannot drift onto different subsets of identity fields.
|
|
124
|
+
*/
|
|
125
|
+
export function linkedinPersonDncKeys(input) {
|
|
126
|
+
const keys = new Set();
|
|
127
|
+
const providerId = readNonEmptyString(input.leadProviderId);
|
|
128
|
+
if (providerId)
|
|
129
|
+
keys.add(providerId);
|
|
130
|
+
const addProfile = (value) => {
|
|
131
|
+
const raw = readNonEmptyString(value);
|
|
132
|
+
if (!raw)
|
|
133
|
+
return;
|
|
134
|
+
const normalized = normalizeLinkedinProfileUrl(raw)?.normalized;
|
|
135
|
+
if (normalized)
|
|
136
|
+
keys.add(normalized);
|
|
137
|
+
};
|
|
138
|
+
addProfile(input.leadProfileUrl);
|
|
139
|
+
for (const key of LINKEDIN_PERSON_ROW_KEYS)
|
|
140
|
+
addProfile(input.rowValues?.[key]);
|
|
141
|
+
return [...keys];
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* Extract only explicit company identities from an enrollment row. We never
|
|
145
|
+
* derive a company-wide block from a person's email domain: one person's opt-out
|
|
146
|
+
* must not silently suppress their entire employer.
|
|
147
|
+
*/
|
|
148
|
+
export function companyDncIdentitiesFromRowValues(rowValues) {
|
|
149
|
+
if (!rowValues)
|
|
150
|
+
return [];
|
|
151
|
+
const identities = new Map();
|
|
152
|
+
for (const key of COMPANY_DOMAIN_ROW_KEYS) {
|
|
153
|
+
const value = normalizeCompanyDncDomain(rowValues[key]);
|
|
154
|
+
if (!value)
|
|
155
|
+
continue;
|
|
156
|
+
const identity = {
|
|
157
|
+
kind: "domain",
|
|
158
|
+
value,
|
|
159
|
+
key: companyDncIdentityKey("domain", value),
|
|
160
|
+
};
|
|
161
|
+
identities.set(identity.key, identity);
|
|
162
|
+
}
|
|
163
|
+
for (const key of LINKEDIN_COMPANY_ROW_KEYS) {
|
|
164
|
+
const value = normalizeLinkedinCompanyUrl(rowValues[key]);
|
|
165
|
+
if (!value)
|
|
166
|
+
continue;
|
|
167
|
+
const identity = {
|
|
168
|
+
kind: "linkedin_company",
|
|
169
|
+
value,
|
|
170
|
+
key: companyDncIdentityKey("linkedin_company", value),
|
|
171
|
+
};
|
|
172
|
+
identities.set(identity.key, identity);
|
|
173
|
+
}
|
|
174
|
+
return [...identities.values()];
|
|
175
|
+
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { type Row } from "read-excel-file/node";
|
|
2
2
|
import { type ImportColumnDataType } from "./column-types.js";
|
|
3
|
+
export { MAX_BUFFERED_IMPORT_PARSE_BYTES } from "./import-limits.js";
|
|
3
4
|
export type RowsFileFormat = "json" | "jsonl" | "csv" | "xlsx";
|
|
4
5
|
export type ImportTableColumn = {
|
|
5
6
|
label: string;
|
|
@@ -11,7 +12,6 @@ export type NewTableImportRows = {
|
|
|
11
12
|
rows: Record<string, unknown>[];
|
|
12
13
|
keyBySource: Record<string, string>;
|
|
13
14
|
};
|
|
14
|
-
export declare const MAX_BUFFERED_IMPORT_PARSE_BYTES: number;
|
|
15
15
|
export declare function inferRowsFileFormat(path: string): RowsFileFormat;
|
|
16
16
|
export declare function normalizeRowsFormat(value: string | undefined, fallback: RowsFileFormat): RowsFileFormat;
|
|
17
17
|
export declare function parseRowsFileBuffer(buffer: Buffer, format: RowsFileFormat, options?: {
|
|
@@ -30,6 +30,41 @@ export declare function iterateRowsFileStreamBatches(chunks: AsyncIterable<Uint8
|
|
|
30
30
|
batchSize?: number;
|
|
31
31
|
}): AsyncGenerator<Record<string, unknown>[]>;
|
|
32
32
|
export declare function parseRowsText(text: string, format: Exclude<RowsFileFormat, "xlsx">): Record<string, unknown>[];
|
|
33
|
+
/**
|
|
34
|
+
* How many leading bytes of a staged upload we read back to derive its shape.
|
|
35
|
+
*
|
|
36
|
+
* A staged import lives in object storage, so the only thing standing between a
|
|
37
|
+
* 100MB CSV and a preview is how much of it we are willing to pull into a
|
|
38
|
+
* request handler. 2MiB is the same ceiling the copilot tabular preview uses and
|
|
39
|
+
* comfortably covers a header row plus enough rows to infer column types.
|
|
40
|
+
*/
|
|
41
|
+
export declare const MAX_IMPORT_SAMPLE_BYTES: number;
|
|
42
|
+
export type RowsSample = {
|
|
43
|
+
rows: Record<string, unknown>[];
|
|
44
|
+
headers: string[];
|
|
45
|
+
/** True when a partial trailing record was discarded because the sample was cut short. */
|
|
46
|
+
droppedPartialRecord: boolean;
|
|
47
|
+
};
|
|
48
|
+
/**
|
|
49
|
+
* Parse the LEADING SLICE of an import file — the same bytes the worker will
|
|
50
|
+
* later stream, run through the same state machine.
|
|
51
|
+
*
|
|
52
|
+
* `truncated` says the caller handed us a prefix, not a whole file, and it
|
|
53
|
+
* changes exactly one thing: the residue left in the parser when the bytes run
|
|
54
|
+
* out is DISCARDED instead of being flushed as a final record (or raised as an
|
|
55
|
+
* unterminated-quote error). That is the whole subtlety here. The obvious
|
|
56
|
+
* alternative — cut the slice at the last newline and parse that — is wrong for
|
|
57
|
+
* real exports: a quoted free-text column routinely contains literal newlines,
|
|
58
|
+
* so the last `\n` in a 2MiB window is very often INSIDE a quoted field, and
|
|
59
|
+
* cutting there splits one row into two bogus ones. Only the state machine knows
|
|
60
|
+
* whether a newline closed a record, so only it may decide where the sample ends.
|
|
61
|
+
*
|
|
62
|
+
* With `truncated: false` the sample IS the file and behaviour is identical to
|
|
63
|
+
* `parseRowsText`, unterminated-quote error included.
|
|
64
|
+
*/
|
|
65
|
+
export declare function parseRowsSampleText(text: string, format: Exclude<RowsFileFormat, "xlsx">, options: {
|
|
66
|
+
truncated: boolean;
|
|
67
|
+
}): RowsSample;
|
|
33
68
|
export declare function inferImportColumnLabels(rows: Record<string, unknown>[]): string[];
|
|
34
69
|
export declare function normalizeRowsForNewTable(rows: Record<string, unknown>[]): NewTableImportRows;
|
|
35
70
|
export declare function normalizeImportColumnKey(value: string): string;
|
|
@@ -3,7 +3,8 @@ import readXlsxFile from "read-excel-file/node";
|
|
|
3
3
|
import { inferImportColumnDataType, parseDateValueToIso, } from "./column-types.js";
|
|
4
4
|
import { OxygenError } from "./cli-result.js";
|
|
5
5
|
import { makeUniqueIdentifier, toSnakeIdentifier } from "./identifiers.js";
|
|
6
|
-
|
|
6
|
+
import { MAX_BUFFERED_IMPORT_PARSE_BYTES } from "./import-limits.js";
|
|
7
|
+
export { MAX_BUFFERED_IMPORT_PARSE_BYTES } from "./import-limits.js";
|
|
7
8
|
export function inferRowsFileFormat(path) {
|
|
8
9
|
const extension = extname(path).toLowerCase();
|
|
9
10
|
if (extension === ".jsonl" || extension === ".ndjson")
|
|
@@ -87,6 +88,84 @@ export function parseRowsText(text, format) {
|
|
|
87
88
|
}
|
|
88
89
|
return parseCsvRows(text);
|
|
89
90
|
}
|
|
91
|
+
/**
|
|
92
|
+
* How many leading bytes of a staged upload we read back to derive its shape.
|
|
93
|
+
*
|
|
94
|
+
* A staged import lives in object storage, so the only thing standing between a
|
|
95
|
+
* 100MB CSV and a preview is how much of it we are willing to pull into a
|
|
96
|
+
* request handler. 2MiB is the same ceiling the copilot tabular preview uses and
|
|
97
|
+
* comfortably covers a header row plus enough rows to infer column types.
|
|
98
|
+
*/
|
|
99
|
+
export const MAX_IMPORT_SAMPLE_BYTES = 2 * 1024 * 1024;
|
|
100
|
+
/**
|
|
101
|
+
* Parse the LEADING SLICE of an import file — the same bytes the worker will
|
|
102
|
+
* later stream, run through the same state machine.
|
|
103
|
+
*
|
|
104
|
+
* `truncated` says the caller handed us a prefix, not a whole file, and it
|
|
105
|
+
* changes exactly one thing: the residue left in the parser when the bytes run
|
|
106
|
+
* out is DISCARDED instead of being flushed as a final record (or raised as an
|
|
107
|
+
* unterminated-quote error). That is the whole subtlety here. The obvious
|
|
108
|
+
* alternative — cut the slice at the last newline and parse that — is wrong for
|
|
109
|
+
* real exports: a quoted free-text column routinely contains literal newlines,
|
|
110
|
+
* so the last `\n` in a 2MiB window is very often INSIDE a quoted field, and
|
|
111
|
+
* cutting there splits one row into two bogus ones. Only the state machine knows
|
|
112
|
+
* whether a newline closed a record, so only it may decide where the sample ends.
|
|
113
|
+
*
|
|
114
|
+
* With `truncated: false` the sample IS the file and behaviour is identical to
|
|
115
|
+
* `parseRowsText`, unterminated-quote error included.
|
|
116
|
+
*/
|
|
117
|
+
export function parseRowsSampleText(text, format, options) {
|
|
118
|
+
if (format === "csv")
|
|
119
|
+
return parseCsvSample(text, options.truncated);
|
|
120
|
+
if (format === "jsonl") {
|
|
121
|
+
// A truncated JSONL sample ends mid-line; that line is not a document yet.
|
|
122
|
+
const lines = text.split(/\r?\n/);
|
|
123
|
+
const complete = options.truncated ? lines.slice(0, -1) : lines;
|
|
124
|
+
const rows = normalizeRowObjects(complete
|
|
125
|
+
.map((line, index) => ({ line, lineNumber: index + 1 }))
|
|
126
|
+
.filter(({ line }) => line.trim())
|
|
127
|
+
.map(({ line, lineNumber }) => parseJsonImportLine(line, lineNumber)));
|
|
128
|
+
return {
|
|
129
|
+
rows,
|
|
130
|
+
headers: inferImportColumnLabels(rows),
|
|
131
|
+
droppedPartialRecord: options.truncated && lines.length > complete.length,
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
// A JSON array has no record boundary to resume from — half a document is not
|
|
135
|
+
// a document. Callers must refuse a truncated JSON sample before asking.
|
|
136
|
+
if (options.truncated) {
|
|
137
|
+
throw new OxygenError("import_sample_unsupported_format", "A JSON array cannot be previewed from a partial file; it must be parsed whole.", { details: { format }, exitCode: 1 });
|
|
138
|
+
}
|
|
139
|
+
const rows = normalizeRowObjects(parseJsonArray(text));
|
|
140
|
+
return { rows, headers: inferImportColumnLabels(rows), droppedPartialRecord: false };
|
|
141
|
+
}
|
|
142
|
+
function parseCsvSample(text, truncated) {
|
|
143
|
+
const state = {
|
|
144
|
+
header: null,
|
|
145
|
+
batch: [],
|
|
146
|
+
record: [],
|
|
147
|
+
field: "",
|
|
148
|
+
inQuotes: false,
|
|
149
|
+
};
|
|
150
|
+
for (let index = 0; index < text.length; index += 1) {
|
|
151
|
+
const result = applyCsvCharacter(state, text.charAt(index), text.charAt(index + 1));
|
|
152
|
+
if (result.skipNext)
|
|
153
|
+
index += 1;
|
|
154
|
+
if (!result.recordComplete)
|
|
155
|
+
continue;
|
|
156
|
+
appendCsvRecordToBatch(state, finishCsvRecord(state), Number.POSITIVE_INFINITY);
|
|
157
|
+
}
|
|
158
|
+
const hasResidue = state.inQuotes || Boolean(state.field) || state.record.length > 0;
|
|
159
|
+
if (truncated) {
|
|
160
|
+
return { rows: state.batch, headers: state.header ?? [], droppedPartialRecord: hasResidue };
|
|
161
|
+
}
|
|
162
|
+
if (state.inQuotes)
|
|
163
|
+
throw unterminatedCsvError();
|
|
164
|
+
if (state.field || state.record.length > 0) {
|
|
165
|
+
appendCsvRecordToBatch(state, finishCsvRecord(state), Number.POSITIVE_INFINITY);
|
|
166
|
+
}
|
|
167
|
+
return { rows: state.batch, headers: state.header ?? [], droppedPartialRecord: false };
|
|
168
|
+
}
|
|
90
169
|
export function inferImportColumnLabels(rows) {
|
|
91
170
|
const labels = [];
|
|
92
171
|
const seen = new Set();
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Identify an image by its BYTES, never by what the uploader claimed.
|
|
3
|
+
*
|
|
4
|
+
* These files are served back from our own domain to a third party that fetches
|
|
5
|
+
* them unauthenticated, so `Content-Type` cannot come from client input: an
|
|
6
|
+
* attacker who can choose the type of a file served from oxygen-agent.com — the
|
|
7
|
+
* origin holding the session cookie — can serve HTML from it. Sniffing is the
|
|
8
|
+
* whole defence, so it runs both when the file is stored and again when it is
|
|
9
|
+
* served.
|
|
10
|
+
*/
|
|
11
|
+
export type SniffedImageType = "image/png" | "image/jpeg" | "image/webp";
|
|
12
|
+
/**
|
|
13
|
+
* SVG is deliberately absent and must stay absent.
|
|
14
|
+
*
|
|
15
|
+
* An SVG is a document, not a raster image: it can carry <script>, and a viewer
|
|
16
|
+
* who navigates directly to it runs that script on our origin. That is stored
|
|
17
|
+
* XSS against every signed-in user. (The directory-listing logo route can accept
|
|
18
|
+
* SVG precisely because Vercel Blob serves it from a foreign origin, where the
|
|
19
|
+
* same file is inert against us.) GIF is merely unnecessary rather than
|
|
20
|
+
* dangerous; adding it would be two more magic numbers here.
|
|
21
|
+
*/
|
|
22
|
+
export declare function sniffImageContentType(bytes: Uint8Array): SniffedImageType | null;
|
|
23
|
+
export declare function imageExtensionFor(type: SniffedImageType): "png" | "jpg" | "webp";
|
|
24
|
+
/** The filenames the public serve route will answer to. Nothing else is addressable. */
|
|
25
|
+
export declare const INBOX_AVATAR_FILE_NAMES: readonly ["avatar.png", "avatar.jpg", "avatar.webp"];
|
|
26
|
+
export type InboxAvatarFileName = (typeof INBOX_AVATAR_FILE_NAMES)[number];
|
|
27
|
+
export declare function inboxAvatarFileNameFor(type: SniffedImageType): InboxAvatarFileName;
|
|
28
|
+
export declare function contentTypeForInboxAvatarFileName(name: string): SniffedImageType | null;
|
|
29
|
+
/**
|
|
30
|
+
* How many leading bytes are enough to identify any supported format. Every
|
|
31
|
+
* magic number above lives in the first 12.
|
|
32
|
+
*/
|
|
33
|
+
export declare const IMAGE_SNIFF_BYTES = 32;
|
|
34
|
+
/**
|
|
35
|
+
* The upload ceiling. Larger than the 2MB used for directory logos because this
|
|
36
|
+
* is a photo of a person, and phone cameras routinely produce 4–8MB files; a
|
|
37
|
+
* cap that rejects "the photo I just took" is a cap that gets worked around.
|
|
38
|
+
*/
|
|
39
|
+
export declare const MAX_INBOX_AVATAR_BYTES: number;
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Identify an image by its BYTES, never by what the uploader claimed.
|
|
3
|
+
*
|
|
4
|
+
* These files are served back from our own domain to a third party that fetches
|
|
5
|
+
* them unauthenticated, so `Content-Type` cannot come from client input: an
|
|
6
|
+
* attacker who can choose the type of a file served from oxygen-agent.com — the
|
|
7
|
+
* origin holding the session cookie — can serve HTML from it. Sniffing is the
|
|
8
|
+
* whole defence, so it runs both when the file is stored and again when it is
|
|
9
|
+
* served.
|
|
10
|
+
*/
|
|
11
|
+
const PNG_MAGIC = [0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a];
|
|
12
|
+
const JPEG_MAGIC = [0xff, 0xd8, 0xff];
|
|
13
|
+
/**
|
|
14
|
+
* SVG is deliberately absent and must stay absent.
|
|
15
|
+
*
|
|
16
|
+
* An SVG is a document, not a raster image: it can carry <script>, and a viewer
|
|
17
|
+
* who navigates directly to it runs that script on our origin. That is stored
|
|
18
|
+
* XSS against every signed-in user. (The directory-listing logo route can accept
|
|
19
|
+
* SVG precisely because Vercel Blob serves it from a foreign origin, where the
|
|
20
|
+
* same file is inert against us.) GIF is merely unnecessary rather than
|
|
21
|
+
* dangerous; adding it would be two more magic numbers here.
|
|
22
|
+
*/
|
|
23
|
+
export function sniffImageContentType(bytes) {
|
|
24
|
+
if (startsWith(bytes, PNG_MAGIC))
|
|
25
|
+
return "image/png";
|
|
26
|
+
if (startsWith(bytes, JPEG_MAGIC))
|
|
27
|
+
return "image/jpeg";
|
|
28
|
+
// WebP is a RIFF container: "RIFF" .... "WEBP"
|
|
29
|
+
if (bytes.byteLength >= 12
|
|
30
|
+
&& startsWith(bytes, [0x52, 0x49, 0x46, 0x46])
|
|
31
|
+
&& bytes[8] === 0x57
|
|
32
|
+
&& bytes[9] === 0x45
|
|
33
|
+
&& bytes[10] === 0x42
|
|
34
|
+
&& bytes[11] === 0x50) {
|
|
35
|
+
return "image/webp";
|
|
36
|
+
}
|
|
37
|
+
return null;
|
|
38
|
+
}
|
|
39
|
+
export function imageExtensionFor(type) {
|
|
40
|
+
if (type === "image/png")
|
|
41
|
+
return "png";
|
|
42
|
+
if (type === "image/jpeg")
|
|
43
|
+
return "jpg";
|
|
44
|
+
return "webp";
|
|
45
|
+
}
|
|
46
|
+
/** The filenames the public serve route will answer to. Nothing else is addressable. */
|
|
47
|
+
export const INBOX_AVATAR_FILE_NAMES = ["avatar.png", "avatar.jpg", "avatar.webp"];
|
|
48
|
+
export function inboxAvatarFileNameFor(type) {
|
|
49
|
+
return `avatar.${imageExtensionFor(type)}`;
|
|
50
|
+
}
|
|
51
|
+
export function contentTypeForInboxAvatarFileName(name) {
|
|
52
|
+
if (name === "avatar.png")
|
|
53
|
+
return "image/png";
|
|
54
|
+
if (name === "avatar.jpg")
|
|
55
|
+
return "image/jpeg";
|
|
56
|
+
if (name === "avatar.webp")
|
|
57
|
+
return "image/webp";
|
|
58
|
+
return null;
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* How many leading bytes are enough to identify any supported format. Every
|
|
62
|
+
* magic number above lives in the first 12.
|
|
63
|
+
*/
|
|
64
|
+
export const IMAGE_SNIFF_BYTES = 32;
|
|
65
|
+
/**
|
|
66
|
+
* The upload ceiling. Larger than the 2MB used for directory logos because this
|
|
67
|
+
* is a photo of a person, and phone cameras routinely produce 4–8MB files; a
|
|
68
|
+
* cap that rejects "the photo I just took" is a cap that gets worked around.
|
|
69
|
+
*/
|
|
70
|
+
export const MAX_INBOX_AVATAR_BYTES = 8 * 1024 * 1024;
|
|
71
|
+
function startsWith(bytes, magic) {
|
|
72
|
+
if (bytes.byteLength < magic.length)
|
|
73
|
+
return false;
|
|
74
|
+
return magic.every((value, index) => bytes[index] === value);
|
|
75
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Import size ceilings, in a module the BROWSER can import.
|
|
3
|
+
*
|
|
4
|
+
* They would naturally live next to the parser (`file-import.ts`) and the plan
|
|
5
|
+
* matrix (`plan-limits.ts`), but neither is client-safe: file-import pulls in
|
|
6
|
+
* `node:path` and the XLSX reader at module scope, and plan-limits is reached
|
|
7
|
+
* through the shared barrel. The import modal has to know these numbers to
|
|
8
|
+
* refuse an oversize file BEFORE uploading it — which is the whole point, since
|
|
9
|
+
* the failure it replaces was an upload that died with nothing to say. So the
|
|
10
|
+
* constants live here alone, and both of those modules re-export them, keeping
|
|
11
|
+
* one definition rather than a copy that drifts.
|
|
12
|
+
*/
|
|
13
|
+
/**
|
|
14
|
+
* Formats that must be held in memory to parse at all — a JSON array has no
|
|
15
|
+
* record boundary to stream from, and XLSX is a zip container. CSV and JSONL
|
|
16
|
+
* stream, so only these two are capped this low.
|
|
17
|
+
*/
|
|
18
|
+
export declare const MAX_BUFFERED_IMPORT_PARSE_BYTES: number;
|
|
19
|
+
/**
|
|
20
|
+
* The platform's request-body ceiling — infrastructure, not a plan limit.
|
|
21
|
+
*
|
|
22
|
+
* Anything sent through a serverless function is cut off here before any handler
|
|
23
|
+
* runs, which is why a large import uploads straight to object storage instead.
|
|
24
|
+
* Kept slightly under the real ~4.5MB so a body plus its multipart framing fits.
|
|
25
|
+
*/
|
|
26
|
+
export declare const VERCEL_REQUEST_BODY_LIMIT_BYTES: number;
|
|
27
|
+
/** True for the formats that cannot stream and so carry the buffered ceiling. */
|
|
28
|
+
export declare function isBufferedOnlyImportFormat(format: string): boolean;
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Import size ceilings, in a module the BROWSER can import.
|
|
3
|
+
*
|
|
4
|
+
* They would naturally live next to the parser (`file-import.ts`) and the plan
|
|
5
|
+
* matrix (`plan-limits.ts`), but neither is client-safe: file-import pulls in
|
|
6
|
+
* `node:path` and the XLSX reader at module scope, and plan-limits is reached
|
|
7
|
+
* through the shared barrel. The import modal has to know these numbers to
|
|
8
|
+
* refuse an oversize file BEFORE uploading it — which is the whole point, since
|
|
9
|
+
* the failure it replaces was an upload that died with nothing to say. So the
|
|
10
|
+
* constants live here alone, and both of those modules re-export them, keeping
|
|
11
|
+
* one definition rather than a copy that drifts.
|
|
12
|
+
*/
|
|
13
|
+
/**
|
|
14
|
+
* Formats that must be held in memory to parse at all — a JSON array has no
|
|
15
|
+
* record boundary to stream from, and XLSX is a zip container. CSV and JSONL
|
|
16
|
+
* stream, so only these two are capped this low.
|
|
17
|
+
*/
|
|
18
|
+
export const MAX_BUFFERED_IMPORT_PARSE_BYTES = 10 * 1024 * 1024;
|
|
19
|
+
/**
|
|
20
|
+
* The platform's request-body ceiling — infrastructure, not a plan limit.
|
|
21
|
+
*
|
|
22
|
+
* Anything sent through a serverless function is cut off here before any handler
|
|
23
|
+
* runs, which is why a large import uploads straight to object storage instead.
|
|
24
|
+
* Kept slightly under the real ~4.5MB so a body plus its multipart framing fits.
|
|
25
|
+
*/
|
|
26
|
+
export const VERCEL_REQUEST_BODY_LIMIT_BYTES = 4 * 1024 * 1024;
|
|
27
|
+
/** True for the formats that cannot stream and so carry the buffered ceiling. */
|
|
28
|
+
export function isBufferedOnlyImportFormat(format) {
|
|
29
|
+
return format === "json" || format === "xlsx";
|
|
30
|
+
}
|
|
@@ -35,6 +35,7 @@ export * from "./linkedin-post-url.js";
|
|
|
35
35
|
export * from "./linkedin-quota-denial.js";
|
|
36
36
|
export * from "./linkedin-url.js";
|
|
37
37
|
export * from "./linkedin-sequences.js";
|
|
38
|
+
export * from "./member-columns.js";
|
|
38
39
|
export * from "./microsoft-consent-url.js";
|
|
39
40
|
export * from "./networks.js";
|
|
40
41
|
export * from "./recipes.js";
|
|
@@ -46,6 +47,7 @@ export * from "./call-outcomes.js";
|
|
|
46
47
|
export * from "./dial-guardrail-overrides.js";
|
|
47
48
|
export * from "./sequences.js";
|
|
48
49
|
export * from "./suppression-entries.js";
|
|
50
|
+
export * from "./dnc-identities.js";
|
|
49
51
|
export * from "./table-limits.js";
|
|
50
52
|
export * from "./log.js";
|
|
51
53
|
export * from "./axiom-field-budget.js";
|
|
@@ -35,6 +35,7 @@ export * from "./linkedin-post-url.js";
|
|
|
35
35
|
export * from "./linkedin-quota-denial.js";
|
|
36
36
|
export * from "./linkedin-url.js";
|
|
37
37
|
export * from "./linkedin-sequences.js";
|
|
38
|
+
export * from "./member-columns.js";
|
|
38
39
|
export * from "./microsoft-consent-url.js";
|
|
39
40
|
export * from "./networks.js";
|
|
40
41
|
export * from "./recipes.js";
|
|
@@ -46,6 +47,7 @@ export * from "./call-outcomes.js";
|
|
|
46
47
|
export * from "./dial-guardrail-overrides.js";
|
|
47
48
|
export * from "./sequences.js";
|
|
48
49
|
export * from "./suppression-entries.js";
|
|
50
|
+
export * from "./dnc-identities.js";
|
|
49
51
|
export * from "./table-limits.js";
|
|
50
52
|
export * from "./log.js";
|
|
51
53
|
export * from "./axiom-field-budget.js";
|