salesprompter-cli 0.1.82 → 0.1.84
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +71 -10
- package/dist/cli.js +92 -40
- package/dist/company-identity-review.js +78 -0
- package/dist/company-leads.js +170 -16
- package/dist/company-recovery.js +84 -0
- package/dist/research-browser-preference.js +37 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -44,29 +44,90 @@ For headless or automation use, generate a CLI token in the app and run `salespr
|
|
|
44
44
|
|
|
45
45
|
`leads:at-companies` researches Director, Head-of, VP and C-level contacts across functions. It exports a shortlist for review without starting email enrichment or outreach.
|
|
46
46
|
|
|
47
|
-
|
|
47
|
+
Start with company names; no LinkedIn IDs are needed in your brief:
|
|
48
|
+
|
|
49
|
+
```json
|
|
50
|
+
{ "companies": [{ "name": "Your target company", "maxContacts": 20 }] }
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
salesprompter auth:login
|
|
55
|
+
salesprompter browser:connect
|
|
56
|
+
salesprompter leads:at-companies --brief companies.json --out-dir ./research --resolve-companies --all
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
After a successful connection, Chrome is remembered for company research. If names are ambiguous, an interactive terminal shows verified candidate locations, websites and evidence links. Choose a number, then confirm; nothing is preselected. The CLI saves the decision and resumes without editing your brief. Use `--review` to review a saved ambiguity later. Enter 0 or press Enter to leave it unresolved.
|
|
60
|
+
|
|
61
|
+
JSON, quiet and non-interactive runs never prompt. Their `outcome` is `complete`, `incomplete` or `needs_review`; `nextCommand` gives the terminal command for pending collection or identity review. `collectionComplete` refers only to collection, so it can be true while role review remains. Add `--require-complete` for exit code 2 whenever collection or review is unfinished. Without it, exit 0 means the command ran, not that all research is complete. `--no-interactive` disables prompts in a terminal too.
|
|
62
|
+
|
|
63
|
+
Guided review fetches native details for at most ten exact-name candidates. Non-exact, composite or larger identity choices still need independently checked `--review-file` decisions. Missing/mismatched details cannot be approved. Explicit connection flags override the remembered choice; failed Chrome authentication never falls back silently to another identity.
|
|
64
|
+
|
|
65
|
+
The checkpoint, shortlist and review exports refresh after every collected page, so long runs expose saved progress immediately. Company labels with punctuation, such as `(TKMS)`, are escaped safely in search queries.
|
|
48
66
|
|
|
49
67
|
Use `--report-only` to refresh saved reports without contacting LinkedIn. The workspace-bound checkpoint is still checked. `progress` separates unique shortlisted people, review rows, review people, review-only people, pending searches, partial/unknown coverage and unresolved companies. `collectionComplete` is false while any coverage gap remains; `nextActions` explains what is left. Unrestricted geography is labeled explicitly.
|
|
50
68
|
|
|
51
|
-
|
|
69
|
+
### Recover saved research without recollecting
|
|
70
|
+
|
|
71
|
+
Version 0.1.83 separates collection evidence from review decisions. Keep the original brief and output directory:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
# Recheck saved roles and write an actionable recovery.json; no LinkedIn requests.
|
|
75
|
+
salesprompter leads:at-companies --brief companies.json --out-dir ./research --report-only
|
|
76
|
+
# Verify canonical employer names against native company details for selected targets.
|
|
77
|
+
salesprompter leads:at-companies --brief companies.json --out-dir ./research \
|
|
78
|
+
--browser chrome --verify-employers --company 'Example (CH)'
|
|
79
|
+
# Continue partial searches until company shortlist ceilings are filled.
|
|
80
|
+
salesprompter leads:at-companies --brief companies.json --out-dir ./research \
|
|
81
|
+
--browser chrome --continue-partial quota --all
|
|
82
|
+
# Or request all accessible pages for one exact brief company name.
|
|
83
|
+
salesprompter leads:at-companies --brief companies.json --out-dir ./research \
|
|
84
|
+
--browser chrome --continue-partial exhaustive --company 'Example (CH)'
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
`--company` is repeatable and limits network work, not the overall report. `--verify-employers` checks scoped employers with saved candidates and caches their canonical-name evidence. Only equivalent presentation/legal-suffix names (for example, `BioNTech` / `BioNTech SE`) qualify automatically; group, division and subsidiary differences still require review. The exact numeric current-employer ID check always applies.
|
|
88
|
+
|
|
89
|
+
`recovery.json` contains alias proposals, identity-search attempts and candidate evidence, explicit unresolved reasons, and a fingerprint-bound review template. Proposals are **not** approvals. Copy its `reviewFileTemplate` into a private JSON file and add only independently verified decisions:
|
|
90
|
+
|
|
91
|
+
```json
|
|
92
|
+
{
|
|
93
|
+
"fingerprint": "COPY_THE_64_CHARACTER_FINGERPRINT_FROM_RECOVERY_JSON",
|
|
94
|
+
"employerAliases": [
|
|
95
|
+
{ "companyId": "123", "name": "Exact source name", "evidenceUrl": "https://official.example/evidence" }
|
|
96
|
+
],
|
|
97
|
+
"identities": [
|
|
98
|
+
{ "targetName": "Exact name in brief", "companyId": "456", "canonicalName": "Verified source name", "evidenceUrl": "https://official.example/evidence" }
|
|
99
|
+
]
|
|
100
|
+
}
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
salesprompter leads:at-companies --brief companies.json --out-dir ./research \
|
|
105
|
+
--review-file ./verified-decisions.json --report-only
|
|
106
|
+
# Then collect any newly resolved identities through the same checkpoint.
|
|
107
|
+
salesprompter leads:at-companies --brief companies.json --out-dir ./research --browser chrome --all
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
The CLI validates and persists the complete review file as `review-decisions.json`, checks its workspace/brief fingerprint, and rejects conflicting/shared identities. Supplied evidence URLs record human decisions; their contents are not independently verified by `--review-file`. Later runs reuse the decisions. An alias never approves an ambiguous role. For a new collection, `verifiedEmployerAliases` in the company brief remains supported; changing the original targeting brief still requires a new output directory.
|
|
111
|
+
|
|
112
|
+
Partial continuation checkpoints the raw next offset and deduplicates people. Old checkpoints without an offset replay the first page once before advancing. `quota` stops at the shortlist ceiling; `exhaustive` means attempting the accessible 2,500-result window, **not** a promise of exhaustive coverage. Count changes, empty/repeated pages and larger result sets remain explicitly incomplete. Stopped searches require investigation; the CLI does not automatically clear `stoppedReason`. `--all` controls the number of jobs, not these safety limits. Make a private copy of the output directory before comparing old and new role-policy results.
|
|
52
113
|
|
|
53
|
-
For standalone browser research, sign in once to the CLI-owned Chrome profile
|
|
114
|
+
For standalone browser research, sign in once to the CLI-owned Chrome profile:
|
|
54
115
|
|
|
55
116
|
```bash
|
|
56
117
|
salesprompter browser:connect
|
|
57
118
|
salesprompter leads:at-companies --brief companies.json --out-dir ./research \
|
|
58
|
-
--resolve-companies --all
|
|
119
|
+
--resolve-companies --all
|
|
59
120
|
```
|
|
60
121
|
|
|
61
122
|
`browser:connect` opens a separate Chrome window and waits up to ten minutes for your manual Sales Navigator sign-in (change with `--wait <seconds>`). It confirms a successful native authenticated request, then closes its own browser. Google Chrome must be installed. The private profile lives at `~/.config/salesprompter/chrome-research` (or under `SALESPROMPTER_CONFIG_DIR`). Your everyday Chrome profile stays untouched. The control connection uses a private OS pipe, not a debugging port. No cookies or browser storage are read/exported. The CLI keeps only allowlisted native request headers in memory.
|
|
62
123
|
|
|
63
124
|
Research runs reuse that profile in background Chrome, execute the GET requests themselves, and close their own browser when finished. No Codex task, relay worker, extension, or manual request polling is needed. A first sign-in, expired login, or security challenge still requires you: rerun `browser:connect`, then repeat the research command with the same output directory. Rate limits stop immediately without switching accounts. `--browser chrome`, `--browser-relay-port`, and `--wait-for-session` are mutually exclusive. Chrome connects lazily, so completed checkpoints and report-only runs never launch it.
|
|
64
125
|
|
|
65
|
-
The
|
|
126
|
+
`browser:connect` remembers Chrome only after confirming a successful authenticated request. `browser:connect --check` checks an existing sign-in in background and also remembers it; it does not open a sign-in window. The preference is private and scoped to the CLI configuration directory. Explicit `--browser session` selects extension credentials; without a remembered choice, session remains the fallback. Explicit relay/wait flags take precedence over the preference. Credentials load only for pending requests; on an auth failure the session route tries changed credentials for the same identity once. Add `--wait-for-session 300` for a bounded refresh wait. The optional Codex `--browser-relay-port` route still needs its external worker. Salesprompter workspace authentication applies in every mode.
|
|
66
127
|
|
|
67
128
|
Ctrl+C, SIGTERM and terminal hangup stop Chrome research gracefully: completed searches stay saved, the owned browser closes, and the research lock is released before exit. Repeat the same command to resume only pending work. A forced kill or power loss cannot run cleanup; if a stale lock remains, confirm that its process stopped before removing only `.research-lock` as the CLI error explains.
|
|
68
129
|
|
|
69
|
-
Create a JSON brief with verified numeric LinkedIn company IDs. Use
|
|
130
|
+
Create a JSON brief with company names or verified numeric LinkedIn company IDs. Use `--resolve-companies` for names. Each subsidiary must resolve to its own ID; unresolved identities stay in the coverage report and are never guessed.
|
|
70
131
|
|
|
71
132
|
For a name-only list, the executable end-to-end research mode is:
|
|
72
133
|
|
|
@@ -74,7 +135,7 @@ For a name-only list, the executable end-to-end research mode is:
|
|
|
74
135
|
salesprompter leads:at-companies --brief companies.json --out-dir ./research --resolve-companies --all
|
|
75
136
|
```
|
|
76
137
|
|
|
77
|
-
The CLI first collects prepared IDs, then resolves missing IDs through paginated Account Search and immediately collects their contacts. It
|
|
138
|
+
The CLI first collects prepared IDs, then resolves missing IDs through paginated Account Search and immediately collects their contacts. It tries up to three name queries, normalizing only presentation/legal-suffix differences. Automatic resolution requires a unique matching identity in a complete result set of at most 1,000 companies plus matching native company-detail evidence. Fuzzy, shared-ID, oversized, ambiguous and composite parent/subsidiary targets remain unresolved with diagnostics in `recovery.json`. Resolutions and misses are saved in the workspace-bound checkpoint and `company-resolutions.json`, without editing the input brief. Repeat the command to resume. Add `--retry-unresolved --resolve-companies` to intentionally retry saved identity misses without repeating successful work; add `--company` to narrow the retry. `--all` removes the invocation's job-count limit, not LinkedIn pacing or per-function candidate limits. A working LinkedIn session is required for new requests; use `--browser chrome` to eliminate the external browser-worker dependency.
|
|
78
139
|
|
|
79
140
|
```bash
|
|
80
141
|
salesprompter leads:at-companies --brief companies.json --out-dir ./research \
|
|
@@ -102,13 +163,13 @@ The default six functions are Digital Marketing & CRM, Digital Product & UX, Sof
|
|
|
102
163
|
|
|
103
164
|
Contacts must match the current company ID, a senior title and the requested function. Selection alternates between functions, deduplicates profile URLs and never pads a company to its ceiling. The default ceiling is 20; each company can override it. Head-of roles classified as experienced managers are searched, but generic manager titles do not pass the final seniority check.
|
|
104
165
|
|
|
105
|
-
Default role matching includes talent acquisition, business intelligence, digital workplace and German department-head titles. Broad digital/transformation, communications and operational-technology matches go into `review.csv`, not the selected contacts. Mixed-role headlines cannot borrow seniority from an assistant or former role. Current-position evidence overrides a conflicting summary employer or headline.
|
|
166
|
+
Default role matching includes talent acquisition, business intelligence, digital workplace and German department-head titles. Role policy 2 requires software-specific context for Software Development and digital/UX context for Digital Product & UX; generic engineering, physical-vehicle product roles and technical flight/hydraulic data do not automatically qualify. Broad digital/transformation, communications and operational-technology matches go into `review.csv`, not the selected contacts. Mixed-role headlines cannot borrow seniority from an assistant or former role. Current-position evidence overrides a conflicting summary employer or headline. Reports include `rolePolicyVersion`; review rows retain every applicable `reasons` entry as well as the primary `reason`.
|
|
106
167
|
|
|
107
|
-
The export retains the source employer name.
|
|
168
|
+
The export retains the source employer name and accepted alias evidence. Unverified employer-name differences go to review: verify equivalent canonical names with `--verify-employers`, or supply an evidence-backed review decision for non-equivalent names. Parent/subsidiary scope is never inferred.
|
|
108
169
|
|
|
109
170
|
For custom functions, add `reviewTerms` (a subset of `terms`) for discovery terms that should require manual review: `{"name":"IT","terms":["it","digital"],"reviewTerms":["digital"]}`. A specific non-review term in the same senior-role clause can qualify a contact.
|
|
110
171
|
|
|
111
|
-
Outputs are private local `contacts.csv`, `review.csv`, `coverage.json` (including review candidates, rejected/unresolved matches and per-function counts), and `checkpoint.json`. Review counts are candidate/function pairs, not additional unique leads. A completed search batch is not exhaustive coverage:
|
|
172
|
+
Outputs are private local `contacts.csv`, `review.csv`, `coverage.json` (including review candidates, rejected/unresolved matches and per-function counts), `recovery.json`, `company-resolutions.json`, and `checkpoint.json`. Review counts are candidate/function pairs, not additional unique leads. A completed search batch is not exhaustive coverage: ordinary collection stops at `candidatesPerDepartment`; use `--continue-partial` explicitly to advance. The report records LinkedIn's reported total and any shortfall. Default `--max-searches 12` bounds collection jobs and resolution attempts; employer-detail verification is scoped separately with `--company`. Increase job limits explicitly for larger batches. HTTP 429/999 stops the run and preserves saved pages; resume only after cooldown.
|
|
112
173
|
|
|
113
174
|
Live research requires Salesprompter workspace login and a LinkedIn session. `--browser-relay-port` uses the existing signed-in browser relay. Checkpoints are bound to both the brief and workspace; changed criteria require a new output directory. A concurrent run is refused; after a crash, remove only `.research-lock` after confirming the process stopped. This command does not import the shortlist into the workspace, find emails, or create campaigns. Review company identity, role fit and coverage before downstream use.
|
|
114
175
|
|
package/dist/cli.js
CHANGED
|
@@ -16,6 +16,9 @@ import { Command } from "commander";
|
|
|
16
16
|
import { z } from "zod";
|
|
17
17
|
import { AffiliateCopyReviewSchema, renderAffiliateCopyReview } from "./affiliate-copy.js";
|
|
18
18
|
import { CompanyBriefSchema, planCompanySearches, runCompanyResearch } from "./company-leads.js";
|
|
19
|
+
import { CompanyReviewSchema, resolveCompanyIdentity } from "./company-recovery.js";
|
|
20
|
+
import { chooseCompanyIdentity, companyResearchNextCommand, createCompanyReviewPrompt } from "./company-identity-review.js";
|
|
21
|
+
import { rememberResearchBrowser, selectResearchBrowser } from "./research-browser-preference.js";
|
|
19
22
|
import { createLazySessionRecovery, waitForFreshSession, isFreshSessionForIdentity } from "./session-recovery.js";
|
|
20
23
|
import { createChromeResearchBrowser, ChromeResearchInterruptedError } from "./chrome-browser.js";
|
|
21
24
|
import { clearAuthSession, loginWithBrowserConnect, loginWithDeviceFlow, loginWithToken, readAuthSession, requireAuthSession, shouldBypassAuth, verifySession, writeAuthSession } from "./auth.js";
|
|
@@ -15141,10 +15144,13 @@ program
|
|
|
15141
15144
|
.command("browser:connect")
|
|
15142
15145
|
.description("Sign in once to the CLI-owned research Chrome; no external browser worker required.")
|
|
15143
15146
|
.option("--wait <seconds>", "Time allowed for manual LinkedIn sign-in", "600")
|
|
15147
|
+
.option("--check", "Check the saved sign-in in background without opening a sign-in window", false)
|
|
15144
15148
|
.action(async (options) => {
|
|
15145
|
-
const browser = createChromeResearchBrowser({ visible:
|
|
15149
|
+
const browser = createChromeResearchBrowser({ visible: !options.check, loginWaitMs: options.check ? 5000 : z.coerce.number().int().min(1).max(3600).parse(options.wait) * 1000, onProgress: message => process.stderr.write(`${message}\n`) });
|
|
15146
15150
|
try {
|
|
15147
|
-
|
|
15151
|
+
const connection = await browser.connect();
|
|
15152
|
+
await rememberResearchBrowser(getSalesprompterConfigDir(), "chrome");
|
|
15153
|
+
printOutput({ status: "ok", ...connection, defaultResearchBrowser: "chrome" });
|
|
15148
15154
|
}
|
|
15149
15155
|
finally {
|
|
15150
15156
|
await browser.close();
|
|
@@ -15153,19 +15159,32 @@ program
|
|
|
15153
15159
|
program
|
|
15154
15160
|
.command("leads:at-companies")
|
|
15155
15161
|
.description("Find senior contacts at named companies, balanced across functions; export a review shortlist.")
|
|
15156
|
-
.requiredOption("--brief <path>", "JSON company
|
|
15162
|
+
.requiredOption("--brief <path>", "JSON company names or verified IDs, with optional role criteria")
|
|
15157
15163
|
.option("--out-dir <path>", "Private output directory and resume checkpoint", "./company-leads")
|
|
15158
15164
|
.option("--max-searches <number>", "Maximum company/function searches this invocation", "12")
|
|
15159
15165
|
.option("--report-only", "Refresh saved research reports without contacting LinkedIn", false)
|
|
15160
15166
|
.option("--all", "Process all pending company/function searches in this invocation", false)
|
|
15161
15167
|
.option("--resolve-companies", "Resolve unique exact company names and collect their contacts in the same run", false)
|
|
15162
15168
|
.option("--retry-unresolved", "Retry cached identity misses without discarding completed searches (requires --resolve-companies)", false)
|
|
15169
|
+
.option("--review-file <path>", "Apply fingerprint-bound, evidence-backed identity/alias decisions without changing the brief")
|
|
15170
|
+
.option("--review", "Choose unresolved company identities in this terminal, then resume collection", false)
|
|
15171
|
+
.option("--no-interactive", "Never prompt for company review; return needs_review for unresolved choices")
|
|
15172
|
+
.option("--require-complete", "Exit 2 unless collection and review are complete; still write reports", false)
|
|
15173
|
+
.option("--verify-employers", "Verify canonical employer names against LinkedIn company details and save safe aliases", false)
|
|
15174
|
+
.option("--continue-partial <mode>", "Resume saved partial pages: quota or exhaustive (2500-result window)")
|
|
15175
|
+
.option("--company <name>", "Limit recovery to an exact brief company name; repeat for multiple companies", (value, previous) => [...previous, value], [])
|
|
15163
15176
|
.option("--wait-for-session <seconds>", "Wait for a refreshed extension session after an auth failure; never retry rate limits", "0")
|
|
15164
15177
|
.option("--browser-relay-port <number>", "Use the signed-in Codex browser through a loopback relay")
|
|
15165
|
-
.option("--browser <mode>", "
|
|
15178
|
+
.option("--browser <mode>", "Override saved connection: chrome or session; browser:connect remembers chrome")
|
|
15166
15179
|
.option("--dry-run", "Preview targeting and unresolved companies without network calls", false)
|
|
15167
15180
|
.action(async (options) => {
|
|
15168
|
-
const browserMode =
|
|
15181
|
+
const browserMode = await selectResearchBrowser(options, getSalesprompterConfigDir());
|
|
15182
|
+
const canReview = Boolean(process.stdin.isTTY && process.stderr.isTTY && !runtimeOutputOptions.json && !runtimeOutputOptions.quiet && options.interactive);
|
|
15183
|
+
if (options.review && !canReview)
|
|
15184
|
+
throw new Error("--review requires an interactive terminal without --json, --quiet or --no-interactive. Run the displayed nextCommand in a terminal, or use --review-file for an evidence-backed scripted decision.");
|
|
15185
|
+
if (options.review && (options.reportOnly || options.dryRun))
|
|
15186
|
+
throw new Error("--review cannot be combined with --report-only or --dry-run; candidate verification requires live company details.");
|
|
15187
|
+
const interactiveReview = canReview && !options.reportOnly && !options.dryRun && (options.review || options.resolveCompanies);
|
|
15169
15188
|
if (options.retryUnresolved && !options.resolveCompanies)
|
|
15170
15189
|
throw new Error("--retry-unresolved requires --resolve-companies.");
|
|
15171
15190
|
const waitMs = z.coerce.number().int().min(0).max(3600).parse(options.waitForSession) * 1000;
|
|
@@ -15174,19 +15193,39 @@ program
|
|
|
15174
15193
|
if (waitMs && options.browserRelayPort)
|
|
15175
15194
|
throw new Error("--wait-for-session refreshes extension credentials, not a browser relay. Use one connection mode.");
|
|
15176
15195
|
const brief = await readJsonFile(path.resolve(options.brief), CompanyBriefSchema);
|
|
15196
|
+
const review = options.reviewFile ? await readJsonFile(path.resolve(options.reviewFile), CompanyReviewSchema) : undefined;
|
|
15197
|
+
const continuePartial = options.continuePartial ? z.enum(["quota", "exhaustive"]).parse(options.continuePartial) : undefined;
|
|
15198
|
+
for (const name of options.company)
|
|
15199
|
+
if (!brief.companies.some(c => c.name === name))
|
|
15200
|
+
throw new Error(`Unknown target company: ${name}`);
|
|
15177
15201
|
const jobs = planCompanySearches(brief);
|
|
15178
15202
|
const maxSearches = options.all ? Number.MAX_SAFE_INTEGER : z.coerce.number().int().min(1).max(10000).parse(options.maxSearches);
|
|
15179
15203
|
if (options.dryRun) {
|
|
15180
|
-
|
|
15204
|
+
const inScope = (name) => !options.company.length || options.company.includes(name);
|
|
15205
|
+
printOutput({ status: "ok", outcome: "preview", browser: browserMode, dryRun: true, searches: jobs.filter(job => inScope(job.company.name)), unresolvedCompanies: brief.companies.filter(c => !c.companyId && inScope(c.name)), maxSearches, recovery: { companies: options.company, verifyEmployers: options.verifyEmployers, continuePartial: continuePartial ?? null, reviewFile: options.reviewFile ? path.resolve(options.reviewFile) : null, savedCheckpointNotRead: true }, outreachStarted: false, emailEnrichmentStarted: false });
|
|
15181
15206
|
return;
|
|
15182
15207
|
}
|
|
15183
15208
|
const session = await requireAuthSession();
|
|
15184
15209
|
const orgId = session.user.orgId;
|
|
15185
15210
|
if (!orgId)
|
|
15186
15211
|
throw new Error("Choose a Salesprompter workspace before company research.");
|
|
15212
|
+
const finish = (report, reportOnly = false) => {
|
|
15213
|
+
const nextCommand = companyResearchNextCommand({ brief: path.resolve(options.brief), outDir: report.output, browser: browserMode, unresolved: report.progress.unresolvedCompanies, pending: report.progress.remainingSearches, companies: options.company });
|
|
15214
|
+
writeProgress(`\n${report.outcome.replace(/_/g, " ").toUpperCase()}: ${report.selected.length} contacts; ${report.progress.completedSearches}/${report.progress.totalSearches} searches; ${report.progress.reviewOnlyPeople} additional people to review.`);
|
|
15215
|
+
if (nextCommand)
|
|
15216
|
+
writeProgress(`Next: ${nextCommand}`);
|
|
15217
|
+
if (report.review.length)
|
|
15218
|
+
writeProgress(`Role/employer review: ${path.join(report.output, "review.csv")}`);
|
|
15219
|
+
if (process.stdout.isTTY && !runtimeOutputOptions.json)
|
|
15220
|
+
writeProgress(`Contacts: ${path.join(report.output, "contacts.csv")}\nReport: ${path.join(report.output, "coverage.json")}`);
|
|
15221
|
+
else
|
|
15222
|
+
printOutput({ status: report.status, outcome: report.outcome, browser: browserMode, reportOnly, selected: report.selected.length, reviewRequired: report.review.length, progress: report.progress, nextActions: report.nextActions, nextCommand, completedSearches: report.completedSearches, totalSearches: report.totalSearches, coverage: report.coverage, output: report.output, resumable: true, outreachStarted: false, emailEnrichmentStarted: false });
|
|
15223
|
+
if (options.requireComplete && report.outcome !== "complete")
|
|
15224
|
+
process.exitCode = 2;
|
|
15225
|
+
};
|
|
15187
15226
|
if (options.reportOnly) {
|
|
15188
|
-
const report = await runCompanyResearch({ brief, outDir: path.resolve(options.outDir), scope: `${session.apiBaseUrl}:${orgId}`, maxSearches: 0, search: async () => { throw new Error("Report-only cannot collect leads."); } });
|
|
15189
|
-
|
|
15227
|
+
const report = await runCompanyResearch({ brief, review, onlyCompanies: options.company, outDir: path.resolve(options.outDir), scope: `${session.apiBaseUrl}:${orgId}`, maxSearches: 0, search: async () => { throw new Error("Report-only cannot collect leads."); } });
|
|
15228
|
+
finish(report, true);
|
|
15190
15229
|
return;
|
|
15191
15230
|
}
|
|
15192
15231
|
const port = options.browserRelayPort == null ? null : z.coerce.number().int().min(1).max(65535).parse(options.browserRelayPort);
|
|
@@ -15228,41 +15267,54 @@ program
|
|
|
15228
15267
|
},
|
|
15229
15268
|
});
|
|
15230
15269
|
let startedSearches = 0;
|
|
15270
|
+
const paced = async () => { if (startedSearches++ > 0)
|
|
15271
|
+
await delay(randomIntegerBetween(5000, 8000)); };
|
|
15272
|
+
const detailsCache = new Map();
|
|
15273
|
+
const details = async (id) => {
|
|
15274
|
+
if (detailsCache.has(id))
|
|
15275
|
+
return detailsCache.get(id);
|
|
15276
|
+
await paced();
|
|
15277
|
+
process.stderr.write(`Verifying employer ${id}\n`);
|
|
15278
|
+
const response = await recoverSession(config => fetchCliImportSalesNavigatorJson({ url: buildCliImportCompanyDetailsUrl([id]), config, browserRelay: relay, timeoutMs: 30000, label: "Company identity evidence", companyId: id }));
|
|
15279
|
+
const body = cliImportCompanyRecord(response.body?.results);
|
|
15280
|
+
const record = cliImportCompanyRecord(body?.[id]);
|
|
15281
|
+
const explicitId = record?.entityUrn ? extractLinkedInSalesCompanyIdFromUrn(record.entityUrn) : null;
|
|
15282
|
+
const evidence = response.status === 200 && typeof record?.name === "string" && (!explicitId || explicitId === id)
|
|
15283
|
+
? { companyId: id, name: record.name, website: typeof record.website === "string" ? record.website : undefined, location: formatCliImportCompanyLocation(record.headquarters) ?? formatCliImportCompanyLocation(record.location) ?? undefined, evidenceUrl: `https://www.linkedin.com/sales/company/${id}` } : null;
|
|
15284
|
+
detailsCache.set(id, evidence);
|
|
15285
|
+
return evidence;
|
|
15286
|
+
};
|
|
15231
15287
|
try {
|
|
15232
|
-
const report = await runCompanyResearch({ brief, outDir: path.resolve(options.outDir), scope: `${session.apiBaseUrl}:${orgId}`, maxSearches, retryUnresolved: options.retryUnresolved,
|
|
15233
|
-
|
|
15234
|
-
|
|
15235
|
-
|
|
15236
|
-
|
|
15237
|
-
|
|
15238
|
-
|
|
15288
|
+
const report = await runCompanyResearch({ brief, review, onlyCompanies: options.company, continuePartial, outDir: path.resolve(options.outDir), scope: `${session.apiBaseUrl}:${orgId}`, maxSearches, retryUnresolved: options.retryUnresolved,
|
|
15289
|
+
reviewIdentity: !interactiveReview ? undefined : async (company, diagnostic) => {
|
|
15290
|
+
const prompt = createCompanyReviewPrompt();
|
|
15291
|
+
try {
|
|
15292
|
+
return await chooseCompanyIdentity({ targetName: company.name, diagnostic, details, io: prompt.io });
|
|
15293
|
+
}
|
|
15294
|
+
finally {
|
|
15295
|
+
prompt.close();
|
|
15296
|
+
}
|
|
15297
|
+
},
|
|
15298
|
+
verifyEmployer: options.verifyEmployers ? details : undefined,
|
|
15299
|
+
diagnoseCompany: !options.resolveCompanies ? undefined : company => resolveCompanyIdentity(company.name, async (query, start) => {
|
|
15300
|
+
await paced();
|
|
15301
|
+
process.stderr.write(`Resolving ${company.name}: ${query}, offset ${start}\n`);
|
|
15302
|
+
const url = buildCliImportAccountSearchUrl(query).replace(/([?&])start=\d+/, `$1start=${start}`);
|
|
15303
|
+
const response = await recoverSession(config => fetchCliImportSalesNavigatorJson({ url, config, browserRelay: relay, timeoutMs: 30000, label: `Company identity for ${company.name}` }));
|
|
15239
15304
|
if (response.status !== 200 || !response.body)
|
|
15240
15305
|
throw new Error(`Company identity search failed (${response.status}).`);
|
|
15241
15306
|
const elements = extractLocalSalesNavigatorElements(response.body);
|
|
15242
|
-
|
|
15243
|
-
|
|
15244
|
-
|
|
15245
|
-
|
|
15246
|
-
|
|
15247
|
-
|
|
15248
|
-
const
|
|
15249
|
-
|
|
15250
|
-
|
|
15251
|
-
|
|
15252
|
-
|
|
15253
|
-
if (!rows.length)
|
|
15254
|
-
return null;
|
|
15255
|
-
elements.push(...rows);
|
|
15256
|
-
}
|
|
15257
|
-
if (elements.length < total)
|
|
15258
|
-
return null;
|
|
15259
|
-
const accounts = elements.map(element => normalizeLocalSalesNavigatorAccount(element, queryUrl)).filter(Boolean);
|
|
15260
|
-
const exact = accounts.filter(account => normalizeLooseMatchText(String(account.companyName ?? "")) === normalizeLooseMatchText(company.name));
|
|
15261
|
-
const ids = new Set(exact.map(account => String(account.companyId ?? "")).filter(id => /^[1-9]\d*$/.test(id)));
|
|
15262
|
-
if (ids.size !== 1)
|
|
15263
|
-
return null;
|
|
15264
|
-
const companyId = [...ids][0];
|
|
15265
|
-
return { companyId, companyName: company.name, evidenceUrl: `https://www.linkedin.com/sales/company/${companyId}` };
|
|
15307
|
+
return { rawCount: elements.length, total: extractLocalSalesNavigatorTotalResults(response.body), candidates: elements.map(e => normalizeLocalSalesNavigatorAccount(e, url)).filter(a => a && /^[1-9]\d*$/.test(String(a.companyId)) && typeof a.companyName === "string").map(a => ({ companyId: String(a.companyId), name: String(a.companyName), evidenceUrl: `https://www.linkedin.com/sales/company/${a.companyId}` })) };
|
|
15308
|
+
}, details),
|
|
15309
|
+
searchPage: async (job, start, count) => {
|
|
15310
|
+
await paced();
|
|
15311
|
+
process.stderr.write(`Researching ${job.company.name} — ${job.department.name}, offset ${start}\n`);
|
|
15312
|
+
return recoverSession(async (config) => {
|
|
15313
|
+
const request = relay ? { url: buildSalesNavigatorLeadApiUrlFromSearchUrl(job.queryUrl, count), headers: {} } : buildSalesNavigatorApiRequestFromSearchUrl(job.queryUrl, config, count);
|
|
15314
|
+
request.url = withSalesNavigatorPaging(request.url, start, count);
|
|
15315
|
+
const response = relay ? await relay.request(request) : await fetchLocalSalesNavigatorRequest(request, { maxRetries: 0, retryBaseDelayMs: 2000, retryMaxDelayMs: 10000 });
|
|
15316
|
+
return { people: normalizeLocalSalesNavigatorPeople(response.body, request.url), totalResults: extractLocalSalesNavigatorTotalResults(response.body), rawCount: extractLocalSalesNavigatorElements(response.body).length };
|
|
15317
|
+
});
|
|
15266
15318
|
},
|
|
15267
15319
|
search: async (job) => {
|
|
15268
15320
|
if (startedSearches++ > 0)
|
|
@@ -15280,7 +15332,7 @@ program
|
|
|
15280
15332
|
});
|
|
15281
15333
|
},
|
|
15282
15334
|
});
|
|
15283
|
-
|
|
15335
|
+
finish(report);
|
|
15284
15336
|
}
|
|
15285
15337
|
finally {
|
|
15286
15338
|
await relay?.close();
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
import { companyIdentityKey } from "./company-recovery.js";
|
|
2
|
+
import { createInterface } from "node:readline/promises";
|
|
3
|
+
/** Cancelling a terminal question must reject it so research can release its lock. */
|
|
4
|
+
export function createCompanyReviewPrompt(input = process.stdin, output = process.stderr) {
|
|
5
|
+
const rl = createInterface({ input, output, terminal: true });
|
|
6
|
+
const cancellation = new AbortController();
|
|
7
|
+
const cancel = () => cancellation.abort();
|
|
8
|
+
const signals = ["SIGINT", "SIGTERM", "SIGHUP"];
|
|
9
|
+
rl.on("SIGINT", cancel);
|
|
10
|
+
rl.once("close", cancel);
|
|
11
|
+
for (const signal of signals)
|
|
12
|
+
process.on(signal, cancel);
|
|
13
|
+
return {
|
|
14
|
+
io: {
|
|
15
|
+
write: (text) => { output.write(text); },
|
|
16
|
+
ask: async (question) => {
|
|
17
|
+
try {
|
|
18
|
+
return await rl.question(question, { signal: cancellation.signal });
|
|
19
|
+
}
|
|
20
|
+
catch (error) {
|
|
21
|
+
if (cancellation.signal.aborted)
|
|
22
|
+
throw new Error("Company review cancelled. Saved work is preserved; rerun with the same --out-dir to resume.");
|
|
23
|
+
throw error;
|
|
24
|
+
}
|
|
25
|
+
},
|
|
26
|
+
},
|
|
27
|
+
close: () => {
|
|
28
|
+
for (const signal of signals)
|
|
29
|
+
process.off(signal, cancel);
|
|
30
|
+
rl.close();
|
|
31
|
+
},
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
export const terminalText = (text) => text.replace(/[\x00-\x1f\x7f-\x9f\u202a-\u202e\u2066-\u2069]/g, " ").slice(0, 300);
|
|
35
|
+
export async function chooseCompanyIdentity(input) {
|
|
36
|
+
const { targetName, diagnostic, details, io } = input;
|
|
37
|
+
const candidates = [...new Map(diagnostic.candidates.filter(c => /^[1-9]\d*$/.test(c.companyId) && companyIdentityKey(c.name) === companyIdentityKey(targetName)).map(c => [c.companyId, c])).values()];
|
|
38
|
+
io.write(`\nCompany review: ${terminalText(targetName)}\n`);
|
|
39
|
+
if (!candidates.length || candidates.length > 10) {
|
|
40
|
+
io.write("No bounded exact-name choice is available. Inspect recovery.json and use an evidence-backed --review-file; this company remains unresolved.\n");
|
|
41
|
+
return null;
|
|
42
|
+
}
|
|
43
|
+
const verified = [];
|
|
44
|
+
for (const candidate of candidates) {
|
|
45
|
+
const evidence = await details(candidate.companyId);
|
|
46
|
+
if (!evidence || evidence.companyId !== candidate.companyId || evidence.evidenceUrl !== `https://www.linkedin.com/sales/company/${candidate.companyId}` || companyIdentityKey(evidence.name) !== companyIdentityKey(candidate.name))
|
|
47
|
+
continue;
|
|
48
|
+
verified.push(evidence);
|
|
49
|
+
io.write(`\n${verified.length}. ${terminalText(evidence.name)}\n Location: ${terminalText(evidence.location ?? "not provided")}\n Website: ${terminalText(evidence.website ?? "not provided")}\n Evidence: ${terminalText(evidence.evidenceUrl)}\n`);
|
|
50
|
+
}
|
|
51
|
+
if (!verified.length) {
|
|
52
|
+
io.write("No candidate could be verified. Nothing approved.\n");
|
|
53
|
+
return null;
|
|
54
|
+
}
|
|
55
|
+
io.write("\nMatch the location and website to your intended company. No choice is preselected.\n");
|
|
56
|
+
while (true) {
|
|
57
|
+
const answer = (await io.ask("Company number [0 = skip]: ")).trim();
|
|
58
|
+
if (!answer || answer === "0")
|
|
59
|
+
return null;
|
|
60
|
+
if (!/^[1-9]\d*$/.test(answer) || Number(answer) > verified.length) {
|
|
61
|
+
io.write("Enter a listed number, or 0 to skip.\n");
|
|
62
|
+
continue;
|
|
63
|
+
}
|
|
64
|
+
const chosen = verified[Number(answer) - 1];
|
|
65
|
+
const confirm = await io.ask(`Use ${terminalText(chosen.name)} (${terminalText(chosen.location ?? chosen.companyId)}) for ${terminalText(targetName)}? [y/N]: `);
|
|
66
|
+
return /^(y|yes)$/i.test(confirm.trim()) ? chosen : null;
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
export function companyResearchNextCommand(input) {
|
|
70
|
+
const quote = (text) => `'${text.replace(/'/g, "'\\''")}'`;
|
|
71
|
+
const scope = (input.companies ?? []).map(name => ` --company ${quote(name)}`).join("");
|
|
72
|
+
const base = `salesprompter leads:at-companies --brief ${quote(input.brief)} --out-dir ${quote(input.outDir)} --browser ${input.browser}${scope}`;
|
|
73
|
+
if (input.unresolved)
|
|
74
|
+
return `${base} --resolve-companies --review --all`;
|
|
75
|
+
if (input.pending)
|
|
76
|
+
return `${base} --all`;
|
|
77
|
+
return null;
|
|
78
|
+
}
|
package/dist/company-leads.js
CHANGED
|
@@ -3,6 +3,7 @@ import { mkdir, readFile, rename, writeFile, chmod, rmdir } from "node:fs/promis
|
|
|
3
3
|
import path from "node:path";
|
|
4
4
|
import { z } from "zod";
|
|
5
5
|
import { buildSalesNavigatorPeopleSearchUrl } from "./sales-navigator.js";
|
|
6
|
+
import { CompanyReviewSchema, companyIdentityKey } from "./company-recovery.js";
|
|
6
7
|
export const defaultDepartments = [
|
|
7
8
|
{ name: "Digital Marketing & CRM", terms: ["marketing", "crm", "growth", "communication"], reviewTerms: ["communication"] },
|
|
8
9
|
{ name: "Digital Product & UX", terms: ["product", "ux", "design"], reviewTerms: [] },
|
|
@@ -63,7 +64,17 @@ export function classifyCompanyRole(title, department) {
|
|
|
63
64
|
if (!relevant.length)
|
|
64
65
|
return "seniority_not_verified";
|
|
65
66
|
const reviewTerms = new Set(department.reviewTerms.map(words));
|
|
66
|
-
|
|
67
|
+
const contextMatches = (clause) => {
|
|
68
|
+
const t = words(clause);
|
|
69
|
+
if (department.name === "Software Development")
|
|
70
|
+
return /\b(software|cto|developer|developers|devops|frontend|backend|full stack)\b/.test(t);
|
|
71
|
+
if (department.name === "Digital Product & UX")
|
|
72
|
+
return /\b(ux|user experience|digital product|software product|product design|digital design|ui)\b/.test(t) && !/\b(vehicles|mechanical|industrial design|hardware design)\b/.test(t);
|
|
73
|
+
if (department.name === "Data & AI" && /\b(technical data|hydraulic|flight control)\b/.test(t))
|
|
74
|
+
return /\b(analytics|artificial intelligence|machine learning|data science|data engineering)\b/.test(t);
|
|
75
|
+
return true;
|
|
76
|
+
};
|
|
77
|
+
if (relevant.some(clause => contextMatches(clause) && department.terms.some(term => !reviewTerms.has(words(term)) && hasTerm(clause, term))))
|
|
67
78
|
return "match";
|
|
68
79
|
if (relevant.some(clause => department.terms.some(term => hasTerm(clause, term))))
|
|
69
80
|
return "review";
|
|
@@ -123,7 +134,7 @@ export function shortlistCompanyLeads(brief, results) {
|
|
|
123
134
|
const key = `${company.companyId}:${job.department.name}:${profile}`;
|
|
124
135
|
if (!reviewSeen.has(key)) {
|
|
125
136
|
reviewSeen.add(key);
|
|
126
|
-
review.push({ ...row, reason: employerNameDiffers ? "employer_name_differs" : "ambiguous_role" });
|
|
137
|
+
review.push({ ...row, reason: employerNameDiffers ? "employer_name_differs" : "ambiguous_role", reasons: [...(employerNameDiffers ? ["employer_name_differs"] : []), ...(role === "review" ? ["ambiguous_role"] : [])] });
|
|
127
138
|
}
|
|
128
139
|
continue;
|
|
129
140
|
}
|
|
@@ -151,11 +162,11 @@ export function shortlistCompanyLeads(brief, results) {
|
|
|
151
162
|
selected.push(...local);
|
|
152
163
|
coverage.push({ companyName: company.name, companyId: company.companyId ?? null, selected: local.length, reviewRequired: review.filter(p => p.companyId === company.companyId).length, ceiling: company.maxContacts,
|
|
153
164
|
status: !company.companyId ? "unresolved_company" : jobs.some(j => !results[j.key]) ? "pending" : local.length ? "review_ready" : "no_verified_matches",
|
|
154
|
-
departments: jobs.map(j => ({ name: j.department.name, selected: local.filter(p => p.department === j.department.name).length, collected: results[j.key]?.people.length ?? null, reported: results[j.key]?.totalResults ?? null, searchComplete: results[j.key]?.totalResults != null ? results[j.key].people.length >= results[j.key].totalResults : null })) });
|
|
165
|
+
departments: jobs.map(j => ({ name: j.department.name, selected: local.filter(p => p.department === j.department.name).length, collected: results[j.key]?.people.length ?? null, reported: results[j.key]?.totalResults ?? null, stoppedReason: results[j.key]?.stoppedReason ?? null, searchComplete: results[j.key]?.totalResults != null ? !results[j.key].stoppedReason && results[j.key].people.length >= results[j.key].totalResults : null })) });
|
|
155
166
|
}
|
|
156
167
|
const jobs = planCompanySearches(brief);
|
|
157
168
|
const completed = jobs.filter(j => results[j.key]);
|
|
158
|
-
const incomplete = completed.filter(j => results[j.key].totalResults != null && results[j.key].people.length < results[j.key].totalResults);
|
|
169
|
+
const incomplete = completed.filter(j => results[j.key].stoppedReason || results[j.key].totalResults != null && results[j.key].people.length < results[j.key].totalResults);
|
|
159
170
|
const unknown = completed.filter(j => results[j.key].totalResults == null);
|
|
160
171
|
const selectedIds = new Set(selected.map(p => String(p.profileUrl)));
|
|
161
172
|
const reviewIds = new Set(review.map(p => String(p.profileUrl)));
|
|
@@ -171,14 +182,15 @@ export function shortlistCompanyLeads(brief, results) {
|
|
|
171
182
|
};
|
|
172
183
|
const nextActions = [
|
|
173
184
|
...(progress.remainingSearches ? ["Resume the unchanged brief and output directory to collect pending searches."] : []),
|
|
174
|
-
...(unresolved.length ? ["
|
|
175
|
-
...(incomplete.length || unknown.length ? ["
|
|
176
|
-
...(review.length ? ["
|
|
185
|
+
...(unresolved.length ? ["Use --review in a terminal to choose a verified exact-name identity. Inspect recovery.json for non-exact decisions or an intentional --retry-unresolved --resolve-companies retry."] : []),
|
|
186
|
+
...(incomplete.length || unknown.length ? ["Use --continue-partial quota or exhaustive to resume eligible pages; inspect stoppedReason before claiming exhaustive coverage."] : []),
|
|
187
|
+
...(review.length ? ["Use --verify-employers for canonical names, or evidence-backed --review-file aliases. Ambiguous roles remain review-only."] : []),
|
|
177
188
|
];
|
|
178
|
-
|
|
189
|
+
const outcome = unresolved.length ? "needs_review" : !progress.collectionComplete ? "incomplete" : review.length ? "needs_review" : "complete";
|
|
190
|
+
return { outcome, selected, review, rejected, coverage, progress, nextActions, rolePolicyVersion: 2, outreachStarted: false, emailEnrichmentStarted: false };
|
|
179
191
|
}
|
|
180
192
|
export function companyLeadCsv(rows) {
|
|
181
|
-
const columns = ["companyName", "companyId", "fullName", "title", "department", "profileUrl", "location", "sourceQueryUrl", "observedAt", "sourceCompanyName", "reason", "employerAliasEvidence"];
|
|
193
|
+
const columns = ["companyName", "companyId", "fullName", "title", "department", "profileUrl", "location", "sourceQueryUrl", "observedAt", "sourceCompanyName", "reason", "reasons", "employerAliasEvidence"];
|
|
182
194
|
const escape = (value) => { let s = String(value ?? ""); if (/^[=+@\-\t\r]/.test(s))
|
|
183
195
|
s = "'" + s; return `"${s.replace(/"/g, '""')}"`; };
|
|
184
196
|
return [columns.join(","), ...rows.map(r => columns.map(c => escape(r[c])).join(","))].join("\n") + "\n";
|
|
@@ -217,47 +229,189 @@ async function runCompanyResearchLocked(input) {
|
|
|
217
229
|
throw e;
|
|
218
230
|
}
|
|
219
231
|
const save = async (file, value) => { await writeFile(file + ".tmp", value, { mode: 0o600 }); await chmod(file + ".tmp", 0o600); await rename(file + ".tmp", file); };
|
|
232
|
+
let review = input.review ? CompanyReviewSchema.parse(input.review) : undefined;
|
|
233
|
+
if (!review) {
|
|
234
|
+
try {
|
|
235
|
+
review = CompanyReviewSchema.parse(JSON.parse(await readFile(path.join(outDir, "review-decisions.json"), "utf8")));
|
|
236
|
+
}
|
|
237
|
+
catch (e) {
|
|
238
|
+
if (e.code !== "ENOENT")
|
|
239
|
+
throw e;
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
if (review && review.fingerprint !== fingerprint)
|
|
243
|
+
throw new Error("Review decisions belong to a different brief or workspace.");
|
|
244
|
+
for (const identity of review?.identities ?? []) {
|
|
245
|
+
const index = brief.companies.findIndex(c => c.name === identity.targetName);
|
|
246
|
+
if (index < 0 || brief.companies[index].companyId || state.resolutions?.[String(index)] && state.resolutions[String(index)].companyId !== identity.companyId)
|
|
247
|
+
throw new Error("Reviewed identity must target an unresolved company in this brief.");
|
|
248
|
+
const occupied = [...brief.companies.flatMap(c => c.companyId ? [c.companyId] : []), ...Object.entries(state.resolutions ?? {}).filter(([i]) => i !== String(index)).flatMap(([, r]) => r ? [r.companyId] : [])];
|
|
249
|
+
if (occupied.includes(identity.companyId))
|
|
250
|
+
throw new Error("Reviewed identity is already assigned to another target.");
|
|
251
|
+
state.resolutions ??= {};
|
|
252
|
+
state.resolutions[String(index)] = { companyId: identity.companyId, companyName: identity.targetName, canonicalName: identity.canonicalName, evidenceUrl: identity.evidenceUrl };
|
|
253
|
+
}
|
|
220
254
|
const effectiveBrief = () => CompanyBriefSchema.parse({ ...brief, companies: brief.companies.map((company, index) => {
|
|
221
255
|
const resolution = state.resolutions?.[String(index)];
|
|
222
|
-
|
|
256
|
+
const resolved = !company.companyId && resolution ? { ...company, companyId: resolution.companyId } : company;
|
|
257
|
+
const evidence = resolved.companyId ? state.employerEvidence?.[resolved.companyId] : undefined;
|
|
258
|
+
const aliases = [...(resolved.verifiedEmployerAliases ?? []), ...(review?.employerAliases.filter(a => a.companyId === resolved.companyId).map(({ name, evidenceUrl }) => ({ name, evidenceUrl })) ?? [])];
|
|
259
|
+
if (resolution?.canonicalName)
|
|
260
|
+
aliases.push({ name: resolution.canonicalName, evidenceUrl: resolution.evidenceUrl });
|
|
261
|
+
if (evidence && companyIdentityKey(evidence.name) === companyIdentityKey(resolved.name))
|
|
262
|
+
aliases.push({ name: evidence.name, evidenceUrl: evidence.evidenceUrl });
|
|
263
|
+
return { ...resolved, verifiedEmployerAliases: aliases };
|
|
223
264
|
}) });
|
|
224
|
-
const
|
|
265
|
+
const allowed = (name) => !input.onlyCompanies?.length || input.onlyCompanies.includes(name);
|
|
266
|
+
for (const name of input.onlyCompanies ?? [])
|
|
267
|
+
if (!brief.companies.some(c => c.name === name))
|
|
268
|
+
throw new Error(`Unknown target company: ${name}`);
|
|
269
|
+
for (const alias of review?.employerAliases ?? [])
|
|
270
|
+
if (!effectiveBrief().companies.some(c => c.companyId === alias.companyId))
|
|
271
|
+
throw new Error("Reviewed alias targets an unknown company ID.");
|
|
272
|
+
if (input.review) {
|
|
273
|
+
await save(path.join(outDir, "review-decisions.json"), JSON.stringify(review, null, 2));
|
|
274
|
+
await save(checkpoint, JSON.stringify(state));
|
|
275
|
+
}
|
|
276
|
+
const publish = async () => {
|
|
277
|
+
const report = shortlistCompanyLeads(effectiveBrief(), state.results);
|
|
278
|
+
const aliases = new Map();
|
|
279
|
+
for (const row of report.review.filter(r => r.reason === "employer_name_differs")) {
|
|
280
|
+
const id = String(row.companyId), name = String(row.sourceCompanyName);
|
|
281
|
+
aliases.set(`${id}:${name}`, { companyId: id, targetName: String(row.companyName), name, evidenceUrl: `https://www.linkedin.com/sales/company/${id}`, status: "needs_verification" });
|
|
282
|
+
}
|
|
283
|
+
await save(path.join(outDir, "contacts.csv"), companyLeadCsv(report.selected));
|
|
284
|
+
await save(path.join(outDir, "review.csv"), companyLeadCsv(report.review));
|
|
285
|
+
await save(path.join(outDir, "coverage.json"), JSON.stringify(report, null, 2));
|
|
286
|
+
await save(path.join(outDir, "company-resolutions.json"), JSON.stringify(state.resolutions ?? {}, null, 2));
|
|
287
|
+
await save(path.join(outDir, "recovery.json"), JSON.stringify({ fingerprint, aliases: [...aliases.values()], identities: brief.companies.flatMap((c, i) => !c.companyId && !state.resolutions?.[String(i)] ? [{ targetName: c.name, diagnostic: state.diagnostics?.[String(i)] ?? { reason: "legacy_unknown" } }] : []), reviewFileTemplate: { fingerprint, employerAliases: [], identities: [] } }, null, 2));
|
|
288
|
+
return report;
|
|
289
|
+
};
|
|
225
290
|
let performed = 0;
|
|
226
291
|
try {
|
|
292
|
+
if (input.verifyEmployer)
|
|
293
|
+
for (const company of effectiveBrief().companies) {
|
|
294
|
+
if (!company.companyId || !allowed(company.name) || state.employerEvidence?.[company.companyId])
|
|
295
|
+
continue;
|
|
296
|
+
if (!Object.values(state.results).some(result => result.people.some(p => String(p.companyId) === company.companyId)))
|
|
297
|
+
continue;
|
|
298
|
+
const evidence = await input.verifyEmployer(company.companyId);
|
|
299
|
+
if (evidence?.companyId === company.companyId) {
|
|
300
|
+
state.employerEvidence ??= {};
|
|
301
|
+
state.employerEvidence[company.companyId] = evidence;
|
|
302
|
+
await save(checkpoint, JSON.stringify(state));
|
|
303
|
+
await publish();
|
|
304
|
+
}
|
|
305
|
+
}
|
|
227
306
|
// Known IDs run first; an unresolved identity cannot block already prepared work.
|
|
228
307
|
const collect = async () => {
|
|
229
308
|
for (const job of planCompanySearches(effectiveBrief())) {
|
|
230
|
-
if (
|
|
309
|
+
if (!allowed(job.company.name))
|
|
310
|
+
continue;
|
|
311
|
+
const previous = state.results[job.key];
|
|
312
|
+
const defaultBudgetFinished = previous && (previous.nextOffset == null || previous.nextOffset >= brief.candidatesPerDepartment);
|
|
313
|
+
if (previous && (!input.continuePartial && defaultBudgetFinished || previous.totalResults != null && previous.people.length >= previous.totalResults || previous.stoppedReason))
|
|
314
|
+
continue;
|
|
315
|
+
if (previous && input.continuePartial === "quota" && shortlistCompanyLeads(effectiveBrief(), state.results).coverage.find(c => c.companyId === job.company.companyId).selected >= job.company.maxContacts)
|
|
231
316
|
continue;
|
|
232
317
|
if (performed >= input.maxSearches)
|
|
233
318
|
break;
|
|
234
|
-
state.results[job.key] = await input.search(job);
|
|
235
319
|
performed++;
|
|
320
|
+
if (input.searchPage) {
|
|
321
|
+
const result = previous ? { ...previous, people: [...previous.people] } : { people: [], totalResults: null, fetchedPages: 0, nextOffset: 0 };
|
|
322
|
+
// Old checkpoints lack a reliable raw offset; replay the first page once and deduplicate.
|
|
323
|
+
let start = result.nextOffset ?? 0;
|
|
324
|
+
const seen = new Set(result.people.map(p => canonicalProfile(p.profileUrl)));
|
|
325
|
+
const target = input.continuePartial ? 2500 : brief.candidatesPerDepartment;
|
|
326
|
+
while (start < target) {
|
|
327
|
+
const replayingLegacyFirstPage = previous != null && previous.nextOffset == null && start === 0;
|
|
328
|
+
const count = Math.min(100, target - start);
|
|
329
|
+
const page = await input.searchPage(job, start, count);
|
|
330
|
+
result.fetchedPages++;
|
|
331
|
+
if (result.totalResults != null && page.totalResults !== result.totalResults)
|
|
332
|
+
result.stoppedReason = "reported_total_changed";
|
|
333
|
+
result.totalResults ??= page.totalResults;
|
|
334
|
+
let added = 0;
|
|
335
|
+
for (const person of page.people) {
|
|
336
|
+
const key = canonicalProfile(person.profileUrl);
|
|
337
|
+
if (key && !seen.has(key)) {
|
|
338
|
+
seen.add(key);
|
|
339
|
+
result.people.push(person);
|
|
340
|
+
added++;
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
start += page.rawCount;
|
|
344
|
+
result.nextOffset = start;
|
|
345
|
+
if ((page.rawCount === 0 || !added && !replayingLegacyFirstPage) && (result.totalResults == null || result.people.length < result.totalResults))
|
|
346
|
+
result.stoppedReason = "empty_or_duplicate_page";
|
|
347
|
+
state.results[job.key] = result;
|
|
348
|
+
await save(checkpoint, JSON.stringify(state));
|
|
349
|
+
await publish();
|
|
350
|
+
if (result.stoppedReason || result.totalResults != null && start >= result.totalResults)
|
|
351
|
+
break;
|
|
352
|
+
if (input.continuePartial === "quota" && shortlistCompanyLeads(effectiveBrief(), state.results).coverage.find(c => c.companyId === job.company.companyId).selected >= job.company.maxContacts)
|
|
353
|
+
break;
|
|
354
|
+
}
|
|
355
|
+
}
|
|
356
|
+
else
|
|
357
|
+
state.results[job.key] = await input.search(job);
|
|
236
358
|
await save(checkpoint, JSON.stringify(state));
|
|
237
359
|
await publish();
|
|
238
360
|
}
|
|
239
361
|
};
|
|
240
362
|
await collect();
|
|
241
|
-
if (input.resolveCompany) {
|
|
363
|
+
if (input.resolveCompany || input.diagnoseCompany) {
|
|
242
364
|
let resolutionsPerformed = 0;
|
|
243
365
|
for (const [index, company] of brief.companies.entries()) {
|
|
244
366
|
if (performed >= input.maxSearches || resolutionsPerformed >= input.maxSearches)
|
|
245
367
|
break;
|
|
246
368
|
const savedResolution = state.resolutions?.[String(index)];
|
|
247
|
-
if (company.companyId || (Object.hasOwn(state.resolutions ?? {}, String(index)) && !(input.retryUnresolved && savedResolution === null)))
|
|
369
|
+
if (!allowed(company.name) || company.companyId || (Object.hasOwn(state.resolutions ?? {}, String(index)) && !(input.retryUnresolved && savedResolution === null)))
|
|
248
370
|
continue;
|
|
249
|
-
const
|
|
371
|
+
const diagnosed = input.diagnoseCompany ? await input.diagnoseCompany(company) : undefined;
|
|
372
|
+
const resolution = diagnosed ? diagnosed.resolution : await input.resolveCompany(company);
|
|
373
|
+
if (diagnosed) {
|
|
374
|
+
state.diagnostics ??= {};
|
|
375
|
+
state.diagnostics[String(index)] = diagnosed.diagnostic;
|
|
376
|
+
}
|
|
250
377
|
resolutionsPerformed++;
|
|
251
378
|
if (resolution && (!/^[1-9]\d*$/.test(resolution.companyId) || words(resolution.companyName) !== words(company.name)))
|
|
252
379
|
throw new Error("Company resolver returned a non-exact identity.");
|
|
253
380
|
const occupied = new Set(effectiveBrief().companies.map(c => c.companyId).filter(Boolean));
|
|
254
381
|
state.resolutions ??= {};
|
|
255
382
|
state.resolutions[String(index)] = resolution && !occupied.has(resolution.companyId) ? resolution : null;
|
|
383
|
+
if (resolution && occupied.has(resolution.companyId) && state.diagnostics?.[String(index)])
|
|
384
|
+
state.diagnostics[String(index)].reason = "shared_identity";
|
|
256
385
|
await save(checkpoint, JSON.stringify(state));
|
|
257
386
|
await publish();
|
|
258
387
|
await collect();
|
|
259
388
|
}
|
|
260
389
|
}
|
|
390
|
+
if (input.reviewIdentity)
|
|
391
|
+
for (const [index, company] of brief.companies.entries()) {
|
|
392
|
+
if (!allowed(company.name) || company.companyId || state.resolutions?.[String(index)])
|
|
393
|
+
continue;
|
|
394
|
+
const diagnostic = state.diagnostics?.[String(index)];
|
|
395
|
+
if (!diagnostic)
|
|
396
|
+
continue;
|
|
397
|
+
const choice = await input.reviewIdentity(company, diagnostic);
|
|
398
|
+
if (!choice)
|
|
399
|
+
continue;
|
|
400
|
+
if (!diagnostic.candidates.some(c => c.companyId === choice.companyId && companyIdentityKey(c.name) === companyIdentityKey(choice.name)))
|
|
401
|
+
throw new Error("Reviewed identity must match a verified candidate from this research.");
|
|
402
|
+
if (effectiveBrief().companies.some(c => c.companyId === choice.companyId))
|
|
403
|
+
throw new Error("Reviewed identity is already assigned to another target.");
|
|
404
|
+
const identity = { targetName: company.name, companyId: choice.companyId, canonicalName: choice.name, evidenceUrl: choice.evidenceUrl };
|
|
405
|
+
review = CompanyReviewSchema.parse({ fingerprint, employerAliases: review?.employerAliases ?? [], identities: [...(review?.identities ?? []), identity] });
|
|
406
|
+
// The durable review overlay is written first. A crash before the checkpoint
|
|
407
|
+
// write is recovered by the normal review-file replay on the next invocation.
|
|
408
|
+
await save(path.join(outDir, "review-decisions.json"), JSON.stringify(review, null, 2));
|
|
409
|
+
state.resolutions ??= {};
|
|
410
|
+
state.resolutions[String(index)] = { companyId: choice.companyId, companyName: company.name, canonicalName: choice.name, evidenceUrl: choice.evidenceUrl };
|
|
411
|
+
await save(checkpoint, JSON.stringify(state));
|
|
412
|
+
await publish();
|
|
413
|
+
await collect();
|
|
414
|
+
}
|
|
261
415
|
}
|
|
262
416
|
catch (e) {
|
|
263
417
|
await publish();
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
// Only presentation annotations and legal suffixes are interchangeable. Group,
|
|
3
|
+
// division, brand and subsidiary names remain different identities.
|
|
4
|
+
export function companyIdentityKey(name) {
|
|
5
|
+
let key = name.replace(/\s*\((?:CH|DE|AT|UK|US|USA)\)\s*$/i, "")
|
|
6
|
+
.normalize("NFKD").replace(/[\u0300-\u036f]/g, "").replace(/ß/g, "ss").toLowerCase()
|
|
7
|
+
.replace(/&/g, " and ").replace(/[^a-z0-9]+/g, " ")
|
|
8
|
+
.replace(/\s+/g, " ").trim();
|
|
9
|
+
// Do not erase a brand token in names such as AG Insurance or SE Ranking.
|
|
10
|
+
while (/\s(?:gmbh|kgaa|ag|se|ltd|limited|inc|incorporated|llc)$/.test(key))
|
|
11
|
+
key = key.replace(/\s\S+$/, "");
|
|
12
|
+
return key;
|
|
13
|
+
}
|
|
14
|
+
export function companyQueryVariants(name) {
|
|
15
|
+
const plain = name.replace(/\s*\([^)]*\)\s*/g, " ").trim();
|
|
16
|
+
const primary = plain.split(/\s+\/\s+/)[0].trim();
|
|
17
|
+
return [...new Set([name, plain, primary, companyIdentityKey(primary)].filter(Boolean))].slice(0, 3);
|
|
18
|
+
}
|
|
19
|
+
export async function resolveCompanyIdentity(name, search, details) {
|
|
20
|
+
const diagnostic = { reason: "no_exact_match", attempts: [], candidates: [], observedAt: new Date().toISOString() };
|
|
21
|
+
const candidates = new Map();
|
|
22
|
+
const proven = new Map();
|
|
23
|
+
let incomplete = false, oversized = false;
|
|
24
|
+
for (const query of companyQueryVariants(name)) {
|
|
25
|
+
const first = await search(query, 0);
|
|
26
|
+
let count = first.rawCount;
|
|
27
|
+
const total = first.total;
|
|
28
|
+
const queryCandidates = new Map(first.candidates.map(c => [c.companyId, c]));
|
|
29
|
+
let complete = total != null && total <= 1000;
|
|
30
|
+
for (const c of first.candidates)
|
|
31
|
+
candidates.set(c.companyId, c);
|
|
32
|
+
if (complete)
|
|
33
|
+
for (let start = 25; start < total; start += 25) {
|
|
34
|
+
const page = await search(query, start);
|
|
35
|
+
if (page.total !== total || page.rawCount === 0) {
|
|
36
|
+
complete = false;
|
|
37
|
+
break;
|
|
38
|
+
}
|
|
39
|
+
count += page.rawCount;
|
|
40
|
+
for (const c of page.candidates) {
|
|
41
|
+
candidates.set(c.companyId, c);
|
|
42
|
+
queryCandidates.set(c.companyId, c);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
complete = complete && count >= total && queryCandidates.size >= total;
|
|
46
|
+
diagnostic.attempts.push({ query, total, collected: count, complete });
|
|
47
|
+
incomplete ||= !complete;
|
|
48
|
+
oversized ||= total != null && total > 1000;
|
|
49
|
+
// A complete exact query is sufficient; avoid repeating equivalent searches.
|
|
50
|
+
if (complete)
|
|
51
|
+
for (const c of queryCandidates.values())
|
|
52
|
+
proven.set(c.companyId, c);
|
|
53
|
+
if (complete && [...queryCandidates.values()].some(c => companyIdentityKey(c.name) === companyIdentityKey(name))) {
|
|
54
|
+
incomplete = false;
|
|
55
|
+
oversized = false;
|
|
56
|
+
break;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
diagnostic.candidates = [...candidates.values()].slice(0, 100);
|
|
60
|
+
const exact = [...proven.values()].filter(c => companyIdentityKey(c.name) === companyIdentityKey(name));
|
|
61
|
+
if (/\/|\([^)]*(?:\/|,)[^)]*\)/.test(name))
|
|
62
|
+
diagnostic.reason = "entity_decision_required";
|
|
63
|
+
else if (oversized)
|
|
64
|
+
diagnostic.reason = "oversized";
|
|
65
|
+
else if (incomplete)
|
|
66
|
+
diagnostic.reason = "incomplete";
|
|
67
|
+
else if (exact.length > 1)
|
|
68
|
+
diagnostic.reason = "ambiguous";
|
|
69
|
+
else if (exact.length === 1) {
|
|
70
|
+
const evidence = await details(exact[0].companyId);
|
|
71
|
+
if (evidence?.companyId === exact[0].companyId && companyIdentityKey(evidence.name) === companyIdentityKey(name)) {
|
|
72
|
+
diagnostic.reason = "resolved";
|
|
73
|
+
return { resolution: { companyId: evidence.companyId, companyName: name, canonicalName: evidence.name, evidenceUrl: evidence.evidenceUrl }, diagnostic };
|
|
74
|
+
}
|
|
75
|
+
diagnostic.reason = "detail_mismatch";
|
|
76
|
+
}
|
|
77
|
+
return { resolution: null, diagnostic };
|
|
78
|
+
}
|
|
79
|
+
const evidenceUrl = z.url().refine(value => /^https:\/\//i.test(value), "Evidence must use HTTPS");
|
|
80
|
+
export const CompanyReviewSchema = z.object({
|
|
81
|
+
fingerprint: z.string().regex(/^[a-f0-9]{64}$/),
|
|
82
|
+
employerAliases: z.array(z.object({ companyId: z.string().regex(/^[1-9]\d*$/), name: z.string().trim().min(1), evidenceUrl }).strict()).default([]),
|
|
83
|
+
identities: z.array(z.object({ targetName: z.string().trim().min(1), companyId: z.string().regex(/^[1-9]\d*$/), canonicalName: z.string().trim().min(1), evidenceUrl }).strict()).default([]),
|
|
84
|
+
}).strict();
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { randomUUID } from "node:crypto";
|
|
2
|
+
import { mkdir, readFile, rename, rm, writeFile } from "node:fs/promises";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
import { z } from "zod";
|
|
5
|
+
const Preference = z.object({ browser: z.enum(["chrome", "session"]) }).strict();
|
|
6
|
+
export async function readResearchBrowserPreference(configDir) {
|
|
7
|
+
const file = path.join(configDir, "research-browser.json");
|
|
8
|
+
try {
|
|
9
|
+
return Preference.parse(JSON.parse(await readFile(file, "utf8"))).browser;
|
|
10
|
+
}
|
|
11
|
+
catch (error) {
|
|
12
|
+
if (error.code === "ENOENT")
|
|
13
|
+
return "session";
|
|
14
|
+
throw new Error(`Cannot read research browser preference at ${file}. Run browser:connect again, or choose --browser chrome or --browser session explicitly.`);
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
export async function rememberResearchBrowser(configDir, browser) {
|
|
18
|
+
const value = Preference.parse({ browser });
|
|
19
|
+
await mkdir(configDir, { recursive: true, mode: 0o700 });
|
|
20
|
+
const file = path.join(configDir, "research-browser.json");
|
|
21
|
+
const temporary = `${file}.${randomUUID()}.tmp`;
|
|
22
|
+
try {
|
|
23
|
+
await writeFile(temporary, JSON.stringify(value) + "\n", { mode: 0o600, flag: "wx" });
|
|
24
|
+
await rename(temporary, file);
|
|
25
|
+
}
|
|
26
|
+
finally {
|
|
27
|
+
await rm(temporary, { force: true });
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
export async function selectResearchBrowser(options, configDir) {
|
|
31
|
+
if (options.browser !== undefined)
|
|
32
|
+
return Preference.shape.browser.parse(options.browser);
|
|
33
|
+
// An explicit connection route takes precedence over the saved preference.
|
|
34
|
+
if (options.browserRelayPort !== undefined || Number(options.waitForSession ?? 0) > 0)
|
|
35
|
+
return "session";
|
|
36
|
+
return readResearchBrowserPreference(configDir);
|
|
37
|
+
}
|
package/package.json
CHANGED