@reddoorla/maintenance 0.90.1 → 0.92.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{announce-NDTLRQCN.js → announce-6OS4C57D.js} +3 -3
- package/dist/{blux-YGBGS24U.js → blux-NXQLMS4W.js} +4 -4
- package/dist/blux-NXQLMS4W.js.map +1 -0
- package/dist/{bump-deps-KGH5UYSF.js → bump-deps-GI7PHMOC.js} +6 -6
- package/dist/{chunk-WS6NU475.js → chunk-22GKJO5F.js} +2 -2
- package/dist/{chunk-A6T2R63Z.js → chunk-3HS6LPFX.js} +2 -2
- package/dist/{chunk-UXJZU23T.js → chunk-4NUKMPOO.js} +9 -9
- package/dist/{chunk-BQRFQSGY.js → chunk-4R3YAQNV.js} +2 -2
- package/dist/{chunk-JRHIFA5R.js → chunk-5XMKJX74.js} +10 -4
- package/dist/chunk-5XMKJX74.js.map +1 -0
- package/dist/{chunk-XIDUYLSR.js → chunk-7CJ3DVZW.js} +2 -2
- package/dist/{chunk-NYUYNYQL.js → chunk-7MKT4I5T.js} +15 -1
- package/dist/chunk-7MKT4I5T.js.map +1 -0
- package/dist/{chunk-Z3KYFUN6.js → chunk-B6NUZ2ZX.js} +2 -2
- package/dist/chunk-BJQCIJOY.js +3534 -0
- package/dist/chunk-BJQCIJOY.js.map +1 -0
- package/dist/{chunk-Y53WOQIJ.js → chunk-CVJF2IC5.js} +2 -2
- package/dist/{chunk-W45FZEGE.js → chunk-DMPHP7UP.js} +2 -2
- package/dist/{chunk-KGUKL3CV.js → chunk-E22WAIF2.js} +2 -2
- package/dist/{chunk-VBEKL445.js → chunk-FCS36FDT.js} +2 -2
- package/dist/{chunk-QD427NIO.js → chunk-JB4Q3C3M.js} +7 -7
- package/dist/{chunk-IYSEC543.js → chunk-LS6QPANO.js} +2 -2
- package/dist/{chunk-SJOXWJ5Y.js → chunk-M2XF6JAI.js} +80 -9
- package/dist/chunk-M2XF6JAI.js.map +1 -0
- package/dist/{chunk-ZOCJBJQV.js → chunk-MVDMKKBN.js} +2 -2
- package/dist/{chunk-7RJ2HMMJ.js → chunk-NTLSMJHW.js} +2 -2
- package/dist/{chunk-MFHS7IQ2.js → chunk-OCNSIBMP.js} +27 -1
- package/dist/{chunk-MFHS7IQ2.js.map → chunk-OCNSIBMP.js.map} +1 -1
- package/dist/chunk-PJ23RDHT.js +48 -0
- package/dist/chunk-PJ23RDHT.js.map +1 -0
- package/dist/{chunk-2NJQLSEU.js → chunk-PPIZHDJL.js} +2 -2
- package/dist/chunk-T7ZUUOZ2.js +496 -0
- package/dist/chunk-T7ZUUOZ2.js.map +1 -0
- package/dist/{chunk-TVL5MVUF.js → chunk-TCF6S7AD.js} +4 -4
- package/dist/{chunk-GSWTT2J6.js → chunk-WHWVYL7B.js} +4 -4
- package/dist/{chunk-3G25KIWW.js → chunk-ZTPPCHDB.js} +4 -4
- package/dist/cli/bin.js +30 -27
- package/dist/cli/bin.js.map +1 -1
- package/dist/cli/commands/audit.js +14 -14
- package/dist/client-M2V6FHH5.js +11 -0
- package/dist/{convert-to-pnpm-NWB5FCQ2.js → convert-to-pnpm-ONFP3JKF.js} +6 -6
- package/dist/{db-LOSJ74ZJ.js → db-DJVW6ZSE.js} +11 -11
- package/dist/{digest-CS6KLCZB.js → digest-EOA4JLTV.js} +13 -13
- package/dist/{digest-collectors-NI2M3UTC.js → digest-collectors-ZA3BTS3R.js} +7 -7
- package/dist/{email-LJQT5J7F.js → email-I4BCETI4.js} +16 -64
- package/dist/email-I4BCETI4.js.map +1 -0
- package/dist/{ensure-site-FFDUMY6E.js → ensure-site-SF6KICTN.js} +2 -2
- package/dist/forms/index.d.ts +5 -1
- package/dist/forms/index.js +1 -1
- package/dist/forms/index.js.map +1 -1
- package/dist/forms/prismic.d.ts +69 -0
- package/dist/forms/prismic.js +75 -0
- package/dist/forms/prismic.js.map +1 -0
- package/dist/{forms-notify-target-IHJX7NWA.js → forms-notify-target-MNNOGSJ7.js} +5 -4
- package/dist/{forms-notify-target-IHJX7NWA.js.map → forms-notify-target-MNNOGSJ7.js.map} +1 -1
- package/dist/{github-signals-FCLNHKMF.js → github-signals-3WLIJ5RR.js} +5 -5
- package/dist/{header-image-7A7XLVEU.js → header-image-DOCDFFRO.js} +2 -2
- package/dist/{health-endpoint-UVX6R5N5.js → health-endpoint-DHKJUPRM.js} +6 -6
- package/dist/{health-mirror-GMSF3Z5R.js → health-mirror-55RRU6BI.js} +3 -3
- package/dist/index.js +36 -29
- package/dist/index.js.map +1 -1
- package/dist/{init-NXZ4NXTL.js → init-ZRKWYCKA.js} +15 -15
- package/dist/{launch-5V7RLUV7.js → launch-6X5OXPF2.js} +13 -13
- package/dist/migrate-K4JETR36.js +7 -0
- package/dist/{notify-4HP4GZ2H.js → notify-FYSOWKMJ.js} +4 -3
- package/dist/{onboard-BKCYBAVW.js → onboard-6NHGCKGB.js} +6 -6
- package/dist/{orchestrate-6HROYTM6.js → orchestrate-KMQIUP3H.js} +5 -5
- package/dist/pipeline-AQBFG5WM.js +31 -0
- package/dist/{preflight-LIN5NCYU.js → preflight-VPHQYWTY.js} +6 -6
- package/dist/{prismic-ci-6YGRIOHT.js → prismic-ci-A2HI7NNJ.js} +8 -8
- package/dist/{prismic-models-VNGIKS2B.js → prismic-models-QTKIMFBC.js} +5 -5
- package/dist/prospect/types.d.ts +809 -2
- package/dist/prospect/types.js +8 -0
- package/dist/{prospect-audit-JCWTKJTS.js → prospect-audit-RKUVSTLN.js} +27 -6
- package/dist/prospect-audit-RKUVSTLN.js.map +1 -0
- package/dist/{prospect-audits-3PMIO73D.js → prospect-audits-RVYEA4KH.js} +6 -4
- package/dist/recipes/sync-configs.js +3 -3
- package/dist/{chunk-5PWB3JHJ.js → render-7G43BSK5.js} +28 -9
- package/dist/render-7G43BSK5.js.map +1 -0
- package/dist/{replay-4S6ENYCH.js → replay-HV7MGKBU.js} +3 -2
- package/dist/{replay-4S6ENYCH.js.map → replay-HV7MGKBU.js.map} +1 -1
- package/dist/{report-AAY2HZZR.js → report-Y4LXDHRV.js} +11 -11
- package/dist/{report-mirror-GMNSH5HF.js → report-mirror-6ZDCAQZB.js} +3 -3
- package/dist/{schema-75PJECAE.js → schema-2V7BGYVT.js} +1 -1
- package/dist/{schema-75PJECAE.js.map → schema-2V7BGYVT.js.map} +1 -1
- package/dist/{self-updating-VQIW24RZ.js → self-updating-D5XOWS7V.js} +5 -5
- package/dist/{selftest-CUX2FIQD.js → selftest-IFGSJP4X.js} +5 -5
- package/dist/{site-mirror-GCFLBBWK.js → site-mirror-3E7JYTJG.js} +3 -3
- package/dist/{smoke-suite-4JFZZYVW.js → smoke-suite-GXC5JZPH.js} +6 -6
- package/dist/{submissions-7LGJJSDL.js → submissions-UMJS3YST.js} +2 -2
- package/dist/{svelte-codemods-QAPDZC6F.js → svelte-codemods-A3EARSRP.js} +6 -6
- package/dist/{sync-configs-PVRLZRKL.js → sync-configs-KSESOHFF.js} +6 -6
- package/dist/{upgrade-IR2OO3KP.js → upgrade-ZLNVEIEA.js} +6 -6
- package/dist/{webflow-624FEBFD.js → webflow-IRKOIT7S.js} +4 -4
- package/package.json +6 -1
- package/dist/blux-YGBGS24U.js.map +0 -1
- package/dist/chunk-5PWB3JHJ.js.map +0 -1
- package/dist/chunk-DXKF552B.js +0 -1630
- package/dist/chunk-DXKF552B.js.map +0 -1
- package/dist/chunk-JRHIFA5R.js.map +0 -1
- package/dist/chunk-NYUYNYQL.js.map +0 -1
- package/dist/chunk-SJOXWJ5Y.js.map +0 -1
- package/dist/client-JYKPVCJX.js +0 -11
- package/dist/email-LJQT5J7F.js.map +0 -1
- package/dist/migrate-YITCXBLS.js +0 -7
- package/dist/pipeline-MQJD4CKN.js +0 -16
- package/dist/prospect-audit-JCWTKJTS.js.map +0 -1
- package/dist/render-P3SBUFVB.js +0 -10
- package/dist/render-P3SBUFVB.js.map +0 -1
- /package/dist/{announce-NDTLRQCN.js.map → announce-6OS4C57D.js.map} +0 -0
- /package/dist/{bump-deps-KGH5UYSF.js.map → bump-deps-GI7PHMOC.js.map} +0 -0
- /package/dist/{chunk-WS6NU475.js.map → chunk-22GKJO5F.js.map} +0 -0
- /package/dist/{chunk-A6T2R63Z.js.map → chunk-3HS6LPFX.js.map} +0 -0
- /package/dist/{chunk-UXJZU23T.js.map → chunk-4NUKMPOO.js.map} +0 -0
- /package/dist/{chunk-BQRFQSGY.js.map → chunk-4R3YAQNV.js.map} +0 -0
- /package/dist/{chunk-XIDUYLSR.js.map → chunk-7CJ3DVZW.js.map} +0 -0
- /package/dist/{chunk-Z3KYFUN6.js.map → chunk-B6NUZ2ZX.js.map} +0 -0
- /package/dist/{chunk-Y53WOQIJ.js.map → chunk-CVJF2IC5.js.map} +0 -0
- /package/dist/{chunk-W45FZEGE.js.map → chunk-DMPHP7UP.js.map} +0 -0
- /package/dist/{chunk-KGUKL3CV.js.map → chunk-E22WAIF2.js.map} +0 -0
- /package/dist/{chunk-VBEKL445.js.map → chunk-FCS36FDT.js.map} +0 -0
- /package/dist/{chunk-QD427NIO.js.map → chunk-JB4Q3C3M.js.map} +0 -0
- /package/dist/{chunk-IYSEC543.js.map → chunk-LS6QPANO.js.map} +0 -0
- /package/dist/{chunk-ZOCJBJQV.js.map → chunk-MVDMKKBN.js.map} +0 -0
- /package/dist/{chunk-7RJ2HMMJ.js.map → chunk-NTLSMJHW.js.map} +0 -0
- /package/dist/{chunk-2NJQLSEU.js.map → chunk-PPIZHDJL.js.map} +0 -0
- /package/dist/{chunk-TVL5MVUF.js.map → chunk-TCF6S7AD.js.map} +0 -0
- /package/dist/{chunk-GSWTT2J6.js.map → chunk-WHWVYL7B.js.map} +0 -0
- /package/dist/{chunk-3G25KIWW.js.map → chunk-ZTPPCHDB.js.map} +0 -0
- /package/dist/{client-JYKPVCJX.js.map → client-M2V6FHH5.js.map} +0 -0
- /package/dist/{convert-to-pnpm-NWB5FCQ2.js.map → convert-to-pnpm-ONFP3JKF.js.map} +0 -0
- /package/dist/{db-LOSJ74ZJ.js.map → db-DJVW6ZSE.js.map} +0 -0
- /package/dist/{digest-CS6KLCZB.js.map → digest-EOA4JLTV.js.map} +0 -0
- /package/dist/{digest-collectors-NI2M3UTC.js.map → digest-collectors-ZA3BTS3R.js.map} +0 -0
- /package/dist/{ensure-site-FFDUMY6E.js.map → ensure-site-SF6KICTN.js.map} +0 -0
- /package/dist/{github-signals-FCLNHKMF.js.map → github-signals-3WLIJ5RR.js.map} +0 -0
- /package/dist/{header-image-7A7XLVEU.js.map → header-image-DOCDFFRO.js.map} +0 -0
- /package/dist/{health-endpoint-UVX6R5N5.js.map → health-endpoint-DHKJUPRM.js.map} +0 -0
- /package/dist/{health-mirror-GMSF3Z5R.js.map → health-mirror-55RRU6BI.js.map} +0 -0
- /package/dist/{init-NXZ4NXTL.js.map → init-ZRKWYCKA.js.map} +0 -0
- /package/dist/{launch-5V7RLUV7.js.map → launch-6X5OXPF2.js.map} +0 -0
- /package/dist/{migrate-YITCXBLS.js.map → migrate-K4JETR36.js.map} +0 -0
- /package/dist/{notify-4HP4GZ2H.js.map → notify-FYSOWKMJ.js.map} +0 -0
- /package/dist/{onboard-BKCYBAVW.js.map → onboard-6NHGCKGB.js.map} +0 -0
- /package/dist/{orchestrate-6HROYTM6.js.map → orchestrate-KMQIUP3H.js.map} +0 -0
- /package/dist/{pipeline-MQJD4CKN.js.map → pipeline-AQBFG5WM.js.map} +0 -0
- /package/dist/{preflight-LIN5NCYU.js.map → preflight-VPHQYWTY.js.map} +0 -0
- /package/dist/{prismic-ci-6YGRIOHT.js.map → prismic-ci-A2HI7NNJ.js.map} +0 -0
- /package/dist/{prismic-models-VNGIKS2B.js.map → prismic-models-QTKIMFBC.js.map} +0 -0
- /package/dist/{prospect-audits-3PMIO73D.js.map → prospect-audits-RVYEA4KH.js.map} +0 -0
- /package/dist/{report-AAY2HZZR.js.map → report-Y4LXDHRV.js.map} +0 -0
- /package/dist/{report-mirror-GMNSH5HF.js.map → report-mirror-6ZDCAQZB.js.map} +0 -0
- /package/dist/{self-updating-VQIW24RZ.js.map → self-updating-D5XOWS7V.js.map} +0 -0
- /package/dist/{selftest-CUX2FIQD.js.map → selftest-IFGSJP4X.js.map} +0 -0
- /package/dist/{site-mirror-GCFLBBWK.js.map → site-mirror-3E7JYTJG.js.map} +0 -0
- /package/dist/{smoke-suite-4JFZZYVW.js.map → smoke-suite-GXC5JZPH.js.map} +0 -0
- /package/dist/{submissions-7LGJJSDL.js.map → submissions-UMJS3YST.js.map} +0 -0
- /package/dist/{svelte-codemods-QAPDZC6F.js.map → svelte-codemods-A3EARSRP.js.map} +0 -0
- /package/dist/{sync-configs-PVRLZRKL.js.map → sync-configs-KSESOHFF.js.map} +0 -0
- /package/dist/{upgrade-IR2OO3KP.js.map → upgrade-ZLNVEIEA.js.map} +0 -0
- /package/dist/{webflow-624FEBFD.js.map → webflow-IRKOIT7S.js.map} +0 -0
package/dist/prospect/types.d.ts
CHANGED
|
@@ -1,5 +1,654 @@
|
|
|
1
1
|
import { a as AuditResult } from '../types-OIQZ5hWN.js';
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* Who does the engine actually answer this category with?
|
|
5
|
+
*
|
|
6
|
+
* `visibilityScore` alone is close to useless as a report headline: across the
|
|
7
|
+
* 12 audits stored to date it is 0 for eight of them and takes only four
|
|
8
|
+
* distinct values in total, so it cannot rank two thirds of prospects against
|
|
9
|
+
* each other at all. Worse, a bare 0 invites the one question we cannot
|
|
10
|
+
* honestly answer — "how do we make it go up?"
|
|
11
|
+
*
|
|
12
|
+
* What IS honest, and what a prospect can act on, is the shape of the answer
|
|
13
|
+
* they are absent from. Two zeros mean opposite things:
|
|
14
|
+
*
|
|
15
|
+
* - Revogen scores 0, and the engine answers their category with Stryker,
|
|
16
|
+
* Arthrex, Conmed, Globus, NCBI and the FDA. No website edit puts anyone
|
|
17
|
+
* in that answer. The honest advice is to not buy AEO at all.
|
|
18
|
+
* - Beachfront Dentistry scores 0, and the engine answers "dentist in
|
|
19
|
+
* Redondo Beach CA" with Yelp plus five other local practices' own
|
|
20
|
+
* websites — businesses exactly their size. Being in that answer is
|
|
21
|
+
* plainly possible; they simply are not.
|
|
22
|
+
*
|
|
23
|
+
* Same number, opposite counsel. This module computes the evidence that tells
|
|
24
|
+
* them apart. Everything here is arithmetic over citations the engine actually
|
|
25
|
+
* returned — no judgment, no weighting, nothing we chose. A label sits on top
|
|
26
|
+
* of it elsewhere; the numbers below are what the label has to survive.
|
|
27
|
+
*/
|
|
28
|
+
type SourceCount = {
|
|
29
|
+
domain: string;
|
|
30
|
+
/** Total citations, counting repeats within a single answer. */
|
|
31
|
+
count: number;
|
|
32
|
+
/** `count` as a fraction of every category citation, 0..1. */
|
|
33
|
+
share: number;
|
|
34
|
+
};
|
|
35
|
+
type AnswerSpace = {
|
|
36
|
+
/** Category answers that produced at least one citation. Answers that cited
|
|
37
|
+
* nothing are excluded here but still counted in `queriesAsked` — an engine
|
|
38
|
+
* that declined to cite anything is a fact about the query, not about the
|
|
39
|
+
* prospect, and folding it into the shares would dilute them with silence. */
|
|
40
|
+
answersWithCitations: number;
|
|
41
|
+
queriesAsked: number;
|
|
42
|
+
/** Every citation across every category answer, repeats included. */
|
|
43
|
+
citationsTotal: number;
|
|
44
|
+
/** Distinct domains behind those citations. The fragmentation headline: on
|
|
45
|
+
* the benchmark this ran 53–77 per site across five queries.
|
|
46
|
+
*
|
|
47
|
+
* Scales with how many queries ran, so it is NOT comparable between a
|
|
48
|
+
* 3-query run and a 5-query run (the early Reddoor audits sit at 14–38 for
|
|
49
|
+
* that reason alone, not because their categories are less crowded).
|
|
50
|
+
* `medianWidthPerAnswer` is the per-query figure to compare across sites. */
|
|
51
|
+
distinctDomains: number;
|
|
52
|
+
/** Ranked, most-cited first. */
|
|
53
|
+
topSources: SourceCount[];
|
|
54
|
+
/** How many distinct domains it takes to account for half of all citations.
|
|
55
|
+
* This is the number that killed the "get listed in the three directories
|
|
56
|
+
* the engine reads" pitch: on the benchmark it ran 10–18, meaning no such
|
|
57
|
+
* short list exists to buy your way onto. Null when nothing was cited. */
|
|
58
|
+
domainsToHalf: number | null;
|
|
59
|
+
/** Median distinct domains cited within a SINGLE answer — how crowded one
|
|
60
|
+
* reply is, as opposed to the category across five of them. Null when no
|
|
61
|
+
* answer cited anything. */
|
|
62
|
+
medianWidthPerAnswer: number | null;
|
|
63
|
+
/** 1-based position of the prospect's own domain in `topSources`, or null
|
|
64
|
+
* when it was never cited.
|
|
65
|
+
*
|
|
66
|
+
* An earlier read of a 9-site benchmark held that every site scoring above
|
|
67
|
+
* zero was the TOP source in its own category. At 12 sites that breaks:
|
|
68
|
+
* Ludlow Kingsley scores 40 from rank 4. Being cited at all is what tracks
|
|
69
|
+
* the score — which is very nearly a restatement of how the score is
|
|
70
|
+
* computed, so this field is evidence for the reader, not an independent
|
|
71
|
+
* finding. Do not build a claim on rank 1. */
|
|
72
|
+
ownDomainRank: number | null;
|
|
73
|
+
ownDomainCount: number;
|
|
74
|
+
/** The most-cited source that is NOT the prospect, on the queries we ran.
|
|
75
|
+
*
|
|
76
|
+
* ⚠️ This is NOT "who owns your category", and the report must never call
|
|
77
|
+
* it that. Real shares from the benchmark: across the audited sites the
|
|
78
|
+
* biggest rival held 4%, 6% and 8% of citations. At those shares there is
|
|
79
|
+
* no owner —
|
|
80
|
+
* the answer space is fragmented (see `domainsToHalf`, which ran 5–18).
|
|
81
|
+
* What this legitimately answers is narrower and still useful: "on the
|
|
82
|
+
* specific searches we ran, here is who came back instead of you."
|
|
83
|
+
* The character of the whole source list — local practices vs. Stryker and
|
|
84
|
+
* the FDA — is the actual finding; no single row carries it. */
|
|
85
|
+
topRival: SourceCount | null;
|
|
86
|
+
};
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* What is broken, and what is heavy.
|
|
90
|
+
*
|
|
91
|
+
* Lighthouse audits one page and scores it. It does not crawl, so it never sees
|
|
92
|
+
* a link that 404s three pages in, and it reports a performance number without
|
|
93
|
+
* naming the four-megabyte hero image that caused it. Both of those are the
|
|
94
|
+
* actionable half: "your performance is 60" is a grade, "this one photograph is
|
|
95
|
+
* 4.2 MB and it is on every page" is a job.
|
|
96
|
+
*
|
|
97
|
+
* Everything here costs requests to someone else's server, so it is bounded and
|
|
98
|
+
* paced, and the bounds are reported. `linksFound` against `linksChecked` is
|
|
99
|
+
* how the report avoids saying "every link works" when it tested forty of two
|
|
100
|
+
* hundred — the same discipline as `anchorCount` in the extract.
|
|
101
|
+
*
|
|
102
|
+
* SSRF: every URL probed here came out of the prospect's markup, which means an
|
|
103
|
+
* attacker who controls the page controls what we fetch. `isPrivateOrLoopbackHost`
|
|
104
|
+
* is the same guard the crawler applies to its own entry point and redirects,
|
|
105
|
+
* and it is applied here for the same reason.
|
|
106
|
+
*/
|
|
107
|
+
type ProbedUrl = {
|
|
108
|
+
url: string;
|
|
109
|
+
/** HTTP status, or null when the request itself failed. Those are different
|
|
110
|
+
* claims: a 404 is the prospect's broken link, a transport failure might be
|
|
111
|
+
* our network, and only the first belongs in a report as their defect. */
|
|
112
|
+
status: number | null;
|
|
113
|
+
/** Size in bytes, from `content-length`. Null when the server did not say —
|
|
114
|
+
* which is common, and is reported as unknown rather than guessed at. */
|
|
115
|
+
bytes: number | null;
|
|
116
|
+
error: string | null;
|
|
117
|
+
/** Crawled pages that reference it, so a fix has an address. */
|
|
118
|
+
referencedBy: string[];
|
|
119
|
+
};
|
|
120
|
+
/**
|
|
121
|
+
* Why a probe produced no evidence about the prospect's site.
|
|
122
|
+
*
|
|
123
|
+
* Every value here is OUR missing data, never their defect — which is the whole
|
|
124
|
+
* reason the type exists. Grouped rather than free text so a report can count
|
|
125
|
+
* them and a test can assert on them.
|
|
126
|
+
*
|
|
127
|
+
* auth-required 401. The URL exists and is gated; a visitor with an account
|
|
128
|
+
* sees it and we do not.
|
|
129
|
+
* refused 403. Overwhelmingly bot management, not a dead URL — a CDN
|
|
130
|
+
* declining a non-browser client for an image the page paints
|
|
131
|
+
* perfectly well.
|
|
132
|
+
* rate-limited 429. We caused this one, by asking too fast.
|
|
133
|
+
* server-error 5xx. The server was having a bad moment when we asked;
|
|
134
|
+
* saying so as "your link is broken" outlives the moment.
|
|
135
|
+
* no-response The request never got an answer at all — possibly our
|
|
136
|
+
* network, possibly a timeout we set.
|
|
137
|
+
* other Any other non-2xx we are not willing to characterise.
|
|
138
|
+
*/
|
|
139
|
+
type UnverifiedReason = "auth-required" | "refused" | "rate-limited" | "server-error" | "no-response" | "other";
|
|
140
|
+
type UnverifiedGroup = {
|
|
141
|
+
reason: UnverifiedReason;
|
|
142
|
+
count: number;
|
|
143
|
+
/** A sentence a report can print as-is, phrased as our limit rather than
|
|
144
|
+
* their fault. */
|
|
145
|
+
detail: string;
|
|
146
|
+
/** One URL from the group, so a reader can reproduce it themselves. */
|
|
147
|
+
example: string;
|
|
148
|
+
};
|
|
149
|
+
/** The half of the probe budget that came back without evidence. Reported
|
|
150
|
+
* beside the broken lists, never inside them: "we could not check 6 of your
|
|
151
|
+
* 40 links" is an honest sentence, and "6 broken links" would have been a
|
|
152
|
+
* false one. */
|
|
153
|
+
type UnverifiedProbes = {
|
|
154
|
+
count: number;
|
|
155
|
+
groups: UnverifiedGroup[];
|
|
156
|
+
};
|
|
157
|
+
type AssetCheck = {
|
|
158
|
+
/** Internal links that did not resolve to something a visitor can read.
|
|
159
|
+
* Only answers that prove absence — see `classifyProbe`. */
|
|
160
|
+
brokenLinks: ProbedUrl[];
|
|
161
|
+
/** Images that did not load. A broken image is visible to every visitor and
|
|
162
|
+
* is usually a one-line fix, which makes it the cheapest finding here. */
|
|
163
|
+
brokenImages: ProbedUrl[];
|
|
164
|
+
/**
|
|
165
|
+
* Links we asked about and learned nothing from.
|
|
166
|
+
*
|
|
167
|
+
* Optional because this type also describes runs deserialized from
|
|
168
|
+
* `prospect_audits.result_json`, and every report stored before this field
|
|
169
|
+
* existed lacks it — and lacks it for the worst reason: in those reports the
|
|
170
|
+
* unverifiable answers are sitting in `brokenLinks`. `checkAssets` always
|
|
171
|
+
* sets it; a reader must treat absence as "not measured", never as "nothing
|
|
172
|
+
* went unverified".
|
|
173
|
+
*/
|
|
174
|
+
linksUnverified?: UnverifiedProbes;
|
|
175
|
+
/** Images we asked about and learned nothing from. Optional for the same
|
|
176
|
+
* reason as `linksUnverified`, and with the same reading of absence. */
|
|
177
|
+
imagesUnverified?: UnverifiedProbes;
|
|
178
|
+
/** Heaviest images first, capped — the ones worth naming. */
|
|
179
|
+
heaviestImages: ProbedUrl[];
|
|
180
|
+
/** Summed `content-length` of every image we got a size for. Null when no
|
|
181
|
+
* image reported one, so the report never prints "0 MB of images" for a
|
|
182
|
+
* page full of pictures whose server is quiet about sizes. */
|
|
183
|
+
imageBytesMeasured: number | null;
|
|
184
|
+
/** How many images contributed to that sum, against how many were checked —
|
|
185
|
+
* the honest denominator for it. */
|
|
186
|
+
imagesWithKnownSize: number;
|
|
187
|
+
linksFound: number;
|
|
188
|
+
linksChecked: number;
|
|
189
|
+
imagesFound: number;
|
|
190
|
+
imagesChecked: number;
|
|
191
|
+
};
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* What one AI crawler gets when it asks for the homepage, next to what a
|
|
195
|
+
* browser gets.
|
|
196
|
+
*
|
|
197
|
+
* `blocked` is always a COMPARISON — a site that is down answers everyone
|
|
198
|
+
* badly, and that is an outage, not a crawler policy — and it is now also
|
|
199
|
+
* always CONFIRMED: two requests had to agree before this is true. The finding
|
|
200
|
+
* it supports is "this agent is served something a browser is not", which is
|
|
201
|
+
* what we actually observed. It is not "the site blocks this vendor": bot
|
|
202
|
+
* management keys on IP reputation, geography and rate as well as the header,
|
|
203
|
+
* none of which we can characterise from here.
|
|
204
|
+
*/
|
|
205
|
+
type CrawlerReach = {
|
|
206
|
+
agent: string;
|
|
207
|
+
/** The last status we saw, or null when no request got an answer. */
|
|
208
|
+
status: number | null;
|
|
209
|
+
/** Confirmed served differently from a browser. Only ever true when
|
|
210
|
+
* `measured` is true. */
|
|
211
|
+
blocked: boolean;
|
|
212
|
+
/**
|
|
213
|
+
* Do we have evidence either way about this agent?
|
|
214
|
+
*
|
|
215
|
+
* False for a transient answer (429, 5xx), a failed request, or two requests
|
|
216
|
+
* that disagreed. Each of those is OUR missing data — most obviously the 429,
|
|
217
|
+
* which our own request rate causes — and a report must say "not measured"
|
|
218
|
+
* rather than name a vendor on the strength of it.
|
|
219
|
+
*
|
|
220
|
+
* Optional because this type also describes runs deserialized from
|
|
221
|
+
* `prospect_audits.result_json`, and every report stored before this field
|
|
222
|
+
* existed lacks it — those `blocked` values came from a single unpaced
|
|
223
|
+
* sample. `checkCrawlerReach` always sets it; a reader must treat absence as
|
|
224
|
+
* "we do not know how this was judged".
|
|
225
|
+
*/
|
|
226
|
+
measured?: boolean;
|
|
227
|
+
/** Why we could not judge it, in words a report can print. Null when
|
|
228
|
+
* `measured` is true, absent on reports stored before it existed. */
|
|
229
|
+
unverifiedReason?: string | null;
|
|
230
|
+
error: string | null;
|
|
231
|
+
};
|
|
232
|
+
type CrawlerReachability = {
|
|
233
|
+
/** False when the browser control itself failed or came back 4xx/5xx, so
|
|
234
|
+
* nothing below can be attributed to crawler policy and the report must say
|
|
235
|
+
* "not measured". */
|
|
236
|
+
measured: boolean;
|
|
237
|
+
browserStatus: number | null;
|
|
238
|
+
agents: CrawlerReach[];
|
|
239
|
+
/** Agents confirmed to be served something a browser is not. The finding —
|
|
240
|
+
* and it is a description of what we saw, never an accusation about intent. */
|
|
241
|
+
blocked: string[];
|
|
242
|
+
/** Agents we asked about and learned nothing from. Neither reachable nor
|
|
243
|
+
* blocked: not measured, and reported as such so the report's denominator is
|
|
244
|
+
* honest. Optional for reports stored before it existed, where absence means
|
|
245
|
+
* "not measured" rather than "nothing went unverified". */
|
|
246
|
+
unverified?: string[];
|
|
247
|
+
};
|
|
248
|
+
/** One reachability answer. `measured: false` means the request failed for a
|
|
249
|
+
* reason that is ours or the network's, not the site's — and a report must say
|
|
250
|
+
* "not measured" rather than convert our own failure into their defect. */
|
|
251
|
+
type Reachability = {
|
|
252
|
+
measured: boolean;
|
|
253
|
+
/** What was requested, so the finding has an address the reader can try. */
|
|
254
|
+
url: string;
|
|
255
|
+
ok: boolean;
|
|
256
|
+
/** Where it landed, when it landed anywhere. */
|
|
257
|
+
landedOn: string | null;
|
|
258
|
+
error: string | null;
|
|
259
|
+
};
|
|
260
|
+
type BasicsCheck = {
|
|
261
|
+
/**
|
|
262
|
+
* Typing the address without `https://`. Browsers still default to plain
|
|
263
|
+
* http for a bare hostname in plenty of places — a pasted link in a text
|
|
264
|
+
* message, an old bookmark, a printed card — and a site that does not
|
|
265
|
+
* redirect either shows a "Not secure" warning or does not answer at all.
|
|
266
|
+
*/
|
|
267
|
+
insecureEntry: Reachability;
|
|
268
|
+
/**
|
|
269
|
+
* The other of www / apex. Whichever one the site does not use, somebody
|
|
270
|
+
* types anyway; if it has no DNS record the visitor gets a browser error page
|
|
271
|
+
* with the site's name on it.
|
|
272
|
+
*/
|
|
273
|
+
hostVariant: Reachability & {
|
|
274
|
+
host: string;
|
|
275
|
+
};
|
|
276
|
+
/**
|
|
277
|
+
* A URL that cannot exist. Two things can be wrong: the server answers 200
|
|
278
|
+
* (a "soft 404" — search engines index the junk and a mistyped link looks
|
|
279
|
+
* like a real page), or it answers 404 with a bare server error page carrying
|
|
280
|
+
* no way back into the site.
|
|
281
|
+
*/
|
|
282
|
+
notFound: Reachability & {
|
|
283
|
+
status: number | null;
|
|
284
|
+
/** Did the response carry any link back into the site? A default nginx or
|
|
285
|
+
* Apache error page carries none, and a visitor who lands there leaves. */
|
|
286
|
+
linksBackToSite: boolean;
|
|
287
|
+
};
|
|
288
|
+
/**
|
|
289
|
+
* Images loaded over plain http on an https page. Browsers block or refuse to
|
|
290
|
+
* upgrade these, so they are broken images for some visitors and a mixed-
|
|
291
|
+
* content warning for the rest. `measured` is false when the site is not on
|
|
292
|
+
* https at all, in which case `insecureEntry` is the finding instead.
|
|
293
|
+
*/
|
|
294
|
+
mixedContent: {
|
|
295
|
+
measured: boolean;
|
|
296
|
+
imageUrls: string[];
|
|
297
|
+
imagesSeen: number;
|
|
298
|
+
};
|
|
299
|
+
/** Alt text coverage across the pages examined. An image counted once per page
|
|
300
|
+
* it appears on — the ratio is the honest reading, not the totals. */
|
|
301
|
+
altText: {
|
|
302
|
+
imagesTotal: number;
|
|
303
|
+
imagesWithAlt: number;
|
|
304
|
+
pagesExamined: number;
|
|
305
|
+
};
|
|
306
|
+
/** Titles used by more than one page. The browser tab, the bookmark and the
|
|
307
|
+
* search result all show this text, and repeating it makes them
|
|
308
|
+
* indistinguishable. */
|
|
309
|
+
duplicateTitles: {
|
|
310
|
+
title: string;
|
|
311
|
+
pages: string[];
|
|
312
|
+
}[];
|
|
313
|
+
/**
|
|
314
|
+
* What each AI crawler is actually served, as opposed to what robots.txt says
|
|
315
|
+
* it may have.
|
|
316
|
+
*
|
|
317
|
+
* These are different questions and the audit used to answer only the first
|
|
318
|
+
* while reporting the second. robots.txt is a request the site publishes; a
|
|
319
|
+
* CDN's bot management is an answer it enforces, and the second can contradict
|
|
320
|
+
* the first without the owner knowing. Verified on a live prospect: see the
|
|
321
|
+
* note on CRAWLER_AGENTS.
|
|
322
|
+
*
|
|
323
|
+
* Undefined on reports stored before this existed — absence is "not measured",
|
|
324
|
+
* never "nothing blocked".
|
|
325
|
+
*/
|
|
326
|
+
crawlerReachability?: CrawlerReachability;
|
|
327
|
+
};
|
|
328
|
+
|
|
329
|
+
/**
|
|
330
|
+
* Does the site tell the same story on every page?
|
|
331
|
+
*
|
|
332
|
+
* Two findings live here, and they share a property that makes them worth
|
|
333
|
+
* checking together: each is cheap to fix, embarrassing to leave, and invisible
|
|
334
|
+
* from any single page. You only see them by comparing pages, which is exactly
|
|
335
|
+
* what nobody does when they look at their own site.
|
|
336
|
+
*
|
|
337
|
+
* - A stale copyright year. It says nobody has touched this in years, to
|
|
338
|
+
* every visitor, on every page, for free.
|
|
339
|
+
* - Pages that do not share the site's navigation. Usually a landing page
|
|
340
|
+
* built outside the template — a visitor who lands there is in a different
|
|
341
|
+
* website with no way back into this one.
|
|
342
|
+
*
|
|
343
|
+
* The contact numbers and addresses are an INVENTORY, not a third finding, and
|
|
344
|
+
* that is a correction rather than an omission. This module used to treat more
|
|
345
|
+
* than one phone number as "a business that disagrees with itself". Replayed
|
|
346
|
+
* over every stored audit, that rule failed 8 of 22 sites and every single hit
|
|
347
|
+
* was legitimate: our own site's labelled California and Texas office lines, a
|
|
348
|
+
* nonprofit listing thirteen partner helplines on a resources page, a company's
|
|
349
|
+
* fax line, a firm publishing separate general and business-inquiry numbers. A
|
|
350
|
+
* business with two numbers is not confused; it has two numbers. The count was
|
|
351
|
+
* correct data and the claim built on it was false, which is the worst of both
|
|
352
|
+
* — and it was refutable by the reader from their own contact page.
|
|
353
|
+
*
|
|
354
|
+
* What survives is real and actionable: WHICH numbers appear, on which pages,
|
|
355
|
+
* and whether each was ever written as a `tel:` link (see `linked`). Consumers
|
|
356
|
+
* must render the list as a receipt. Any future "you have too many numbers"
|
|
357
|
+
* finding needs a way to tell a second office from a contradiction first, and
|
|
358
|
+
* this data does not carry one.
|
|
359
|
+
*
|
|
360
|
+
* Deliberately NOT checked: the postal address. Addresses cannot be pulled out
|
|
361
|
+
* of free text reliably enough to accuse someone of inconsistency, and a false
|
|
362
|
+
* positive here would have a prospect checking a page that is perfectly fine.
|
|
363
|
+
* When a site publishes a `PostalAddress` in its schema we already read it; a
|
|
364
|
+
* text scrape would be a guess wearing a finding's clothes.
|
|
365
|
+
*/
|
|
366
|
+
type ContactVariant = {
|
|
367
|
+
/** Digits only for a phone, lower-cased for an email — what makes two
|
|
368
|
+
* spellings of the same thing compare equal. */
|
|
369
|
+
normalized: string;
|
|
370
|
+
/** Every spelling actually seen, so the report shows the receipts rather
|
|
371
|
+
* than asserting a mismatch the reader cannot check. */
|
|
372
|
+
seenAs: string[];
|
|
373
|
+
pages: string[];
|
|
374
|
+
/**
|
|
375
|
+
* Was it ever written as a `tel:` / `mailto:` link, anywhere on the site?
|
|
376
|
+
*
|
|
377
|
+
* False means the number exists only as prose. On a phone — which is where
|
|
378
|
+
* most people read a number and where the intent to call is highest — that is
|
|
379
|
+
* a piece of text you cannot tap, and the visitor has to memorise it and
|
|
380
|
+
* switch apps. It is a one-attribute fix, and it is invisible from a desktop,
|
|
381
|
+
* which is exactly where nobody looks.
|
|
382
|
+
*
|
|
383
|
+
* Optional: reports stored before this was recorded lack it, and a reader must
|
|
384
|
+
* treat its absence as "not measured" rather than as "not a link".
|
|
385
|
+
*/
|
|
386
|
+
linked?: boolean;
|
|
387
|
+
};
|
|
388
|
+
type ConsistencyResult = {
|
|
389
|
+
phones: ContactVariant[];
|
|
390
|
+
emails: ContactVariant[];
|
|
391
|
+
/** Every copyright year found in page text, ascending. Empty when the site
|
|
392
|
+
* publishes no copyright line at all, which is not a defect. */
|
|
393
|
+
copyrightYears: number[];
|
|
394
|
+
/** The newest year found, or null when none was. */
|
|
395
|
+
newestCopyrightYear: number | null;
|
|
396
|
+
/** Pages carrying none of the site's shared navigation links. Empty when
|
|
397
|
+
* there is no shared navigation to compare against — see `sharedNavLinks`. */
|
|
398
|
+
pagesOffTemplate: string[];
|
|
399
|
+
/** How many links appear on EVERY page examined. This is the site's shared
|
|
400
|
+
* navigation, derived rather than assumed: no `<nav>` element is required,
|
|
401
|
+
* because plenty of sites do not use one. */
|
|
402
|
+
sharedNavLinks: number;
|
|
403
|
+
pagesExamined: number;
|
|
404
|
+
};
|
|
405
|
+
|
|
406
|
+
/**
|
|
407
|
+
* Is this cited domain someone else's, or another one of the prospect's own?
|
|
408
|
+
*
|
|
409
|
+
* Built after getting it wrong. The accuracy stage read every cited domain that
|
|
410
|
+
* was not the prospect's as "somewhere else the engine looked", and on the first
|
|
411
|
+
* real run that produced the finding "an AI describes your practice using
|
|
412
|
+
* dochopkins.com, cited four times against one citation of your own site" —
|
|
413
|
+
* which was true except for the part that mattered: dochopkins.com is theirs
|
|
414
|
+
* too, an old site they never took down.
|
|
415
|
+
*
|
|
416
|
+
* Getting this backwards is expensive in both directions. Calling a client's own
|
|
417
|
+
* legacy site "a third party" is a factual error about their business, in a
|
|
418
|
+
* document arguing that factual errors about their business are the problem.
|
|
419
|
+
* And the corrected finding is the better one anyway: an engine preferring your
|
|
420
|
+
* old site to your current one is a concrete thing to fix, where "a directory
|
|
421
|
+
* outranks you" often is not.
|
|
422
|
+
*
|
|
423
|
+
* A shared phone number is the signal. Businesses change domains, names and
|
|
424
|
+
* copy; the number on the door tends to survive all three, and two sites
|
|
425
|
+
* publishing the same number are almost never unrelated. A redirect onto the
|
|
426
|
+
* prospect's own host settles it outright.
|
|
427
|
+
*/
|
|
428
|
+
/**
|
|
429
|
+
* Four answers, because three of them lead somewhere different.
|
|
430
|
+
*
|
|
431
|
+
* `yours` — your site, or another domain you own. Fixable by you today.
|
|
432
|
+
* `platform` — a directory, review site or social profile. A listing ABOUT you
|
|
433
|
+
* that you can usually claim and correct, which is a different job from
|
|
434
|
+
* writing a page and worth naming separately.
|
|
435
|
+
* `theirs` — no connection to the prospect that we could find.
|
|
436
|
+
*
|
|
437
|
+
* Worded deliberately as an absence of evidence, because that is all this
|
|
438
|
+
* module establishes: no shared phone number, and no redirect home. It is
|
|
439
|
+
* NOT a finding that the domain belongs to somebody else — asserting that
|
|
440
|
+
* about a third party is a claim we cannot check, in a report whose whole
|
|
441
|
+
* argument is that unchecked claims are the problem. The user-facing string
|
|
442
|
+
* lives in `because` and says so; the variant name is kept because
|
|
443
|
+
* `DomainVerdict` already crosses a module boundary into accuracy.ts, and
|
|
444
|
+
* nothing renders the name itself.
|
|
445
|
+
* `unknown` — we could not tell, and say so.
|
|
446
|
+
*/
|
|
447
|
+
type DomainOwner = "yours" | "platform" | "theirs" | "unknown";
|
|
448
|
+
type DomainVerdict = {
|
|
449
|
+
domain: string;
|
|
450
|
+
owner: DomainOwner;
|
|
451
|
+
/** What decided it, in words a client can check. */
|
|
452
|
+
because: string;
|
|
453
|
+
};
|
|
454
|
+
|
|
455
|
+
type AssertionVerdict = "confirmed" | "contradicted" | "absent" | "unverified";
|
|
456
|
+
type Assertion = {
|
|
457
|
+
/** The statement, in plain words. */
|
|
458
|
+
claim: string;
|
|
459
|
+
verdict: AssertionVerdict;
|
|
460
|
+
/** Verbatim from the engine's answer — verified as a real substring of it. */
|
|
461
|
+
engineQuote: string;
|
|
462
|
+
/** Verbatim from the prospect's own site, verified. Null when there is none. */
|
|
463
|
+
siteQuote: string | null;
|
|
464
|
+
/** Why we could not judge it. Null unless the verdict is `unverified`. */
|
|
465
|
+
unverifiedReason: string | null;
|
|
466
|
+
/**
|
|
467
|
+
* Something related the site DOES say, on an assertion we are still calling
|
|
468
|
+
* absent — the obvious objection, answered before it is raised.
|
|
469
|
+
*
|
|
470
|
+
* The case this exists for, from a real run: an engine said a practice was
|
|
471
|
+
* "formerly known as <a person's name> DDS"; the site never says that, but
|
|
472
|
+
* its team page lists a clinician with a similar name. Suppressing the
|
|
473
|
+
* finding loses the most valuable line in the section. Printing it bare
|
|
474
|
+
* invites a client to open their own team page and conclude we cannot read.
|
|
475
|
+
* Printing it with "your site mentions <the name it does list>, but not
|
|
476
|
+
* this" keeps the finding and shows the work.
|
|
477
|
+
*/
|
|
478
|
+
nearbyMention: string | null;
|
|
479
|
+
/** Domains the engine cited on the answer this came from, excluding the
|
|
480
|
+
* prospect's own — who it was reading instead. */
|
|
481
|
+
sourceDomains: string[];
|
|
482
|
+
/** Which branded query produced it. */
|
|
483
|
+
query: string;
|
|
484
|
+
engine: string;
|
|
485
|
+
};
|
|
486
|
+
type AccuracyResult = {
|
|
487
|
+
assertions: Assertion[];
|
|
488
|
+
/**
|
|
489
|
+
* Who owns each domain the engine cited.
|
|
490
|
+
*
|
|
491
|
+
* Kept beside the assertions rather than copied into each one: the same domain
|
|
492
|
+
* backs several claims, and a verdict about who owns a website should exist in
|
|
493
|
+
* exactly one place. Without this the report called a client's own abandoned
|
|
494
|
+
* site "somewhere else the engine looked" — a factual error about their
|
|
495
|
+
* business, inside a document about factual errors about their business.
|
|
496
|
+
*/
|
|
497
|
+
sources: DomainVerdict[];
|
|
498
|
+
/** False when the site was too large to send whole. Every `absent` verdict is
|
|
499
|
+
* suppressed to `unverified` when this is false, because "the site does not
|
|
500
|
+
* say it" is not a claim we can make about pages we did not read. */
|
|
501
|
+
siteFullyRead: boolean;
|
|
502
|
+
pagesRead: number;
|
|
503
|
+
pagesTotal: number;
|
|
504
|
+
/** Which branded answers we had full text for. A run whose probes predate
|
|
505
|
+
* `fullAnswer` reports zero and no assertions, rather than "nothing wrong". */
|
|
506
|
+
answersRead: number;
|
|
507
|
+
/** The engine confused this business with others of the same name. For a
|
|
508
|
+
* common name this is the headline of the branded search, and it was only
|
|
509
|
+
* ever visible inside a truncated quote. `engineQuote` is verified against
|
|
510
|
+
* the answer text like every other quote; a conflation whose quote is not
|
|
511
|
+
* in the answer is discarded, not trusted. */
|
|
512
|
+
conflation: Conflation;
|
|
513
|
+
};
|
|
514
|
+
type Conflation = {
|
|
515
|
+
detected: boolean;
|
|
516
|
+
/** The OTHER businesses the engine named — never this one. */
|
|
517
|
+
otherNames: string[];
|
|
518
|
+
engineQuote: string | null;
|
|
519
|
+
};
|
|
520
|
+
|
|
521
|
+
/**
|
|
522
|
+
* Can a visitor do the one thing this site needs them to do?
|
|
523
|
+
*
|
|
524
|
+
* Every other check in this audit is generic — crawlable, readable, not broken.
|
|
525
|
+
* Those are worth measuring and they are the same questions for every site,
|
|
526
|
+
* which is exactly the problem: a dentist's site succeeds when somebody books an
|
|
527
|
+
* appointment and a branding studio's succeeds when a qualified enquiry arrives
|
|
528
|
+
* with a budget attached. Grading both against one template scores neither
|
|
529
|
+
* against what it is for.
|
|
530
|
+
*
|
|
531
|
+
* So one goal is named, and the findings are read against it. The goal is
|
|
532
|
+
* inferred from the site by the analyze stage and may be overridden by the
|
|
533
|
+
* operator — and an inference we get wrong is itself worth reporting, because if
|
|
534
|
+
* a model that has just read twenty pages cannot tell what the site is for,
|
|
535
|
+
* neither can a visitor.
|
|
536
|
+
*
|
|
537
|
+
* Findings carry a `scope`, and the report orders by it. That ordering is the
|
|
538
|
+
* only place the commercial ladder appears: a stale phone link is an afternoon,
|
|
539
|
+
* "you have never published a price and every buyer asks" is a content
|
|
540
|
+
* engagement, and "there is no way to publish or update any of this" is a
|
|
541
|
+
* platform conversation. The report must never say those words. Ordering does
|
|
542
|
+
* the work, and a prospect who reads to the bottom arrives at the conversation
|
|
543
|
+
* on their own.
|
|
544
|
+
*/
|
|
545
|
+
type SiteGoal = "book" | "enquire" | "call" | "visit" | "buy" | "demo" | "partner" | "unknown";
|
|
546
|
+
declare const GOAL_LABELS: Record<SiteGoal, string>;
|
|
547
|
+
/** How much work putting one thing right is. Orders the report; never printed
|
|
548
|
+
* as a tier, and never priced. */
|
|
549
|
+
type Scope = "quick" | "content" | "structural";
|
|
550
|
+
/**
|
|
551
|
+
* Three states, not two — and the third is the one that matters.
|
|
552
|
+
*
|
|
553
|
+
* Caught on live data: `reachable` and `tappable-phone` both read their input
|
|
554
|
+
* from `checks`, which older stored reports do not carry. With a boolean they
|
|
555
|
+
* came back `false`, and the report said "no way to reach you from where they
|
|
556
|
+
* land" about sites that have one — turning our own missing measurement into
|
|
557
|
+
* the prospect's defect. That is the single error this whole codebase is built
|
|
558
|
+
* not to make, and a two-state field makes it the default.
|
|
559
|
+
*/
|
|
560
|
+
type RequirementStatus = "met" | "missing" | "unmeasured";
|
|
561
|
+
type GoalRequirement = {
|
|
562
|
+
key: string;
|
|
563
|
+
/** What the visitor needs, in the visitor's terms. */
|
|
564
|
+
label: string;
|
|
565
|
+
status: RequirementStatus;
|
|
566
|
+
/** Where we found it — the receipt. Null when we did not. */
|
|
567
|
+
evidence: string | null;
|
|
568
|
+
/** Why this one matters for THIS goal, not in general. */
|
|
569
|
+
why: string;
|
|
570
|
+
scope: Scope;
|
|
571
|
+
};
|
|
572
|
+
type GoalFit = {
|
|
573
|
+
goal: SiteGoal;
|
|
574
|
+
/** "operator" when supplied at dispatch, "inferred" when the analyze stage
|
|
575
|
+
* read it off the site. A reader is owed that distinction. */
|
|
576
|
+
source: "inferred" | "operator";
|
|
577
|
+
requirements: GoalRequirement[];
|
|
578
|
+
met: number;
|
|
579
|
+
/** Requirements we could actually judge — `met` + `missing`. Excludes
|
|
580
|
+
* `unmeasured`, so "3 of 4" never counts something we did not look at. */
|
|
581
|
+
total: number;
|
|
582
|
+
};
|
|
583
|
+
declare function orderRequirements(reqs: GoalRequirement[]): GoalRequirement[];
|
|
584
|
+
|
|
585
|
+
/**
|
|
586
|
+
* Can a visitor actually get from where they landed to a way of contacting you?
|
|
587
|
+
*
|
|
588
|
+
* Everything else in this audit measures whether a site can be found and read.
|
|
589
|
+
* This measures whether it can be ACTED ON, which is the part that decides
|
|
590
|
+
* whether traffic becomes a phone call.
|
|
591
|
+
*
|
|
592
|
+
* The measurement that matters is click distance, and it matters because of
|
|
593
|
+
* where visitors actually land. A search engine — and an answer engine citing a
|
|
594
|
+
* page — sends people to a deep page, not the homepage. A site whose only
|
|
595
|
+
* contact form sits behind the homepage nav is asking a stranger who arrived on
|
|
596
|
+
* a blog post to go looking. Some do. Most leave.
|
|
597
|
+
*
|
|
598
|
+
* So this builds a link graph across the pages the crawl retrieved, finds every
|
|
599
|
+
* way of making contact on each one, and walks outward to find how far the
|
|
600
|
+
* nearest one is. A page with no path at all is a dead end, and a dead end is a
|
|
601
|
+
* defect with a price attached.
|
|
602
|
+
*
|
|
603
|
+
* The honest limit, which every consumer must carry into what it prints: the
|
|
604
|
+
* crawl retrieves a handful of pages, not the whole site. "Dead end" here means
|
|
605
|
+
* "no path among the pages we looked at" — real, worth reporting, and not the
|
|
606
|
+
* same claim as "no path exists". `pagesExamined` is reported so the sentence
|
|
607
|
+
* can say which one it means.
|
|
608
|
+
*/
|
|
609
|
+
type AffordanceKind = "form" | "tel" | "mailto";
|
|
610
|
+
type ContactAffordance = {
|
|
611
|
+
kind: AffordanceKind;
|
|
612
|
+
/** The crawled page it was found on. */
|
|
613
|
+
page: string;
|
|
614
|
+
/** The number, the address, or the form's action — the receipt. */
|
|
615
|
+
detail: string;
|
|
616
|
+
};
|
|
617
|
+
type PageJourney = {
|
|
618
|
+
url: string;
|
|
619
|
+
/** Clicks to the nearest page carrying a contact affordance. 0 means it is on
|
|
620
|
+
* this page; null means no path was found among the pages examined. */
|
|
621
|
+
clicksToContact: number | null;
|
|
622
|
+
/** Links from this page to other pages the crawl retrieved. Zero means a
|
|
623
|
+
* visitor who lands here can go nowhere, which is a different and worse
|
|
624
|
+
* problem than merely being far from the contact page. */
|
|
625
|
+
internalLinks: number;
|
|
626
|
+
};
|
|
627
|
+
type JourneyMap = {
|
|
628
|
+
affordances: ContactAffordance[];
|
|
629
|
+
pages: PageJourney[];
|
|
630
|
+
/** Pages with no path to any contact affordance, among those examined. */
|
|
631
|
+
deadEnds: string[];
|
|
632
|
+
/** The worst distance among pages that DO have a path — the honest headline,
|
|
633
|
+
* because an average hides the one page that strands people. Null when no
|
|
634
|
+
* page has a path. */
|
|
635
|
+
worstClicksToContact: number | null;
|
|
636
|
+
/** How many pages this was computed over. Reported so no consumer can
|
|
637
|
+
* describe a five-page sample as though it were the whole site. */
|
|
638
|
+
pagesExamined: number;
|
|
639
|
+
/**
|
|
640
|
+
* Did every page examined carry a recorded anchor list?
|
|
641
|
+
*
|
|
642
|
+
* False means the crawl did not record links (reports stored before
|
|
643
|
+
* `PageExtract.anchors` existed), and NOTHING here is a finding: `deadEnds`
|
|
644
|
+
* is empty and `worstClicksToContact` is null, because a page whose links we
|
|
645
|
+
* never captured is a page we cannot say anything about. Reading an absent
|
|
646
|
+
* anchors array as "no links" would report our own missing field as every
|
|
647
|
+
* page on the site being a dead end.
|
|
648
|
+
*/
|
|
649
|
+
anchorsMeasured: boolean;
|
|
650
|
+
};
|
|
651
|
+
|
|
3
652
|
/** Every pipeline stage resolves to this. A failed stage degrades its report
|
|
4
653
|
* section to "not measured" — it never kills the run (spec: error handling). */
|
|
5
654
|
type StageResult<T> = {
|
|
@@ -16,6 +665,53 @@ type RobotsAgentAccess = {
|
|
|
16
665
|
/** The deciding rule, e.g. "User-agent: GPTBot → Disallow: /". Null = no rule matched. */
|
|
17
666
|
matchedRule: string | null;
|
|
18
667
|
};
|
|
668
|
+
/** One `<a href>`. `href` is exactly as authored — relative, absolute, `tel:`,
|
|
669
|
+
* `mailto:`, `#anchor` — and is resolved against the page URL by whoever needs
|
|
670
|
+
* an absolute one. Resolving here would discard the distinction between a
|
|
671
|
+
* genuinely absolute link and a relative one, which is itself a finding when a
|
|
672
|
+
* site hardcodes a staging host. */
|
|
673
|
+
type PageAnchor = {
|
|
674
|
+
href: string;
|
|
675
|
+
text: string;
|
|
676
|
+
rel: string;
|
|
677
|
+
};
|
|
678
|
+
/**
|
|
679
|
+
* What a form is FOR, inferred from its shape.
|
|
680
|
+
*
|
|
681
|
+
* - `enquiry` asks for a way to reply AND for something else — a name, a
|
|
682
|
+
* message, a budget. This is the one that reaches a human.
|
|
683
|
+
* - `subscribe` asks for a way to reply and nothing else: a lone email box.
|
|
684
|
+
* A newsletter signup is a conversion, but it is not a way to
|
|
685
|
+
* get an answer, and a visitor with a question is not served by
|
|
686
|
+
* one.
|
|
687
|
+
* - `other` everything else — search, filters, logins, calculators. No
|
|
688
|
+
* attempt is made to tell those apart; the shapes overlap too
|
|
689
|
+
* much to do it honestly from markup.
|
|
690
|
+
*
|
|
691
|
+
* The distinction earns its keep on real data. One audited site carries a
|
|
692
|
+
* one-field email box in the footer of every page. Counting that as a contact
|
|
693
|
+
* route put the whole site at zero clicks from "reaching them", when in fact
|
|
694
|
+
* the only form that reaches a person was the nine-field one on its contact
|
|
695
|
+
* page.
|
|
696
|
+
*/
|
|
697
|
+
type FormKind = "enquiry" | "subscribe" | "other";
|
|
698
|
+
/** One `<form>`, in enough detail to tell an enquiry form from a newsletter box
|
|
699
|
+
* or a search field. That distinction is the whole point: a site nobody can
|
|
700
|
+
* actually reach must not score as though it has a conversion path. */
|
|
701
|
+
type FormShape = {
|
|
702
|
+
kind: FormKind;
|
|
703
|
+
/** As authored, or null for a form that posts to its own URL. */
|
|
704
|
+
action: string | null;
|
|
705
|
+
/** Lower-cased; defaults to "get", which is what a browser does. */
|
|
706
|
+
method: string;
|
|
707
|
+
/** Visible, named controls — hidden inputs, submits and buttons excluded, so
|
|
708
|
+
* a one-field newsletter box does not read the same as a real enquiry form. */
|
|
709
|
+
fieldCount: number;
|
|
710
|
+
/** Does it ask for an email address or a phone number? A search box does not,
|
|
711
|
+
* and this is what separates "can be contacted" from "can be searched". */
|
|
712
|
+
hasContactField: boolean;
|
|
713
|
+
hasSubmit: boolean;
|
|
714
|
+
};
|
|
19
715
|
type PageExtract = {
|
|
20
716
|
title: string | null;
|
|
21
717
|
metaDescription: string | null;
|
|
@@ -36,6 +732,28 @@ type PageExtract = {
|
|
|
36
732
|
hasViewportMeta: boolean;
|
|
37
733
|
/** Visible text, whitespace-collapsed. */
|
|
38
734
|
text: string;
|
|
735
|
+
/**
|
|
736
|
+
* Anchors in document order, CAPPED — see `anchorCount` for the true total.
|
|
737
|
+
*
|
|
738
|
+
* Capped because this whole extract is persisted into
|
|
739
|
+
* `prospect_audits.result_json`, once per page per audit, and a navigation-
|
|
740
|
+
* heavy page can carry several hundred anchors. The cap is generous enough
|
|
741
|
+
* that no ordinary page reaches it.
|
|
742
|
+
*
|
|
743
|
+
* The count is reported separately rather than left implicit, because a
|
|
744
|
+
* truncated list that looks complete is exactly the kind of quiet lie this
|
|
745
|
+
* audit is built not to tell: "we checked every link" and "we checked the
|
|
746
|
+
* first 300" are different claims.
|
|
747
|
+
*
|
|
748
|
+
* Optional: reports stored before this existed lack it, and a reader must
|
|
749
|
+
* treat its absence as "not measured" rather than "no links".
|
|
750
|
+
*/
|
|
751
|
+
anchors?: PageAnchor[];
|
|
752
|
+
/** True number of `<a href>` on the page, before `anchors` was capped. */
|
|
753
|
+
anchorCount?: number;
|
|
754
|
+
/** `src` of each `<img>`, as authored. Same resolution note as `anchors`. */
|
|
755
|
+
imageSrcs?: string[];
|
|
756
|
+
forms?: FormShape[];
|
|
39
757
|
};
|
|
40
758
|
type PageCapture = {
|
|
41
759
|
url: string;
|
|
@@ -138,10 +856,22 @@ type ChecksResult = {
|
|
|
138
856
|
llmsTxtMeasured: boolean;
|
|
139
857
|
llmsTxtPresent: boolean;
|
|
140
858
|
viewportOk: boolean;
|
|
859
|
+
/** Can a visitor get from where they landed to a way of contacting you?
|
|
860
|
+
* Optional: reports stored before this was measured lack it, and a reader
|
|
861
|
+
* must say "not measured" rather than "no path". See journey.ts. */
|
|
862
|
+
journey?: JourneyMap;
|
|
863
|
+
/** Does the site tell the same story on every page? See consistency.ts. */
|
|
864
|
+
consistency?: ConsistencyResult;
|
|
141
865
|
};
|
|
142
866
|
type BuyerQuestion = {
|
|
867
|
+
/** Stable key from the fixed set in questions.ts. Absent on reports stored
|
|
868
|
+
* before the set was fixed, when the model still wrote its own questions. */
|
|
869
|
+
id?: string;
|
|
143
870
|
question: string;
|
|
144
|
-
|
|
871
|
+
/** "unknown" is ours, never the model's: it means we did not get an answer
|
|
872
|
+
* for a question we asked, so the row is shown and excluded from the score.
|
|
873
|
+
* A missing answer of ours must never be scored as a "no" about them. */
|
|
874
|
+
answered: "yes" | "partial" | "no" | "unknown";
|
|
145
875
|
/** Is there a passage an AI answer could quote verbatim? */
|
|
146
876
|
quotable: boolean;
|
|
147
877
|
page: string | null;
|
|
@@ -153,6 +883,17 @@ type Fix = {
|
|
|
153
883
|
impact: "high" | "medium" | "low";
|
|
154
884
|
effort: "low" | "medium" | "high";
|
|
155
885
|
tier: "crawl" | "content" | "technical";
|
|
886
|
+
/** The goal-requirement key this fix would satisfy, when it maps to one —
|
|
887
|
+
* the handle `reconcileFixes` uses to drop a fix for something we already
|
|
888
|
+
* measured as present. Null is the normal case: most good fixes (a heavy
|
|
889
|
+
* image, a broken link, a stale year) answer to no requirement at all.
|
|
890
|
+
* Absent on reports stored before the model was asked to tag them. */
|
|
891
|
+
addresses?: string | null;
|
|
892
|
+
/** Where the fix came from. "measured": produced by code from a check that
|
|
893
|
+
* ran on this audit — a finding. "recommendation": written by the model —
|
|
894
|
+
* judgement, rendered below the findings and labelled as such. Absent on
|
|
895
|
+
* reports stored before the split, which were all model-written. */
|
|
896
|
+
origin?: "measured" | "recommendation";
|
|
156
897
|
};
|
|
157
898
|
type AnalyzeResult = {
|
|
158
899
|
/** The business's proper name, as a searchable proper noun — "Acme Roofing",
|
|
@@ -166,7 +907,15 @@ type AnalyzeResult = {
|
|
|
166
907
|
score: number;
|
|
167
908
|
missing: string[];
|
|
168
909
|
};
|
|
910
|
+
/** The one action this site is built to produce — the lens goals.ts reads
|
|
911
|
+
* every requirement through. Optional for reports stored before it existed;
|
|
912
|
+
* absence means the goal section reads "not measured". */
|
|
913
|
+
primaryGoal?: SiteGoal;
|
|
169
914
|
buyerQuestions: BuyerQuestion[];
|
|
915
|
+
/** Which fixed question set was asked — `${goal}-v${QUESTION_SET_VERSION}`.
|
|
916
|
+
* Absent on reports stored before the set was fixed. Two audits' Answers
|
|
917
|
+
* scores are comparable exactly when this matches; see sameQuestionSet. */
|
|
918
|
+
questionSetId?: string;
|
|
170
919
|
/** Standalone searches for the visibility probes — what a buyer types before
|
|
171
920
|
* they know this company exists. Distinct from `buyerQuestions`, which are
|
|
172
921
|
* phrased about this site and are unanswerable on their own; see the schema
|
|
@@ -206,6 +955,21 @@ type ProbeAnswer = {
|
|
|
206
955
|
citedDomains: string[];
|
|
207
956
|
/** First ~300 chars of the engine's answer — the report's receipt. */
|
|
208
957
|
snippet: string;
|
|
958
|
+
/**
|
|
959
|
+
* The engine's answer in full, kept for BRANDED answers only.
|
|
960
|
+
*
|
|
961
|
+
* The accuracy stage reads these to work out which statements about the
|
|
962
|
+
* business its own site actually supports, and a 300-character snippet cuts
|
|
963
|
+
* off mid-sentence — every claim past it would be invisible, and invisible
|
|
964
|
+
* reads the same as absent. Branded answers are two or three per run, so
|
|
965
|
+
* keeping them whole costs little; category and competitor answers are many
|
|
966
|
+
* and nothing downstream reads them in full, so they do not carry it.
|
|
967
|
+
*
|
|
968
|
+
* Optional because reports stored before the accuracy stage existed do not
|
|
969
|
+
* have it, and there the stage must report that it could not run rather than
|
|
970
|
+
* report an absence of claims.
|
|
971
|
+
*/
|
|
972
|
+
fullAnswer?: string;
|
|
209
973
|
/** True when `snippet` is a truncated prefix of a longer answer — set
|
|
210
974
|
* right where the truncation happens (probes.ts's SNIPPET_CHARS), so the
|
|
211
975
|
* renderer never has to re-derive "was this cut short?" from a length
|
|
@@ -254,6 +1018,31 @@ type ProbesResult = {
|
|
|
254
1018
|
attempted: number;
|
|
255
1019
|
answered: number;
|
|
256
1020
|
};
|
|
1021
|
+
/**
|
|
1022
|
+
* The shape of the answer the prospect is absent from — see answer-space.ts.
|
|
1023
|
+
*
|
|
1024
|
+
* `visibilityScore` cannot carry a report on its own: across the 12 audits
|
|
1025
|
+
* stored to date it is 0 for eight of them and takes four distinct values in
|
|
1026
|
+
* total, so it cannot rank most prospects against each other at all — and a
|
|
1027
|
+
* bare 0 invites the one question we cannot honestly answer, "how do we make
|
|
1028
|
+
* it go up?". Two zeros can mean opposite things: a category answered by
|
|
1029
|
+
* Stryker, Arthrex and the FDA (no website work reaches that answer — the
|
|
1030
|
+
* honest counsel is not to buy AEO) versus one answered by five local
|
|
1031
|
+
* practices exactly the prospect's size (plainly reachable, and they simply
|
|
1032
|
+
* are not there). This is the evidence that tells those apart.
|
|
1033
|
+
*
|
|
1034
|
+
* Related to `competitorsSeen` but not a replacement for it: that field
|
|
1035
|
+
* predates this one and is what stored reports and the current renderer read,
|
|
1036
|
+
* so it stays. What is new here is the denominator — how many distinct
|
|
1037
|
+
* sources the engine drew on, how many of them it takes to cover half the
|
|
1038
|
+
* citations, and where the prospect's own domain ranks among them.
|
|
1039
|
+
*
|
|
1040
|
+
* Optional because this type also describes runs deserialized from
|
|
1041
|
+
* `prospect_audits.result_json`, and every report stored before this field
|
|
1042
|
+
* existed lacks it. `runVisibilityProbes` always sets it; readers must still
|
|
1043
|
+
* handle its absence rather than assume a stored report has it.
|
|
1044
|
+
*/
|
|
1045
|
+
answerSpace?: AnswerSpace;
|
|
257
1046
|
};
|
|
258
1047
|
type LighthouseScores = {
|
|
259
1048
|
performance: number | null;
|
|
@@ -303,6 +1092,24 @@ type ProspectAuditResult = {
|
|
|
303
1092
|
lighthouse: StageResult<LighthouseScores>;
|
|
304
1093
|
analyze: StageResult<AnalyzeResult>;
|
|
305
1094
|
probes: StageResult<ProbesResult>;
|
|
1095
|
+
/** Broken links, broken images and image weight. Its own stage rather than
|
|
1096
|
+
* part of `checks` because it is the only site check that makes requests —
|
|
1097
|
+
* `runChecks` is pure and synchronous over the crawl, and it is worth
|
|
1098
|
+
* keeping it that way. Optional for reports stored before it existed. */
|
|
1099
|
+
assets?: StageResult<AssetCheck>;
|
|
1100
|
+
/** The things a stranger checks first — whether the address works typed the
|
|
1101
|
+
* ordinary ways, what a missing page does, and the handful of basics the
|
|
1102
|
+
* crawl already has the evidence for. See basics.ts. Optional for reports
|
|
1103
|
+
* stored before it existed. */
|
|
1104
|
+
basics?: StageResult<BasicsCheck>;
|
|
1105
|
+
/** Can a visitor do the one thing this site needs them to do? Read against a
|
|
1106
|
+
* single named goal rather than a generic template — see goals.ts. Optional
|
|
1107
|
+
* for reports stored before it existed. */
|
|
1108
|
+
goalFit?: StageResult<GoalFit>;
|
|
1109
|
+
/** When an engine describes this business, where is it getting that from —
|
|
1110
|
+
* each statement sorted by SOURCE, never by truth. See accuracy.ts. Optional
|
|
1111
|
+
* for reports stored before the stage was wired in. */
|
|
1112
|
+
accuracy?: StageResult<AccuracyResult>;
|
|
306
1113
|
};
|
|
307
1114
|
|
|
308
|
-
export type
|
|
1115
|
+
export { type AnalyzeResult, type AnswerSpace, type AssetCheck, type BasicsCheck, type BuyerQuestion, type ChecksResult, type ConsistencyResult, type ContactAffordance, type ContactVariant, type CrawlResult, type Fix, type FormKind, type FormShape, GOAL_LABELS, type GoalFit, type GoalRequirement, type JourneyMap, type LighthouseScores, type PageAnchor, type PageCapture, type PageExtract, type PageJourney, type ProbeAnswer, type ProbedUrl, type ProbesResult, type ProspectAuditResult, type Reachability, type RobotsAgentAccess, type Scope, type Scores, type SiteGoal, type SourceCount, type StageResult, orderRequirements };
|