@littlefriend/cli 0.1.4 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +40 -2
- package/dist/index.js +2245 -2139
- package/package.json +3 -3
package/dist/index.js
CHANGED
|
@@ -131,7 +131,7 @@ import { createHash, randomBytes, timingSafeEqual } from "node:crypto";
|
|
|
131
131
|
import { createServer } from "node:http";
|
|
132
132
|
|
|
133
133
|
// src/version.ts
|
|
134
|
-
var VERSION = "0.1.
|
|
134
|
+
var VERSION = "0.1.5";
|
|
135
135
|
var USER_AGENT = `littlefriend-cli/${VERSION}`;
|
|
136
136
|
|
|
137
137
|
// src/oauth.ts
|
|
@@ -631,2388 +631,2485 @@ var Api = class {
|
|
|
631
631
|
}
|
|
632
632
|
};
|
|
633
633
|
|
|
634
|
-
//
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
634
|
+
// ../contract/dist/actors.js
|
|
635
|
+
var ACTOR_CLASS_LABELS = {
|
|
636
|
+
browser: "Interactive browser",
|
|
637
|
+
search_crawler: "Search crawler",
|
|
638
|
+
ai_crawler: "AI crawler",
|
|
639
|
+
ai_agent: "User-directed AI agent",
|
|
640
|
+
monitor: "Uptime and monitoring",
|
|
641
|
+
preview: "Link preview",
|
|
642
|
+
automation: "Suspected automation",
|
|
643
|
+
unknown: "Unknown"
|
|
644
|
+
};
|
|
645
|
+
|
|
646
|
+
// ../contract/dist/limits.js
|
|
647
|
+
var LIMITS = {
|
|
648
|
+
/** Max request body accepted by the collector, in bytes (before decompression). */
|
|
649
|
+
maxBodyBytes: 16 * 1024,
|
|
650
|
+
/** Max events in one batch. */
|
|
651
|
+
maxEventsPerBatch: 50,
|
|
652
|
+
/** Max custom properties on one event. */
|
|
653
|
+
maxProps: 8,
|
|
654
|
+
maxPropKeyLength: 32,
|
|
655
|
+
maxPropStringLength: 64,
|
|
656
|
+
maxEventNameLength: 64,
|
|
657
|
+
maxRouteLength: 256,
|
|
658
|
+
maxHostLength: 253,
|
|
659
|
+
maxCampaignValueLength: 64,
|
|
660
|
+
/** Events observed further in the past than this are rejected as stale. */
|
|
661
|
+
maxPastSkewMs: 24 * 60 * 60 * 1e3,
|
|
662
|
+
/** Events observed further in the future than this are rejected as clock skew. */
|
|
663
|
+
maxFutureSkewMs: 10 * 60 * 1e3,
|
|
664
|
+
/** Browser SDK flush triggers. */
|
|
665
|
+
browserFlushEvents: 10,
|
|
666
|
+
browserFlushMs: 5e3,
|
|
667
|
+
/** Browser SDK bounded pending queue, in bytes of serialized events. */
|
|
668
|
+
browserMaxPendingBytes: 64 * 1024,
|
|
669
|
+
/** Journey sessions: inactivity timeout and hard lifetime. */
|
|
670
|
+
sessionIdleMs: 30 * 60 * 1e3,
|
|
671
|
+
sessionMaxMs: 24 * 60 * 60 * 1e3
|
|
672
|
+
};
|
|
673
|
+
|
|
674
|
+
// ../contract/dist/apps.js
|
|
675
|
+
var PLATFORMS = ["web", "ios", "macos", "android"];
|
|
676
|
+
var APP_ID_RE = /^[A-Za-z0-9][A-Za-z0-9_-]*(\.[A-Za-z0-9_-]+)+$/;
|
|
677
|
+
var APP_ID_MAX = 155;
|
|
678
|
+
function validAppId(id) {
|
|
679
|
+
return typeof id === "string" && id.length <= APP_ID_MAX && APP_ID_RE.test(id);
|
|
680
|
+
}
|
|
681
|
+
function normalizeAppId(id) {
|
|
682
|
+
return validAppId(id) ? id.toLowerCase() : null;
|
|
683
|
+
}
|
|
684
|
+
var UTM_FIELDS = [
|
|
685
|
+
["utm_source", "s"],
|
|
686
|
+
["utm_medium", "m"],
|
|
687
|
+
["utm_campaign", "c"]
|
|
688
|
+
];
|
|
689
|
+
var UTM_EXTENDED_FIELDS = [
|
|
690
|
+
...UTM_FIELDS,
|
|
691
|
+
["utm_term", "t"],
|
|
692
|
+
["utm_content", "n"]
|
|
693
|
+
];
|
|
694
|
+
|
|
695
|
+
// ../contract/dist/channels.js
|
|
696
|
+
var AI_ASSISTANT_HOSTS = [
|
|
697
|
+
{ host: "chatgpt.com", label: "ChatGPT", operator: "OpenAI" },
|
|
698
|
+
{ host: "chat.openai.com", label: "ChatGPT", operator: "OpenAI" },
|
|
699
|
+
{ host: "perplexity.ai", label: "Perplexity", operator: "Perplexity" },
|
|
700
|
+
{ host: "claude.ai", label: "Claude", operator: "Anthropic" },
|
|
701
|
+
{ host: "gemini.google.com", label: "Gemini", operator: "Google" },
|
|
702
|
+
{ host: "copilot.microsoft.com", label: "Microsoft Copilot", operator: "Microsoft" },
|
|
703
|
+
{ host: "you.com", label: "You.com", operator: "You.com" },
|
|
704
|
+
{ host: "phind.com", label: "Phind", operator: "Phind" }
|
|
705
|
+
];
|
|
706
|
+
|
|
707
|
+
// ../contract/dist/flows.js
|
|
708
|
+
var FLOW_START_DEFINITIONS = [
|
|
709
|
+
{
|
|
710
|
+
key: "flow_step",
|
|
711
|
+
label: "Step",
|
|
712
|
+
definition: "Page views in a row of the same page type (or the same page) are one step. A goal is a step of its own, right after the page where it happened. Anything before the first page view is left out."
|
|
713
|
+
},
|
|
714
|
+
{
|
|
715
|
+
key: "flow_start_sessions",
|
|
716
|
+
label: "Sessions",
|
|
717
|
+
definition: "Journey sessions that started in the range and whose first step is this one. Sessions are not people."
|
|
718
|
+
},
|
|
719
|
+
{
|
|
720
|
+
key: "flow_start_share",
|
|
721
|
+
label: "Share",
|
|
722
|
+
definition: "These sessions divided by all sessions with a step."
|
|
723
|
+
},
|
|
724
|
+
{ key: "flow_went_on", label: "Went on", definition: "Of these sessions, the share with a second step." },
|
|
725
|
+
{
|
|
726
|
+
key: "flow_converted",
|
|
727
|
+
label: "Converted",
|
|
728
|
+
definition: "Of these sessions, the share that reached at least one goal."
|
|
665
729
|
}
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
730
|
+
];
|
|
731
|
+
var FLOW_DEFINITIONS = [
|
|
732
|
+
FLOW_START_DEFINITIONS[0],
|
|
733
|
+
{
|
|
734
|
+
key: "flow_node",
|
|
735
|
+
label: "Sessions at a step",
|
|
736
|
+
definition: "Sessions that started at the starting point and reached this step in this position. The count after it is the page views merged into the step: pages when grouped by type, views when grouped by page."
|
|
737
|
+
},
|
|
738
|
+
{ key: "flow_left", label: "Left", definition: "The session had no more steps and has ended." },
|
|
739
|
+
{
|
|
740
|
+
key: "flow_browsing",
|
|
741
|
+
label: "Still browsing",
|
|
742
|
+
definition: "The session has no more steps yet but has not ended, so it may continue. A session ends after 30 idle minutes."
|
|
743
|
+
},
|
|
744
|
+
{
|
|
745
|
+
key: "flow_other",
|
|
746
|
+
label: "Other",
|
|
747
|
+
definition: "Pages beyond the busiest ones in that step, together. Their links are merged too."
|
|
748
|
+
},
|
|
749
|
+
{
|
|
750
|
+
key: "flow_continued",
|
|
751
|
+
label: "Continued",
|
|
752
|
+
definition: "In the last column: sessions with more steps than the chart shows."
|
|
673
753
|
}
|
|
674
|
-
|
|
675
|
-
}
|
|
676
|
-
function joinList(items, word) {
|
|
677
|
-
if (items.length <= 1) return items.join("");
|
|
678
|
-
return `${items.slice(0, -1).join(", ")} ${word} ${items[items.length - 1]}`;
|
|
679
|
-
}
|
|
680
|
-
var orList = (items) => joinList(items, "or");
|
|
681
|
-
var andList = (items) => joinList(items, "and");
|
|
682
|
-
function requireYes(ctx, what) {
|
|
683
|
-
if (!flag(ctx, "yes")) throw usageError(`This ${what}. Add --yes to confirm.`);
|
|
684
|
-
}
|
|
685
|
-
var YES = { yes: { type: "boolean" } };
|
|
686
|
-
function arg(ctx, index, what) {
|
|
687
|
-
const v = ctx.args[index];
|
|
688
|
-
if (!v) throw usageError(`Missing ${what}.`);
|
|
689
|
-
return v;
|
|
690
|
-
}
|
|
691
|
-
var enc = encodeURIComponent;
|
|
754
|
+
];
|
|
692
755
|
|
|
693
|
-
//
|
|
694
|
-
function
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
}
|
|
706
|
-
function fields(pairs) {
|
|
707
|
-
const width = Math.max(...pairs.map(([k]) => k.length));
|
|
708
|
-
return pairs.map(([k, v]) => `${`${k}:`.padEnd(width + 2)}${cell(v)}`.trimEnd()).join("\n");
|
|
709
|
-
}
|
|
710
|
-
function when(iso) {
|
|
711
|
-
if (!iso) return "";
|
|
712
|
-
const d = new Date(iso);
|
|
713
|
-
if (Number.isNaN(d.getTime())) return iso;
|
|
714
|
-
return `${d.toISOString().slice(0, 16).replace("T", " ")} UTC`;
|
|
715
|
-
}
|
|
716
|
-
function count(n) {
|
|
717
|
-
return typeof n === "number" && Number.isFinite(n) ? n.toLocaleString("en-US") : "";
|
|
718
|
-
}
|
|
719
|
-
function percent(fraction) {
|
|
720
|
-
if (typeof fraction !== "number" || !Number.isFinite(fraction)) return "";
|
|
721
|
-
return `${(fraction * 100).toFixed(1).replace(/\.0$/, "")}%`;
|
|
756
|
+
// ../contract/dist/kya.js
|
|
757
|
+
function principalRank(level) {
|
|
758
|
+
switch (level) {
|
|
759
|
+
case "none":
|
|
760
|
+
return 0;
|
|
761
|
+
case "user_initiated":
|
|
762
|
+
return 1;
|
|
763
|
+
case "signed_user":
|
|
764
|
+
return 2;
|
|
765
|
+
default:
|
|
766
|
+
return 3;
|
|
767
|
+
}
|
|
722
768
|
}
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
769
|
+
var SIGNED_USER_TAGS = ["agent-browser-auth", "agent-payer-auth"];
|
|
770
|
+
function evidenceRank(evidence) {
|
|
771
|
+
switch (evidence) {
|
|
772
|
+
case "none":
|
|
773
|
+
return 0;
|
|
774
|
+
case "self_declared":
|
|
775
|
+
return 1;
|
|
776
|
+
case "network_verified":
|
|
777
|
+
return 2;
|
|
778
|
+
default:
|
|
779
|
+
return 3;
|
|
780
|
+
}
|
|
728
781
|
}
|
|
729
|
-
|
|
730
|
-
|
|
782
|
+
var lower = (s) => s.toLowerCase();
|
|
783
|
+
function pathMatches(path, prefix) {
|
|
784
|
+
if (prefix === "/")
|
|
785
|
+
return true;
|
|
786
|
+
const p = prefix.endsWith("/") ? prefix.slice(0, -1) : prefix;
|
|
787
|
+
return path === p || path.startsWith(`${p}/`);
|
|
731
788
|
}
|
|
732
|
-
function
|
|
733
|
-
|
|
789
|
+
function doorRuleMatches(match, facts) {
|
|
790
|
+
if (match.classes && !match.classes.includes(facts.class))
|
|
791
|
+
return false;
|
|
792
|
+
if (match.purposes && !match.purposes.includes(facts.purpose))
|
|
793
|
+
return false;
|
|
794
|
+
if (match.operators && !(facts.operator && match.operators.map(lower).includes(lower(facts.operator))))
|
|
795
|
+
return false;
|
|
796
|
+
if (match.products && !(facts.product && match.products.map(lower).includes(lower(facts.product))))
|
|
797
|
+
return false;
|
|
798
|
+
if (match.evidenceBelow && !(evidenceRank(facts.evidence) < evidenceRank(match.evidenceBelow)))
|
|
799
|
+
return false;
|
|
800
|
+
if (match.principalBelow && !(principalRank(facts.principal) < principalRank(match.principalBelow)))
|
|
801
|
+
return false;
|
|
802
|
+
if (match.spoofed !== void 0 && match.spoofed !== facts.spoofed)
|
|
803
|
+
return false;
|
|
804
|
+
if (match.paths && !match.paths.some((p) => pathMatches(facts.path, p)))
|
|
805
|
+
return false;
|
|
806
|
+
if (match.methods && !match.methods.map(lower).includes(lower(facts.method)))
|
|
807
|
+
return false;
|
|
808
|
+
return true;
|
|
734
809
|
}
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
const
|
|
739
|
-
|
|
810
|
+
function evaluateDoor(policy, facts) {
|
|
811
|
+
if (!policy || facts.class === "browser")
|
|
812
|
+
return { action: "allow", ruleId: null };
|
|
813
|
+
for (const rule of policy.rules) {
|
|
814
|
+
if (!doorRuleMatches(rule.match, facts))
|
|
815
|
+
continue;
|
|
816
|
+
return rule.action === "limit" && rule.limit ? { action: "limit", ruleId: rule.id, limit: rule.limit } : { action: rule.action, ruleId: rule.id };
|
|
817
|
+
}
|
|
818
|
+
return { action: "allow", ruleId: null };
|
|
740
819
|
}
|
|
741
|
-
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
760
|
-
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
"
|
|
770
|
-
);
|
|
771
|
-
}
|
|
772
|
-
return found;
|
|
773
|
-
}
|
|
774
|
-
async function listProjects(ctx, workspaceId) {
|
|
775
|
-
const res = await ctx.api().get(`/api/workspaces/${enc(workspaceId)}/projects`);
|
|
776
|
-
return res.projects ?? [];
|
|
777
|
-
}
|
|
778
|
-
function describeProjects(list) {
|
|
779
|
-
return list.map((p) => ` ${p.id} ${p.name} ${p.domain} ${p.publicKey}`).join("\n");
|
|
780
|
-
}
|
|
781
|
-
async function resolveProject(ctx) {
|
|
782
|
-
const ref = ctx.projectRef?.trim();
|
|
783
|
-
if (ref?.startsWith("prj_")) {
|
|
784
|
-
try {
|
|
785
|
-
const res = await ctx.api().get(`/api/projects/${enc(ref)}`);
|
|
786
|
-
return res.project;
|
|
787
|
-
} catch (error) {
|
|
788
|
-
if (error instanceof ApiError && error.status === 404) {
|
|
789
|
-
throw new CliError(`Project ${ref} was not found, or you cannot see it.`, "PROJECT_NOT_FOUND");
|
|
790
|
-
}
|
|
791
|
-
throw error;
|
|
792
|
-
}
|
|
793
|
-
}
|
|
794
|
-
const workspaces = ctx.workspaceRef ? [await resolveWorkspace(ctx)] : await listWorkspaces(ctx);
|
|
795
|
-
const projects = (await Promise.all(workspaces.map((w) => listProjects(ctx, w.id)))).flat();
|
|
796
|
-
if (!ref) {
|
|
797
|
-
if (projects.length === 1) return projects[0];
|
|
798
|
-
if (projects.length === 0) {
|
|
799
|
-
throw new CliError(
|
|
800
|
-
"There are no projects yet. Create one: littlefriend projects create --name <name> --domain <host>",
|
|
801
|
-
"NO_PROJECT"
|
|
802
|
-
);
|
|
803
|
-
}
|
|
804
|
-
throw usageError(
|
|
805
|
-
`There are ${projects.length} projects. Pass --project <id|site key|name|domain>:
|
|
806
|
-
${describeProjects(projects)}`
|
|
807
|
-
);
|
|
808
|
-
}
|
|
809
|
-
const lower3 = ref.toLowerCase();
|
|
810
|
-
const byKey = projects.filter((p) => p.publicKey === ref);
|
|
811
|
-
const byName = projects.filter((p) => p.name.toLowerCase() === lower3);
|
|
812
|
-
const byDomain = projects.filter((p) => p.domain === lower3);
|
|
813
|
-
const matches = byKey.length > 0 ? byKey : byName.length > 0 ? byName : byDomain;
|
|
814
|
-
if (matches.length === 1) return matches[0];
|
|
815
|
-
if (matches.length > 1) {
|
|
816
|
-
throw usageError(`More than one project matches ${ref}. Use its id:
|
|
817
|
-
${describeProjects(matches)}`);
|
|
818
|
-
}
|
|
819
|
-
throw new CliError(
|
|
820
|
-
projects.length > 0 ? `No project matches ${ref}. Projects you can see:
|
|
821
|
-
${describeProjects(projects)}` : `No project matches ${ref}, and there are no projects yet.`,
|
|
822
|
-
"PROJECT_NOT_FOUND"
|
|
823
|
-
);
|
|
824
|
-
}
|
|
825
|
-
function projectPath(project, rest = "") {
|
|
826
|
-
return `/api/projects/${enc(project.id)}${rest}`;
|
|
827
|
-
}
|
|
828
|
-
|
|
829
|
-
// src/commands/auth.ts
|
|
830
|
-
async function lookupEmail(io, origin, accessToken) {
|
|
831
|
-
try {
|
|
832
|
-
const res = await io.fetch(`${origin}/api/workspaces`, {
|
|
833
|
-
headers: {
|
|
834
|
-
authorization: `Bearer ${accessToken}`,
|
|
835
|
-
accept: "application/json",
|
|
836
|
-
"user-agent": USER_AGENT
|
|
837
|
-
}
|
|
838
|
-
});
|
|
839
|
-
if (!res.ok) {
|
|
840
|
-
await res.body?.cancel();
|
|
841
|
-
return null;
|
|
842
|
-
}
|
|
843
|
-
const body = await res.json();
|
|
844
|
-
return typeof body?.user?.email === "string" ? body.user.email : null;
|
|
845
|
-
} catch {
|
|
846
|
-
return null;
|
|
847
|
-
}
|
|
848
|
-
}
|
|
849
|
-
var login = {
|
|
850
|
-
path: ["login"],
|
|
851
|
-
summary: "Sign in with your browser (once per machine)",
|
|
852
|
-
usage: "[--no-browser]",
|
|
853
|
-
example: "littlefriend login",
|
|
854
|
-
options: { "no-browser": { type: "boolean" } },
|
|
855
|
-
async run(ctx) {
|
|
856
|
-
const io = ctx.io;
|
|
857
|
-
const origin = ctx.origin();
|
|
858
|
-
if (io.env.LITTLEFRIEND_TOKEN?.trim()) {
|
|
859
|
-
ctx.note("LITTLEFRIEND_TOKEN is set, so other commands keep using it instead of this sign-in.");
|
|
860
|
-
}
|
|
861
|
-
const meta = await discover(io, origin);
|
|
862
|
-
const previous = loadEntry(io, origin);
|
|
863
|
-
const clientId = previous?.clientId ?? await registerClient(io, meta);
|
|
864
|
-
const { verifier, challenge } = pkcePair();
|
|
865
|
-
const state = randomState();
|
|
866
|
-
const loop = await startLoopback({
|
|
867
|
-
state,
|
|
868
|
-
issuer: meta.authorization_response_iss_parameter_supported ? meta.issuer : null,
|
|
869
|
-
timeoutMs: io.loginTimeoutMs
|
|
870
|
-
});
|
|
871
|
-
let code;
|
|
872
|
-
try {
|
|
873
|
-
const url = authorizationUrl(meta, { clientId, redirectUri: loop.redirectUri, challenge, state });
|
|
874
|
-
const minutes = Math.max(1, Math.round(io.loginTimeoutMs / 6e4));
|
|
875
|
-
ctx.note(
|
|
876
|
-
lines(
|
|
877
|
-
flag(ctx, "no-browser") ? "Open this URL in your browser to sign in to Little Friend:" : "Opening your browser to sign in to Little Friend. If it does not open, visit this URL:",
|
|
878
|
-
"",
|
|
879
|
-
` ${url}`,
|
|
880
|
-
"",
|
|
881
|
-
`Waiting for you to approve (up to ${minutes} minute${minutes === 1 ? "" : "s"})...`
|
|
882
|
-
)
|
|
883
|
-
);
|
|
884
|
-
if (!flag(ctx, "no-browser")) io.openUrl(url);
|
|
885
|
-
code = await loop.code;
|
|
886
|
-
} finally {
|
|
887
|
-
await loop.close();
|
|
820
|
+
var DOOR_PRESET_NAMES = [
|
|
821
|
+
"open",
|
|
822
|
+
"no_training",
|
|
823
|
+
"verified_only",
|
|
824
|
+
"commerce"
|
|
825
|
+
];
|
|
826
|
+
var AGENT_CLASSES = [
|
|
827
|
+
"search_crawler",
|
|
828
|
+
"ai_crawler",
|
|
829
|
+
"ai_agent",
|
|
830
|
+
"monitor",
|
|
831
|
+
"preview",
|
|
832
|
+
"automation",
|
|
833
|
+
"unknown"
|
|
834
|
+
];
|
|
835
|
+
var BLOCK_SPOOFED = {
|
|
836
|
+
id: "spoofed",
|
|
837
|
+
label: "Block agents that fail their operator check",
|
|
838
|
+
match: { spoofed: true },
|
|
839
|
+
action: "block"
|
|
840
|
+
};
|
|
841
|
+
var DOOR_PRESETS = {
|
|
842
|
+
open: [],
|
|
843
|
+
no_training: [
|
|
844
|
+
{
|
|
845
|
+
id: "no-training",
|
|
846
|
+
label: "Block AI training crawlers",
|
|
847
|
+
match: { purposes: ["ai_training"] },
|
|
848
|
+
action: "block"
|
|
888
849
|
}
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
const hint = tokens.refresh_token ? "refresh_token" : "access_token";
|
|
899
|
-
await revoke(io, meta, token, hint, clientId).catch(() => {
|
|
900
|
-
});
|
|
901
|
-
throw new CliError(
|
|
902
|
-
`Little Friend at ${origin} granted "${tokens.scope}" without ${MANAGE_SCOPE}, so the CLI cannot manage projects with it. Nothing was saved.`,
|
|
903
|
-
"SCOPE_NOT_GRANTED"
|
|
904
|
-
);
|
|
850
|
+
],
|
|
851
|
+
verified_only: [
|
|
852
|
+
BLOCK_SPOOFED,
|
|
853
|
+
{
|
|
854
|
+
id: "unverified",
|
|
855
|
+
label: "Slow down agents that are not verified",
|
|
856
|
+
match: { classes: AGENT_CLASSES, evidenceBelow: "network_verified" },
|
|
857
|
+
action: "limit",
|
|
858
|
+
limit: { perMinute: 60, burst: 20 }
|
|
905
859
|
}
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
860
|
+
],
|
|
861
|
+
commerce: [
|
|
862
|
+
BLOCK_SPOOFED,
|
|
863
|
+
{
|
|
864
|
+
id: "checkout",
|
|
865
|
+
label: "Checkout and accounts need an agent acting for a signed-in user",
|
|
866
|
+
match: {
|
|
867
|
+
// Not `unknown`: a person whose browser does not look like one must still reach checkout.
|
|
868
|
+
classes: ["ai_agent", "automation"],
|
|
869
|
+
paths: ["/checkout", "/cart", "/account"],
|
|
870
|
+
principalBelow: "signed_user"
|
|
871
|
+
},
|
|
872
|
+
action: "require_signature"
|
|
919
873
|
}
|
|
920
|
-
|
|
921
|
-
data: { api: origin, email, scope: tokens.scope ?? null, expiresAt },
|
|
922
|
-
text: lines(
|
|
923
|
-
`Signed in to ${origin}${email ? ` as ${email}` : ""}.`,
|
|
924
|
-
`Saved to ${credentialsPath(io)}. The CLI refreshes the sign-in on its own.`
|
|
925
|
-
)
|
|
926
|
-
};
|
|
927
|
-
}
|
|
874
|
+
]
|
|
928
875
|
};
|
|
929
|
-
var logout = {
|
|
930
|
-
path: ["logout"],
|
|
931
|
-
summary: "Revoke this machine's sign-in and delete it",
|
|
932
|
-
usage: "",
|
|
933
|
-
example: "littlefriend logout",
|
|
934
|
-
async run(ctx) {
|
|
935
|
-
const io = ctx.io;
|
|
936
|
-
const origin = ctx.origin();
|
|
937
|
-
const entry = loadEntry(io, origin);
|
|
938
|
-
if (!entry) {
|
|
939
|
-
return {
|
|
940
|
-
data: { api: origin, signedOut: false, revoked: false },
|
|
941
|
-
text: `Not signed in to ${origin} on this machine.`
|
|
942
|
-
};
|
|
943
|
-
}
|
|
944
|
-
let problem = null;
|
|
945
|
-
try {
|
|
946
|
-
const meta = await discover(io, origin);
|
|
947
|
-
if (entry.refreshToken) await revoke(io, meta, entry.refreshToken, "refresh_token", entry.clientId);
|
|
948
|
-
else await revoke(io, meta, entry.accessToken, "access_token", entry.clientId);
|
|
949
|
-
} catch (error) {
|
|
950
|
-
problem = error instanceof Error ? error.message : String(error);
|
|
951
|
-
}
|
|
952
|
-
deleteEntry(io, origin);
|
|
953
|
-
if (problem) {
|
|
954
|
-
throw new CliError(
|
|
955
|
-
`Signed out on this machine, but Little Friend could not revoke the sign-in: ${problem} Revoke it in the dashboard under Account, Connected apps.`,
|
|
956
|
-
"REVOKE_FAILED"
|
|
957
|
-
);
|
|
958
|
-
}
|
|
959
|
-
return {
|
|
960
|
-
data: { api: origin, signedOut: true, revoked: true },
|
|
961
|
-
text: `Signed out of ${origin}. The sign-in was revoked and deleted from this machine.`
|
|
962
|
-
};
|
|
963
|
-
}
|
|
964
|
-
};
|
|
965
|
-
var whoami = {
|
|
966
|
-
path: ["whoami"],
|
|
967
|
-
summary: "Show who the CLI is signed in as, and your workspaces",
|
|
968
|
-
usage: "",
|
|
969
|
-
example: "littlefriend whoami",
|
|
970
|
-
async run(ctx) {
|
|
971
|
-
const origin = ctx.origin();
|
|
972
|
-
const session = ctx.api().session;
|
|
973
|
-
const { user, workspaces } = await accountAndWorkspaces(ctx);
|
|
974
|
-
const stored = session.stored;
|
|
975
|
-
const email = user?.email ?? stored?.user.email ?? null;
|
|
976
|
-
return {
|
|
977
|
-
data: {
|
|
978
|
-
api: origin,
|
|
979
|
-
email,
|
|
980
|
-
source: session.source,
|
|
981
|
-
expiresAt: session.source === "stored" ? stored?.expiresAt ?? null : null,
|
|
982
|
-
workspaces
|
|
983
|
-
},
|
|
984
|
-
text: lines(
|
|
985
|
-
fields([
|
|
986
|
-
["API", origin],
|
|
987
|
-
...email ? [["Signed in as", email]] : [],
|
|
988
|
-
[
|
|
989
|
-
"Credentials",
|
|
990
|
-
session.source === "env" ? "LITTLEFRIEND_TOKEN" : `${credentialsPath(ctx.io)} (refreshes on its own)`
|
|
991
|
-
],
|
|
992
|
-
...session.source === "stored" ? [["Access token expires", when(stored?.expiresAt)]] : []
|
|
993
|
-
]),
|
|
994
|
-
"",
|
|
995
|
-
workspaces.length > 0 ? table(
|
|
996
|
-
["WORKSPACE", "SLUG", "NAME", "ROLE"],
|
|
997
|
-
workspaces.map((w) => [w.id, w.slug, w.name, w.role])
|
|
998
|
-
) : "No workspaces yet."
|
|
999
|
-
)
|
|
1000
|
-
};
|
|
1001
|
-
}
|
|
1002
|
-
};
|
|
1003
|
-
var authCommands = [login, logout, whoami];
|
|
1004
876
|
|
|
1005
|
-
// ../
|
|
1006
|
-
var
|
|
1007
|
-
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
877
|
+
// ../contract/dist/origins.js
|
|
878
|
+
var REFUSED_ORIGIN_SCHEMES = [
|
|
879
|
+
// Script and inline content.
|
|
880
|
+
"javascript",
|
|
881
|
+
"vbscript",
|
|
882
|
+
"data",
|
|
883
|
+
"blob",
|
|
884
|
+
"filesystem",
|
|
885
|
+
// Files and blank pages, which send `Origin: null`.
|
|
886
|
+
"file",
|
|
887
|
+
"about",
|
|
888
|
+
"content",
|
|
889
|
+
// Browser pages and extensions.
|
|
890
|
+
"chrome",
|
|
891
|
+
"chrome-extension",
|
|
892
|
+
"chrome-untrusted",
|
|
893
|
+
"chrome-search",
|
|
894
|
+
"devtools",
|
|
895
|
+
"edge",
|
|
896
|
+
"brave",
|
|
897
|
+
"opera",
|
|
898
|
+
"vivaldi",
|
|
899
|
+
"view-source",
|
|
900
|
+
"resource",
|
|
901
|
+
"moz-extension",
|
|
902
|
+
"safari-extension",
|
|
903
|
+
"safari-web-extension",
|
|
904
|
+
"ms-browser-extension",
|
|
905
|
+
// Protocols that are never a page.
|
|
906
|
+
"ftp",
|
|
907
|
+
"ws",
|
|
908
|
+
"wss",
|
|
909
|
+
"mailto",
|
|
910
|
+
"tel",
|
|
911
|
+
"sms",
|
|
912
|
+
"intent",
|
|
913
|
+
"market"
|
|
1023
914
|
];
|
|
1024
|
-
var
|
|
1025
|
-
var IOS_WEBKIT = /\((?:iPhone|iPad|iPod)\b.*\bAppleWebKit\/[\d.]+.*\bMobile\/\w+/;
|
|
1026
|
-
var MAC_WEBKIT = /^Mozilla\/5\.0 \(Macintosh; Intel Mac OS X [\d_]+\) AppleWebKit\/[\d.]+ \(KHTML, like Gecko\)$/;
|
|
1027
|
-
function detectBrowser(ua) {
|
|
1028
|
-
for (const [re, name] of BROWSERS) {
|
|
1029
|
-
if (re.test(ua)) {
|
|
1030
|
-
if (name === "Chrome" && ANDROID_WEBVIEW.test(ua))
|
|
1031
|
-
return "Android WebView";
|
|
1032
|
-
return name;
|
|
1033
|
-
}
|
|
1034
|
-
}
|
|
1035
|
-
if (IOS_WEBKIT.test(ua))
|
|
1036
|
-
return "iOS WebView";
|
|
1037
|
-
if (MAC_WEBKIT.test(ua))
|
|
1038
|
-
return "macOS WebView";
|
|
1039
|
-
return null;
|
|
1040
|
-
}
|
|
1041
|
-
function isBrowserLike(input) {
|
|
1042
|
-
if (typeof input !== "string")
|
|
1043
|
-
return false;
|
|
1044
|
-
const ua = input.slice(0, MAX_UA_LENGTH);
|
|
1045
|
-
if (!/^Mozilla\/5\.0 \(/.test(ua) && !/^Opera\//.test(ua))
|
|
1046
|
-
return false;
|
|
1047
|
-
if (!/AppleWebKit\/|Gecko\/|Trident\/|Presto\//.test(ua))
|
|
1048
|
-
return false;
|
|
1049
|
-
return detectBrowser(ua) !== null;
|
|
1050
|
-
}
|
|
915
|
+
var REFUSED = new Set(REFUSED_ORIGIN_SCHEMES);
|
|
1051
916
|
|
|
1052
|
-
// ../
|
|
1053
|
-
var
|
|
1054
|
-
|
|
1055
|
-
|
|
1056
|
-
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
|
|
1066
|
-
|
|
1067
|
-
|
|
1068
|
-
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1112
|
-
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
|
|
1127
|
-
|
|
1128
|
-
|
|
1129
|
-
|
|
1130
|
-
|
|
1131
|
-
|
|
1132
|
-
|
|
1133
|
-
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
|
|
1144
|
-
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
|
|
1152
|
-
|
|
1153
|
-
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
|
|
1158
|
-
|
|
1159
|
-
|
|
1160
|
-
|
|
1161
|
-
|
|
1162
|
-
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
|
|
1181
|
-
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
|
|
1206
|
-
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
1214
|
-
|
|
1215
|
-
|
|
1216
|
-
|
|
1217
|
-
|
|
1218
|
-
|
|
1219
|
-
|
|
1220
|
-
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
|
|
1229
|
-
|
|
1230
|
-
|
|
1231
|
-
|
|
1232
|
-
|
|
1233
|
-
|
|
1234
|
-
|
|
1235
|
-
|
|
1236
|
-
|
|
1237
|
-
|
|
1238
|
-
|
|
1239
|
-
|
|
1240
|
-
|
|
1241
|
-
|
|
1242
|
-
|
|
1243
|
-
|
|
1244
|
-
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
1248
|
-
|
|
1249
|
-
|
|
1250
|
-
|
|
1251
|
-
|
|
1252
|
-
|
|
1253
|
-
|
|
1254
|
-
|
|
1255
|
-
|
|
1256
|
-
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
{
|
|
1291
|
-
|
|
1292
|
-
|
|
1293
|
-
|
|
1294
|
-
|
|
1295
|
-
|
|
1296
|
-
|
|
1297
|
-
|
|
1298
|
-
|
|
1299
|
-
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
|
|
1303
|
-
|
|
1304
|
-
|
|
1305
|
-
|
|
1306
|
-
|
|
1307
|
-
|
|
1308
|
-
|
|
1309
|
-
|
|
1310
|
-
|
|
1311
|
-
|
|
1312
|
-
|
|
1313
|
-
|
|
1314
|
-
|
|
1315
|
-
|
|
1316
|
-
|
|
1317
|
-
|
|
1318
|
-
|
|
1319
|
-
|
|
1320
|
-
}
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1325
|
-
|
|
1326
|
-
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1330
|
-
|
|
1331
|
-
|
|
917
|
+
// ../contract/dist/referrers.js
|
|
918
|
+
var ANDROID_APP_HOSTS = Object.freeze({
|
|
919
|
+
"com.google.android.gm": "mail.google.com",
|
|
920
|
+
"com.microsoft.office.outlook": "outlook.live.com",
|
|
921
|
+
"com.yahoo.mobile.client.android.mail": "mail.yahoo.com",
|
|
922
|
+
"ch.protonmail.android": "mail.proton.me",
|
|
923
|
+
"com.google.android.googlequicksearchbox": "google.com",
|
|
924
|
+
"com.linkedin.android": "linkedin.com",
|
|
925
|
+
"com.facebook.katana": "facebook.com",
|
|
926
|
+
"com.facebook.orca": "facebook.com",
|
|
927
|
+
"com.facebook.lite": "facebook.com",
|
|
928
|
+
"com.instagram.android": "instagram.com",
|
|
929
|
+
"com.instagram.barcelona": "threads.net",
|
|
930
|
+
"com.twitter.android": "x.com",
|
|
931
|
+
"com.reddit.frontpage": "reddit.com",
|
|
932
|
+
"com.google.android.youtube": "youtube.com",
|
|
933
|
+
"com.zhiliaoapp.musically": "tiktok.com",
|
|
934
|
+
"xyz.blueskyweb.app": "bsky.app",
|
|
935
|
+
"com.pinterest": "pinterest.com",
|
|
936
|
+
"com.slack": "slack.com",
|
|
937
|
+
"com.discord": "discord.com",
|
|
938
|
+
"org.telegram.messenger": "t.me",
|
|
939
|
+
"com.whatsapp": "whatsapp.com",
|
|
940
|
+
"com.openai.chatgpt": "chatgpt.com",
|
|
941
|
+
"ai.perplexity.app.android": "perplexity.ai",
|
|
942
|
+
"com.anthropic.claude": "claude.ai",
|
|
943
|
+
"com.google.android.apps.bard": "gemini.google.com",
|
|
944
|
+
"com.microsoft.copilot": "copilot.microsoft.com"
|
|
945
|
+
});
|
|
946
|
+
|
|
947
|
+
// ../contract/dist/replay.js
|
|
948
|
+
var REPLAY_MASK_CHAR = "\u2588";
|
|
949
|
+
var REPLAY_ALWAYS_BLOCKED = [
|
|
950
|
+
"img",
|
|
951
|
+
"picture",
|
|
952
|
+
"video",
|
|
953
|
+
"audio",
|
|
954
|
+
"iframe",
|
|
955
|
+
"object",
|
|
956
|
+
"embed",
|
|
957
|
+
"canvas",
|
|
958
|
+
"svg image"
|
|
959
|
+
];
|
|
960
|
+
var REPLAY_KEPT_ATTRIBUTES = [
|
|
961
|
+
"class",
|
|
962
|
+
"id",
|
|
963
|
+
"style",
|
|
964
|
+
"role",
|
|
965
|
+
"type",
|
|
966
|
+
"name",
|
|
967
|
+
"for",
|
|
968
|
+
"rel",
|
|
969
|
+
"media",
|
|
970
|
+
"href",
|
|
971
|
+
"width",
|
|
972
|
+
"height",
|
|
973
|
+
"colspan",
|
|
974
|
+
"rowspan",
|
|
975
|
+
"dir",
|
|
976
|
+
"lang",
|
|
977
|
+
"hidden",
|
|
978
|
+
"open",
|
|
979
|
+
"disabled",
|
|
980
|
+
"readonly",
|
|
981
|
+
"multiple",
|
|
982
|
+
"viewbox",
|
|
983
|
+
"d",
|
|
984
|
+
"fill",
|
|
985
|
+
"stroke",
|
|
986
|
+
"stroke-width",
|
|
987
|
+
"points",
|
|
988
|
+
"x",
|
|
989
|
+
"y",
|
|
990
|
+
"cx",
|
|
991
|
+
"cy",
|
|
992
|
+
"r",
|
|
993
|
+
"rx",
|
|
994
|
+
"ry",
|
|
995
|
+
"transform",
|
|
996
|
+
"aria-hidden",
|
|
997
|
+
"aria-expanded",
|
|
998
|
+
"aria-current",
|
|
999
|
+
"aria-disabled",
|
|
1000
|
+
"aria-pressed"
|
|
1001
|
+
];
|
|
1002
|
+
var REPLAY_TEXT_ATTRIBUTES = [
|
|
1003
|
+
"title",
|
|
1004
|
+
"alt",
|
|
1005
|
+
"placeholder",
|
|
1006
|
+
"aria-label",
|
|
1007
|
+
"aria-description"
|
|
1008
|
+
];
|
|
1009
|
+
var REPLAY_UNMASK_PRESET = [
|
|
1010
|
+
"nav",
|
|
1011
|
+
"header",
|
|
1012
|
+
"footer",
|
|
1013
|
+
"h1",
|
|
1014
|
+
"h2",
|
|
1015
|
+
"h3",
|
|
1016
|
+
"h4",
|
|
1017
|
+
"button",
|
|
1018
|
+
"label",
|
|
1019
|
+
"th",
|
|
1020
|
+
"legend",
|
|
1021
|
+
"[role=button]",
|
|
1022
|
+
"[role=tab]",
|
|
1023
|
+
"[role=menuitem]"
|
|
1024
|
+
];
|
|
1025
|
+
var REPLAY_LIMITS = {
|
|
1026
|
+
/** One POST body, compressed. Keepalive requests on page hide must stay under 64 KB. */
|
|
1027
|
+
maxChunkBytes: 256 * 1024,
|
|
1028
|
+
/** One chunk after gunzip. Anything larger is refused before it is parsed. */
|
|
1029
|
+
maxChunkRawBytes: 4 * 1024 * 1024,
|
|
1030
|
+
/**
|
|
1031
|
+
* A chunk that carries a full snapshot may be larger: a heavy page's first snapshot with its
|
|
1032
|
+
* inlined stylesheets can pass maxChunkBytes on its own. Every other chunk keeps the normal limits.
|
|
1033
|
+
*/
|
|
1034
|
+
maxSnapshotChunkBytes: 1024 * 1024,
|
|
1035
|
+
maxSnapshotChunkRawBytes: 10 * 1024 * 1024,
|
|
1036
|
+
maxChunkEvents: 5e3,
|
|
1037
|
+
/** Stored bytes per session, compressed. Past it the collector answers 202 and drops. */
|
|
1038
|
+
maxSessionBytes: 12 * 1024 * 1024,
|
|
1039
|
+
/** Recorded time per page load. The recorder stops after it. */
|
|
1040
|
+
maxPageMs: 60 * 60 * 1e3,
|
|
1041
|
+
/** The recorder flushes a chunk this often, or sooner when it reaches flushBytes raw. */
|
|
1042
|
+
flushMs: 1e4,
|
|
1043
|
+
flushBytes: 128 * 1024,
|
|
1044
|
+
/** Two interactions closer than this are one stretch of activity. Longer gaps are skippable. */
|
|
1045
|
+
idleGapMs: 5e3,
|
|
1046
|
+
/** A rage click: this many clicks within rageWindowMs inside rageRadiusPx. */
|
|
1047
|
+
rageClicks: 3,
|
|
1048
|
+
rageWindowMs: 1e3,
|
|
1049
|
+
rageRadiusPx: 30,
|
|
1050
|
+
maxSelectors: 50,
|
|
1051
|
+
maxSelectorLength: 200,
|
|
1052
|
+
maxExcludeRoutes: 50,
|
|
1053
|
+
/**
|
|
1054
|
+
* A recording with at least this many clicks is kept when its session ends, whatever its
|
|
1055
|
+
* active time: a short visit of a few taps on a phone is still worth watching.
|
|
1056
|
+
*/
|
|
1057
|
+
keepWithClicks: 3,
|
|
1058
|
+
/**
|
|
1059
|
+
* Recordings a workspace may start per UTC calendar month without a card on file. The
|
|
1060
|
+
* collector and the API read REPLAY_FREE_RECORDINGS_PER_MONTH first, so both agree.
|
|
1061
|
+
*/
|
|
1062
|
+
freeRecordingsPerMonth: 10
|
|
1063
|
+
};
|
|
1064
|
+
|
|
1065
|
+
// ../contract/dist/replay-sanitize.js
|
|
1066
|
+
var REPLAY_SOURCE = {
|
|
1067
|
+
Mutation: 0,
|
|
1068
|
+
MouseMove: 1,
|
|
1069
|
+
MouseInteraction: 2,
|
|
1070
|
+
Scroll: 3,
|
|
1071
|
+
ViewportResize: 4,
|
|
1072
|
+
Input: 5,
|
|
1073
|
+
TouchMove: 6,
|
|
1074
|
+
MediaInteraction: 7,
|
|
1075
|
+
StyleSheetRule: 8,
|
|
1076
|
+
CanvasMutation: 9,
|
|
1077
|
+
Font: 10,
|
|
1078
|
+
Log: 11,
|
|
1079
|
+
Drag: 12,
|
|
1080
|
+
StyleDeclaration: 13,
|
|
1081
|
+
Selection: 14,
|
|
1082
|
+
AdoptedStyleSheet: 15,
|
|
1083
|
+
CustomElement: 16
|
|
1084
|
+
};
|
|
1085
|
+
var REPLAY_MAX_SCANNED_TEXT = 32 * 1024;
|
|
1086
|
+
var MASKED_RE = new RegExp(`^[\\s${REPLAY_MASK_CHAR}]*$`, "u");
|
|
1087
|
+
var BLOCKED_TAGS = /* @__PURE__ */ new Set([
|
|
1088
|
+
...REPLAY_ALWAYS_BLOCKED.map((s) => s.split(/\s+/).pop()),
|
|
1089
|
+
"frame",
|
|
1090
|
+
"frameset",
|
|
1091
|
+
"applet",
|
|
1092
|
+
"portal",
|
|
1093
|
+
"fencedframe"
|
|
1094
|
+
]);
|
|
1095
|
+
var KEPT = new Set(REPLAY_KEPT_ATTRIBUTES);
|
|
1096
|
+
var TEXT_ATTRS = new Set(REPLAY_TEXT_ATTRIBUTES);
|
|
1097
|
+
var INTERACTION_SOURCES = /* @__PURE__ */ new Set([
|
|
1098
|
+
REPLAY_SOURCE.MouseMove,
|
|
1099
|
+
REPLAY_SOURCE.MouseInteraction,
|
|
1100
|
+
REPLAY_SOURCE.Scroll,
|
|
1101
|
+
REPLAY_SOURCE.ViewportResize,
|
|
1102
|
+
REPLAY_SOURCE.Input,
|
|
1103
|
+
REPLAY_SOURCE.TouchMove,
|
|
1104
|
+
REPLAY_SOURCE.Drag,
|
|
1105
|
+
REPLAY_SOURCE.Selection
|
|
1106
|
+
]);
|
|
1107
|
+
|
|
1108
|
+
// ../contract/dist/reports.js
|
|
1109
|
+
var FUNNEL_MAX_STEPS = 8;
|
|
1110
|
+
|
|
1111
|
+
// src/command.ts
|
|
1112
|
+
function str(ctx, name) {
|
|
1113
|
+
const v = ctx.values[name];
|
|
1114
|
+
if (Array.isArray(v)) {
|
|
1115
|
+
const last = v[v.length - 1];
|
|
1116
|
+
return typeof last === "string" ? last : void 0;
|
|
1117
|
+
}
|
|
1118
|
+
return typeof v === "string" ? v : void 0;
|
|
1119
|
+
}
|
|
1120
|
+
function flag(ctx, name) {
|
|
1121
|
+
return ctx.values[name] === true;
|
|
1122
|
+
}
|
|
1123
|
+
function many(ctx, name) {
|
|
1124
|
+
const v = ctx.values[name];
|
|
1125
|
+
if (Array.isArray(v)) return v.filter((x) => typeof x === "string");
|
|
1126
|
+
return typeof v === "string" ? [v] : [];
|
|
1127
|
+
}
|
|
1128
|
+
function has(ctx, name) {
|
|
1129
|
+
return ctx.values[name] !== void 0;
|
|
1130
|
+
}
|
|
1131
|
+
function requiredStr(ctx, name, what) {
|
|
1132
|
+
const v = str(ctx, name)?.trim();
|
|
1133
|
+
if (!v) throw usageError(`--${name} is required: ${what}.`);
|
|
1134
|
+
return v;
|
|
1135
|
+
}
|
|
1136
|
+
function intOpt(ctx, name, min, max) {
|
|
1137
|
+
const raw = str(ctx, name);
|
|
1138
|
+
if (raw === void 0) return void 0;
|
|
1139
|
+
const n = Number(raw);
|
|
1140
|
+
if (!/^-?\d+$/.test(raw.trim()) || n < min || n > max) {
|
|
1141
|
+
throw usageError(`--${name} must be a whole number from ${min} to ${max}. Got ${raw}.`);
|
|
1142
|
+
}
|
|
1143
|
+
return n;
|
|
1144
|
+
}
|
|
1145
|
+
function oneOf(ctx, name, allowed) {
|
|
1146
|
+
const raw = str(ctx, name);
|
|
1147
|
+
if (raw === void 0) return void 0;
|
|
1148
|
+
if (!allowed.includes(raw)) {
|
|
1149
|
+
throw usageError(`--${name} must be ${orList(allowed)}. Got ${raw}.`);
|
|
1150
|
+
}
|
|
1151
|
+
return raw;
|
|
1152
|
+
}
|
|
1153
|
+
function joinList(items, word) {
|
|
1154
|
+
if (items.length <= 1) return items.join("");
|
|
1155
|
+
return `${items.slice(0, -1).join(", ")} ${word} ${items[items.length - 1]}`;
|
|
1156
|
+
}
|
|
1157
|
+
var orList = (items) => joinList(items, "or");
|
|
1158
|
+
var andList = (items) => joinList(items, "and");
|
|
1159
|
+
function requireYes(ctx, what) {
|
|
1160
|
+
if (!flag(ctx, "yes")) throw usageError(`This ${what}. Add --yes to confirm.`);
|
|
1161
|
+
}
|
|
1162
|
+
var YES = { yes: { type: "boolean" } };
|
|
1163
|
+
function arg(ctx, index, what) {
|
|
1164
|
+
const v = ctx.args[index];
|
|
1165
|
+
if (!v) throw usageError(`Missing ${what}.`);
|
|
1166
|
+
return v;
|
|
1167
|
+
}
|
|
1168
|
+
var enc = encodeURIComponent;
|
|
1169
|
+
|
|
1170
|
+
// src/format.ts
|
|
1171
|
+
function cell(value) {
|
|
1172
|
+
if (value === null || value === void 0) return "";
|
|
1173
|
+
if (typeof value === "boolean") return value ? "yes" : "no";
|
|
1174
|
+
return String(value);
|
|
1175
|
+
}
|
|
1176
|
+
function table(headers, rows) {
|
|
1177
|
+
const all = [headers, ...rows.map((r) => r.map(cell))];
|
|
1178
|
+
const widths = headers.map((_, i) => Math.max(...all.map((r) => (r[i] ?? "").length)));
|
|
1179
|
+
return all.map(
|
|
1180
|
+
(r) => r.map((c, i) => i === r.length - 1 ? c : c.padEnd(widths[i] ?? 0)).join(" ").trimEnd()
|
|
1181
|
+
).join("\n");
|
|
1182
|
+
}
|
|
1183
|
+
function fields(pairs) {
|
|
1184
|
+
const width = Math.max(...pairs.map(([k]) => k.length));
|
|
1185
|
+
return pairs.map(([k, v]) => `${`${k}:`.padEnd(width + 2)}${cell(v)}`.trimEnd()).join("\n");
|
|
1186
|
+
}
|
|
1187
|
+
function when(iso) {
|
|
1188
|
+
if (!iso) return "";
|
|
1189
|
+
const d = new Date(iso);
|
|
1190
|
+
if (Number.isNaN(d.getTime())) return iso;
|
|
1191
|
+
return `${d.toISOString().slice(0, 16).replace("T", " ")} UTC`;
|
|
1192
|
+
}
|
|
1193
|
+
function count(n) {
|
|
1194
|
+
return typeof n === "number" && Number.isFinite(n) ? n.toLocaleString("en-US") : "";
|
|
1195
|
+
}
|
|
1196
|
+
function percent(fraction) {
|
|
1197
|
+
if (typeof fraction !== "number" || !Number.isFinite(fraction)) return "";
|
|
1198
|
+
return `${(fraction * 100).toFixed(1).replace(/\.0$/, "")}%`;
|
|
1199
|
+
}
|
|
1200
|
+
function duration(ms) {
|
|
1201
|
+
if (typeof ms !== "number" || !Number.isFinite(ms)) return "";
|
|
1202
|
+
const s = Math.round(ms / 1e3);
|
|
1203
|
+
if (s < 60) return `${s}s`;
|
|
1204
|
+
return `${Math.floor(s / 60)}m ${String(s % 60).padStart(2, "0")}s`;
|
|
1205
|
+
}
|
|
1206
|
+
function lines(...parts) {
|
|
1207
|
+
return parts.filter((p) => typeof p === "string").join("\n");
|
|
1208
|
+
}
|
|
1209
|
+
function plural(n, one, many2 = `${one}s`) {
|
|
1210
|
+
return `${n} ${n === 1 ? one : many2}`;
|
|
1211
|
+
}
|
|
1212
|
+
|
|
1213
|
+
// src/resolve.ts
|
|
1214
|
+
async function accountAndWorkspaces(ctx) {
|
|
1215
|
+
const res = await ctx.api().get("/api/workspaces");
|
|
1216
|
+
return { user: res.user ?? null, workspaces: res.workspaces ?? [] };
|
|
1217
|
+
}
|
|
1218
|
+
async function listWorkspaces(ctx) {
|
|
1219
|
+
return (await accountAndWorkspaces(ctx)).workspaces;
|
|
1220
|
+
}
|
|
1221
|
+
function describeWorkspaces(list) {
|
|
1222
|
+
return list.map((w) => ` ${w.slug} ${w.name} (${w.id})`).join("\n");
|
|
1223
|
+
}
|
|
1224
|
+
async function resolveWorkspace(ctx) {
|
|
1225
|
+
const all = await listWorkspaces(ctx);
|
|
1226
|
+
const ref = ctx.workspaceRef?.trim();
|
|
1227
|
+
if (!ref) {
|
|
1228
|
+
if (all.length === 1) return all[0];
|
|
1229
|
+
if (all.length === 0) throw new CliError("You are not a member of any workspace yet.", "NO_WORKSPACE");
|
|
1230
|
+
throw usageError(
|
|
1231
|
+
`You belong to ${all.length} workspaces. Pass --workspace <id|slug|name>:
|
|
1232
|
+
${describeWorkspaces(all)}`
|
|
1233
|
+
);
|
|
1234
|
+
}
|
|
1235
|
+
const lower3 = ref.toLowerCase();
|
|
1236
|
+
const found = all.find((w) => w.id === ref) ?? all.find((w) => w.slug === lower3) ?? all.filter((w) => w.name.toLowerCase() === lower3);
|
|
1237
|
+
if (Array.isArray(found)) {
|
|
1238
|
+
if (found.length === 1) return found[0];
|
|
1239
|
+
if (found.length > 1) {
|
|
1240
|
+
throw usageError(`More than one workspace is named ${ref}. Use its id:
|
|
1241
|
+
${describeWorkspaces(found)}`);
|
|
1242
|
+
}
|
|
1243
|
+
throw new CliError(
|
|
1244
|
+
`No workspace matches ${ref}. Your workspaces:
|
|
1245
|
+
${describeWorkspaces(all)}`,
|
|
1246
|
+
"WORKSPACE_NOT_FOUND"
|
|
1247
|
+
);
|
|
1248
|
+
}
|
|
1249
|
+
return found;
|
|
1250
|
+
}
|
|
1251
|
+
async function listProjects(ctx, workspaceId) {
|
|
1252
|
+
const res = await ctx.api().get(`/api/workspaces/${enc(workspaceId)}/projects`);
|
|
1253
|
+
return res.projects ?? [];
|
|
1254
|
+
}
|
|
1255
|
+
function describeProjects(list) {
|
|
1256
|
+
return list.map((p) => ` ${p.id} ${p.name} ${p.domain} ${p.publicKey}`).join("\n");
|
|
1257
|
+
}
|
|
1258
|
+
async function resolveProject(ctx) {
|
|
1259
|
+
const ref = ctx.projectRef?.trim();
|
|
1260
|
+
if (ref?.startsWith("prj_")) {
|
|
1261
|
+
try {
|
|
1262
|
+
const res = await ctx.api().get(`/api/projects/${enc(ref)}`);
|
|
1263
|
+
return res.project;
|
|
1264
|
+
} catch (error) {
|
|
1265
|
+
if (error instanceof ApiError && error.status === 404) {
|
|
1266
|
+
throw new CliError(`Project ${ref} was not found, or you cannot see it.`, "PROJECT_NOT_FOUND");
|
|
1267
|
+
}
|
|
1268
|
+
throw error;
|
|
1269
|
+
}
|
|
1270
|
+
}
|
|
1271
|
+
const workspaces = ctx.workspaceRef ? [await resolveWorkspace(ctx)] : await listWorkspaces(ctx);
|
|
1272
|
+
const projects = (await Promise.all(workspaces.map((w) => listProjects(ctx, w.id)))).flat();
|
|
1273
|
+
if (!ref) {
|
|
1274
|
+
if (projects.length === 1) return projects[0];
|
|
1275
|
+
if (projects.length === 0) {
|
|
1276
|
+
throw new CliError(
|
|
1277
|
+
"There are no projects yet. Create one: littlefriend projects create --name <name> --domain <host>",
|
|
1278
|
+
"NO_PROJECT"
|
|
1279
|
+
);
|
|
1280
|
+
}
|
|
1281
|
+
throw usageError(
|
|
1282
|
+
`There are ${projects.length} projects. Pass --project <id|site key|name|domain>:
|
|
1283
|
+
${describeProjects(projects)}`
|
|
1284
|
+
);
|
|
1285
|
+
}
|
|
1286
|
+
const lower3 = ref.toLowerCase();
|
|
1287
|
+
const byKey = projects.filter((p) => p.publicKey === ref);
|
|
1288
|
+
const byName = projects.filter((p) => p.name.toLowerCase() === lower3);
|
|
1289
|
+
const byDomain = projects.filter((p) => p.domain === lower3);
|
|
1290
|
+
const matches = byKey.length > 0 ? byKey : byName.length > 0 ? byName : byDomain;
|
|
1291
|
+
if (matches.length === 1) return matches[0];
|
|
1292
|
+
if (matches.length > 1) {
|
|
1293
|
+
throw usageError(`More than one project matches ${ref}. Use its id:
|
|
1294
|
+
${describeProjects(matches)}`);
|
|
1295
|
+
}
|
|
1296
|
+
throw new CliError(
|
|
1297
|
+
projects.length > 0 ? `No project matches ${ref}. Projects you can see:
|
|
1298
|
+
${describeProjects(projects)}` : `No project matches ${ref}, and there are no projects yet.`,
|
|
1299
|
+
"PROJECT_NOT_FOUND"
|
|
1300
|
+
);
|
|
1301
|
+
}
|
|
1302
|
+
function projectPath(project, rest = "") {
|
|
1303
|
+
return `/api/projects/${enc(project.id)}${rest}`;
|
|
1304
|
+
}
|
|
1305
|
+
|
|
1306
|
+
// src/commands/apps.ts
|
|
1307
|
+
var APP_ID_HINT = "Use the app's bundle id or package name, like com.acme.shop.";
|
|
1308
|
+
var appsOf = (p) => [...p.allowedApps ?? []];
|
|
1309
|
+
function listText(project, apps) {
|
|
1310
|
+
return apps.length > 0 ? apps.join("\n") : `${project.name} lists no apps yet. Add one: littlefriend apps add com.acme.shop`;
|
|
1311
|
+
}
|
|
1312
|
+
async function save(ctx, project, allowedApps) {
|
|
1313
|
+
const { project: saved } = await ctx.api().patch(projectPath(project), { allowedApps });
|
|
1314
|
+
return appsOf(saved);
|
|
1315
|
+
}
|
|
1316
|
+
var appsList = {
|
|
1317
|
+
path: ["apps", "list"],
|
|
1318
|
+
summary: "List the apps allowed to send data to the project",
|
|
1319
|
+
usage: "",
|
|
1320
|
+
example: "littlefriend apps list",
|
|
1321
|
+
async run(ctx) {
|
|
1322
|
+
const project = await resolveProject(ctx);
|
|
1323
|
+
const allowedApps = appsOf(project);
|
|
1324
|
+
return { data: { allowedApps }, text: listText(project, allowedApps) };
|
|
1325
|
+
}
|
|
1326
|
+
};
|
|
1327
|
+
var appsAdd = {
|
|
1328
|
+
path: ["apps", "add"],
|
|
1329
|
+
summary: "Allow an app by its bundle id or package name",
|
|
1330
|
+
usage: "<app-id>",
|
|
1331
|
+
example: "littlefriend apps add com.acme.shop",
|
|
1332
|
+
args: { min: 1, max: 1 },
|
|
1333
|
+
async run(ctx) {
|
|
1334
|
+
const raw = arg(ctx, 0, "the app id, such as com.acme.shop");
|
|
1335
|
+
const id = normalizeAppId(raw);
|
|
1336
|
+
if (!id) throw usageError(`${raw} is not an app id. ${APP_ID_HINT}`);
|
|
1337
|
+
const project = await resolveProject(ctx);
|
|
1338
|
+
const current = appsOf(project);
|
|
1339
|
+
if (current.some((a) => a.toLowerCase() === id)) {
|
|
1340
|
+
return {
|
|
1341
|
+
data: { allowedApps: current },
|
|
1342
|
+
text: lines(`${project.name} already allows ${id}.`, "", listText(project, current))
|
|
1343
|
+
};
|
|
1344
|
+
}
|
|
1345
|
+
const allowedApps = await save(ctx, project, [...current, id]);
|
|
1346
|
+
return {
|
|
1347
|
+
data: { allowedApps },
|
|
1348
|
+
text: lines(`Added ${id} to the allowed apps of ${project.name}.`, "", listText(project, allowedApps))
|
|
1349
|
+
};
|
|
1350
|
+
}
|
|
1351
|
+
};
|
|
1352
|
+
var appsRemove = {
|
|
1353
|
+
path: ["apps", "remove"],
|
|
1354
|
+
summary: "Stop accepting data from an app",
|
|
1355
|
+
usage: "<app-id>",
|
|
1356
|
+
example: "littlefriend apps remove com.acme.shop",
|
|
1357
|
+
args: { min: 1, max: 1 },
|
|
1358
|
+
async run(ctx) {
|
|
1359
|
+
const id = arg(ctx, 0, "the app id, such as com.acme.shop").toLowerCase();
|
|
1360
|
+
const project = await resolveProject(ctx);
|
|
1361
|
+
const current = appsOf(project);
|
|
1362
|
+
const next = current.filter((a) => a.toLowerCase() !== id);
|
|
1363
|
+
if (next.length === current.length) {
|
|
1364
|
+
return {
|
|
1365
|
+
data: { allowedApps: current },
|
|
1366
|
+
text: lines(`${project.name} does not list ${id}.`, "", listText(project, current))
|
|
1367
|
+
};
|
|
1368
|
+
}
|
|
1369
|
+
const allowedApps = await save(ctx, project, next);
|
|
1370
|
+
return {
|
|
1371
|
+
data: { allowedApps },
|
|
1372
|
+
text: lines(
|
|
1373
|
+
`Removed ${id} from the allowed apps of ${project.name}.`,
|
|
1374
|
+
"",
|
|
1375
|
+
listText(project, allowedApps)
|
|
1376
|
+
)
|
|
1377
|
+
};
|
|
1378
|
+
}
|
|
1379
|
+
};
|
|
1380
|
+
var appsCommands = [appsList, appsAdd, appsRemove];
|
|
1381
|
+
|
|
1382
|
+
// src/commands/auth.ts
|
|
1383
|
+
async function lookupEmail(io, origin, accessToken) {
|
|
1384
|
+
try {
|
|
1385
|
+
const res = await io.fetch(`${origin}/api/workspaces`, {
|
|
1386
|
+
headers: {
|
|
1387
|
+
authorization: `Bearer ${accessToken}`,
|
|
1388
|
+
accept: "application/json",
|
|
1389
|
+
"user-agent": USER_AGENT
|
|
1390
|
+
}
|
|
1391
|
+
});
|
|
1392
|
+
if (!res.ok) {
|
|
1393
|
+
await res.body?.cancel();
|
|
1394
|
+
return null;
|
|
1395
|
+
}
|
|
1396
|
+
const body = await res.json();
|
|
1397
|
+
return typeof body?.user?.email === "string" ? body.user.email : null;
|
|
1398
|
+
} catch {
|
|
1399
|
+
return null;
|
|
1400
|
+
}
|
|
1401
|
+
}
|
|
1402
|
+
var login = {
|
|
1403
|
+
path: ["login"],
|
|
1404
|
+
summary: "Sign in with your browser (once per machine)",
|
|
1405
|
+
usage: "[--no-browser]",
|
|
1406
|
+
example: "littlefriend login",
|
|
1407
|
+
options: { "no-browser": { type: "boolean" } },
|
|
1408
|
+
async run(ctx) {
|
|
1409
|
+
const io = ctx.io;
|
|
1410
|
+
const origin = ctx.origin();
|
|
1411
|
+
if (io.env.LITTLEFRIEND_TOKEN?.trim()) {
|
|
1412
|
+
ctx.note("LITTLEFRIEND_TOKEN is set, so other commands keep using it instead of this sign-in.");
|
|
1413
|
+
}
|
|
1414
|
+
const meta = await discover(io, origin);
|
|
1415
|
+
const previous = loadEntry(io, origin);
|
|
1416
|
+
const clientId = previous?.clientId ?? await registerClient(io, meta);
|
|
1417
|
+
const { verifier, challenge } = pkcePair();
|
|
1418
|
+
const state = randomState();
|
|
1419
|
+
const loop = await startLoopback({
|
|
1420
|
+
state,
|
|
1421
|
+
issuer: meta.authorization_response_iss_parameter_supported ? meta.issuer : null,
|
|
1422
|
+
timeoutMs: io.loginTimeoutMs
|
|
1423
|
+
});
|
|
1424
|
+
let code;
|
|
1425
|
+
try {
|
|
1426
|
+
const url = authorizationUrl(meta, { clientId, redirectUri: loop.redirectUri, challenge, state });
|
|
1427
|
+
const minutes = Math.max(1, Math.round(io.loginTimeoutMs / 6e4));
|
|
1428
|
+
ctx.note(
|
|
1429
|
+
lines(
|
|
1430
|
+
flag(ctx, "no-browser") ? "Open this URL in your browser to sign in to Little Friend:" : "Opening your browser to sign in to Little Friend. If it does not open, visit this URL:",
|
|
1431
|
+
"",
|
|
1432
|
+
` ${url}`,
|
|
1433
|
+
"",
|
|
1434
|
+
`Waiting for you to approve (up to ${minutes} minute${minutes === 1 ? "" : "s"})...`
|
|
1435
|
+
)
|
|
1436
|
+
);
|
|
1437
|
+
if (!flag(ctx, "no-browser")) io.openUrl(url);
|
|
1438
|
+
code = await loop.code;
|
|
1439
|
+
} finally {
|
|
1440
|
+
await loop.close();
|
|
1441
|
+
}
|
|
1442
|
+
const tokens = await tokenRequest(io, meta, {
|
|
1443
|
+
grant_type: "authorization_code",
|
|
1444
|
+
code,
|
|
1445
|
+
redirect_uri: loop.redirectUri,
|
|
1446
|
+
client_id: clientId,
|
|
1447
|
+
code_verifier: verifier
|
|
1448
|
+
});
|
|
1449
|
+
if (!scopeIncludes(tokens.scope, MANAGE_SCOPE)) {
|
|
1450
|
+
const token = tokens.refresh_token ?? tokens.access_token;
|
|
1451
|
+
const hint = tokens.refresh_token ? "refresh_token" : "access_token";
|
|
1452
|
+
await revoke(io, meta, token, hint, clientId).catch(() => {
|
|
1453
|
+
});
|
|
1454
|
+
throw new CliError(
|
|
1455
|
+
`Little Friend at ${origin} granted "${tokens.scope}" without ${MANAGE_SCOPE}, so the CLI cannot manage projects with it. Nothing was saved.`,
|
|
1456
|
+
"SCOPE_NOT_GRANTED"
|
|
1457
|
+
);
|
|
1458
|
+
}
|
|
1459
|
+
const email = await lookupEmail(io, origin, tokens.access_token);
|
|
1460
|
+
const expiresAt = new Date(io.now() + tokens.expires_in * 1e3).toISOString();
|
|
1461
|
+
saveEntry(io, origin, {
|
|
1462
|
+
clientId,
|
|
1463
|
+
accessToken: tokens.access_token,
|
|
1464
|
+
refreshToken: tokens.refresh_token ?? null,
|
|
1465
|
+
expiresAt,
|
|
1466
|
+
user: { email },
|
|
1467
|
+
...tokens.scope ? { scope: tokens.scope } : {}
|
|
1468
|
+
});
|
|
1469
|
+
if (previous?.refreshToken && previous.refreshToken !== tokens.refresh_token) {
|
|
1470
|
+
await revoke(io, meta, previous.refreshToken, "refresh_token", previous.clientId).catch(() => {
|
|
1471
|
+
});
|
|
1472
|
+
}
|
|
1473
|
+
return {
|
|
1474
|
+
data: { api: origin, email, scope: tokens.scope ?? null, expiresAt },
|
|
1475
|
+
text: lines(
|
|
1476
|
+
`Signed in to ${origin}${email ? ` as ${email}` : ""}.`,
|
|
1477
|
+
`Saved to ${credentialsPath(io)}. The CLI refreshes the sign-in on its own.`
|
|
1478
|
+
)
|
|
1479
|
+
};
|
|
1480
|
+
}
|
|
1481
|
+
};
|
|
1482
|
+
var logout = {
|
|
1483
|
+
path: ["logout"],
|
|
1484
|
+
summary: "Revoke this machine's sign-in and delete it",
|
|
1485
|
+
usage: "",
|
|
1486
|
+
example: "littlefriend logout",
|
|
1487
|
+
async run(ctx) {
|
|
1488
|
+
const io = ctx.io;
|
|
1489
|
+
const origin = ctx.origin();
|
|
1490
|
+
const entry = loadEntry(io, origin);
|
|
1491
|
+
if (!entry) {
|
|
1492
|
+
return {
|
|
1493
|
+
data: { api: origin, signedOut: false, revoked: false },
|
|
1494
|
+
text: `Not signed in to ${origin} on this machine.`
|
|
1495
|
+
};
|
|
1496
|
+
}
|
|
1497
|
+
let problem = null;
|
|
1498
|
+
try {
|
|
1499
|
+
const meta = await discover(io, origin);
|
|
1500
|
+
if (entry.refreshToken) await revoke(io, meta, entry.refreshToken, "refresh_token", entry.clientId);
|
|
1501
|
+
else await revoke(io, meta, entry.accessToken, "access_token", entry.clientId);
|
|
1502
|
+
} catch (error) {
|
|
1503
|
+
problem = error instanceof Error ? error.message : String(error);
|
|
1504
|
+
}
|
|
1505
|
+
deleteEntry(io, origin);
|
|
1506
|
+
if (problem) {
|
|
1507
|
+
throw new CliError(
|
|
1508
|
+
`Signed out on this machine, but Little Friend could not revoke the sign-in: ${problem} Revoke it in the dashboard under Account, Connected apps.`,
|
|
1509
|
+
"REVOKE_FAILED"
|
|
1510
|
+
);
|
|
1511
|
+
}
|
|
1512
|
+
return {
|
|
1513
|
+
data: { api: origin, signedOut: true, revoked: true },
|
|
1514
|
+
text: `Signed out of ${origin}. The sign-in was revoked and deleted from this machine.`
|
|
1515
|
+
};
|
|
1516
|
+
}
|
|
1517
|
+
};
|
|
1518
|
+
var whoami = {
|
|
1519
|
+
path: ["whoami"],
|
|
1520
|
+
summary: "Show who the CLI is signed in as, and your workspaces",
|
|
1521
|
+
usage: "",
|
|
1522
|
+
example: "littlefriend whoami",
|
|
1523
|
+
async run(ctx) {
|
|
1524
|
+
const origin = ctx.origin();
|
|
1525
|
+
const session = ctx.api().session;
|
|
1526
|
+
const { user, workspaces } = await accountAndWorkspaces(ctx);
|
|
1527
|
+
const stored = session.stored;
|
|
1528
|
+
const email = user?.email ?? stored?.user.email ?? null;
|
|
1529
|
+
return {
|
|
1530
|
+
data: {
|
|
1531
|
+
api: origin,
|
|
1532
|
+
email,
|
|
1533
|
+
source: session.source,
|
|
1534
|
+
expiresAt: session.source === "stored" ? stored?.expiresAt ?? null : null,
|
|
1535
|
+
workspaces
|
|
1536
|
+
},
|
|
1537
|
+
text: lines(
|
|
1538
|
+
fields([
|
|
1539
|
+
["API", origin],
|
|
1540
|
+
...email ? [["Signed in as", email]] : [],
|
|
1541
|
+
[
|
|
1542
|
+
"Credentials",
|
|
1543
|
+
session.source === "env" ? "LITTLEFRIEND_TOKEN" : `${credentialsPath(ctx.io)} (refreshes on its own)`
|
|
1544
|
+
],
|
|
1545
|
+
...session.source === "stored" ? [["Access token expires", when(stored?.expiresAt)]] : []
|
|
1546
|
+
]),
|
|
1547
|
+
"",
|
|
1548
|
+
workspaces.length > 0 ? table(
|
|
1549
|
+
["WORKSPACE", "SLUG", "NAME", "ROLE"],
|
|
1550
|
+
workspaces.map((w) => [w.id, w.slug, w.name, w.role])
|
|
1551
|
+
) : "No workspaces yet."
|
|
1552
|
+
)
|
|
1553
|
+
};
|
|
1554
|
+
}
|
|
1555
|
+
};
|
|
1556
|
+
var authCommands = [login, logout, whoami];
|
|
1557
|
+
|
|
1558
|
+
// ../classify/dist/ua.js
|
|
1559
|
+
var MAX_UA_LENGTH = 512;
|
|
1560
|
+
var BROWSERS = [
|
|
1561
|
+
[/\bFBA[NV]\//, "Facebook App"],
|
|
1562
|
+
[/\bInstagram\b/, "Instagram App"],
|
|
1563
|
+
[/\bEdg(?:e|A|iOS)?\/\d/, "Edge"],
|
|
1564
|
+
[/\bOPR\/|\bOPT\/|\bOPiOS\/|\bOpera\b/, "Opera"],
|
|
1565
|
+
[/\bSamsungBrowser\//, "Samsung Internet"],
|
|
1566
|
+
[/\bYaBrowser\//, "Yandex Browser"],
|
|
1567
|
+
[/\bUCBrowser\/|\bUCWEB/, "UC Browser"],
|
|
1568
|
+
[/\bVivaldi\//, "Vivaldi"],
|
|
1569
|
+
[/\bDuckDuckGo\/\d/, "DuckDuckGo"],
|
|
1570
|
+
[/\bElectron\//, "Electron"],
|
|
1571
|
+
[/\bMSIE \d|\bTrident\/\d/, "Internet Explorer"],
|
|
1572
|
+
[/\bFirefox\/\d|\bFxiOS\/\d/, "Firefox"],
|
|
1573
|
+
[/\bCriOS\/\d|\bChrome\/\d/, "Chrome"],
|
|
1574
|
+
[/\bChromium\/\d/, "Chromium"],
|
|
1575
|
+
[/\bVersion\/[\d.]+.*\bSafari\/\d/, "Safari"]
|
|
1576
|
+
];
|
|
1577
|
+
var ANDROID_WEBVIEW = /;\s?wv\)/;
|
|
1578
|
+
var IOS_WEBKIT = /\((?:iPhone|iPad|iPod)\b.*\bAppleWebKit\/[\d.]+.*\bMobile\/\w+/;
|
|
1579
|
+
var MAC_WEBKIT = /^Mozilla\/5\.0 \(Macintosh; Intel Mac OS X [\d_]+\) AppleWebKit\/[\d.]+ \(KHTML, like Gecko\)$/;
|
|
1580
|
+
function detectBrowser(ua) {
|
|
1581
|
+
for (const [re, name] of BROWSERS) {
|
|
1582
|
+
if (re.test(ua)) {
|
|
1583
|
+
if (name === "Chrome" && ANDROID_WEBVIEW.test(ua))
|
|
1584
|
+
return "Android WebView";
|
|
1585
|
+
return name;
|
|
1586
|
+
}
|
|
1587
|
+
}
|
|
1588
|
+
if (IOS_WEBKIT.test(ua))
|
|
1589
|
+
return "iOS WebView";
|
|
1590
|
+
if (MAC_WEBKIT.test(ua))
|
|
1591
|
+
return "macOS WebView";
|
|
1592
|
+
return null;
|
|
1593
|
+
}
|
|
1594
|
+
function isBrowserLike(input) {
|
|
1595
|
+
if (typeof input !== "string")
|
|
1596
|
+
return false;
|
|
1597
|
+
const ua = input.slice(0, MAX_UA_LENGTH);
|
|
1598
|
+
if (!/^Mozilla\/5\.0 \(/.test(ua) && !/^Opera\//.test(ua))
|
|
1599
|
+
return false;
|
|
1600
|
+
if (!/AppleWebKit\/|Gecko\/|Trident\/|Presto\//.test(ua))
|
|
1601
|
+
return false;
|
|
1602
|
+
return detectBrowser(ua) !== null;
|
|
1603
|
+
}
|
|
1604
|
+
|
|
1605
|
+
// ../classify/dist/rules.js
|
|
1606
|
+
var GOOGLE_COMMON_DOCS = "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers";
|
|
1607
|
+
var GOOGLE_SPECIAL_DOCS = "https://developers.google.com/crawling/docs/crawlers-fetchers/google-special-case-crawlers";
|
|
1608
|
+
var GOOGLE_USER_DOCS = "https://developers.google.com/crawling/docs/crawlers-fetchers/google-user-triggered-fetchers";
|
|
1609
|
+
var GOOGLE_COMMON = { reverseDnsSuffixes: ["googlebot.com"], ipRangeSources: ["google-common"] };
|
|
1610
|
+
var GOOGLE_SPECIAL = { reverseDnsSuffixes: ["google.com"], ipRangeSources: ["google-special"] };
|
|
1611
|
+
var GOOGLE_USER = {
|
|
1612
|
+
reverseDnsSuffixes: ["gae.googleusercontent.com", "google.com"],
|
|
1613
|
+
ipRangeSources: ["google-user-fetchers", "google-user-fetchers-google"]
|
|
1614
|
+
};
|
|
1615
|
+
var BING = { reverseDnsSuffixes: ["search.msn.com"] };
|
|
1616
|
+
var YANDEX = { reverseDnsSuffixes: ["yandex.ru", "yandex.net", "yandex.com"] };
|
|
1617
|
+
var ANTHROPIC = { ipRangeSources: ["anthropic"] };
|
|
1618
|
+
var ANTHROPIC_DOCS = "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler";
|
|
1619
|
+
var OPENAI_DOCS = "https://developers.openai.com/api/docs/bots";
|
|
1620
|
+
var META_DOCS = "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/";
|
|
1621
|
+
var AMAZON_DOCS = "https://developer.amazon.com/amazonbot";
|
|
1622
|
+
var MISTRAL_DOCS = "https://docs.mistral.ai/robots";
|
|
1623
|
+
var PERPLEXITY_DOCS = "https://docs.perplexity.ai/guides/bots";
|
|
1624
|
+
var BOT_RULES = [
|
|
1625
|
+
// Link previews that impersonate other previewers go first.
|
|
1332
1626
|
{
|
|
1333
|
-
id: "
|
|
1334
|
-
pattern:
|
|
1627
|
+
id: "imessage-preview",
|
|
1628
|
+
pattern: /facebookexternalhit\/1\.1 Facebot Twitterbot\/1\.0/,
|
|
1335
1629
|
class: "preview",
|
|
1336
1630
|
purpose: "preview",
|
|
1337
|
-
operator: "Microsoft",
|
|
1338
|
-
product: "MicrosoftPreview",
|
|
1339
|
-
verify: BING
|
|
1340
|
-
},
|
|
1341
|
-
{
|
|
1342
|
-
id: "adidxbot",
|
|
1343
|
-
pattern: /\badidxbot\b/i,
|
|
1344
|
-
class: "search_crawler",
|
|
1345
|
-
purpose: "ads",
|
|
1346
|
-
operator: "Microsoft",
|
|
1347
|
-
product: "AdIdxBot",
|
|
1348
|
-
verify: BING
|
|
1349
|
-
},
|
|
1350
|
-
// Bing's index also feeds Copilot answers and Microsoft's model training, unless a page
|
|
1351
|
-
// opts out with the nocache or noarchive robots tags.
|
|
1352
|
-
{
|
|
1353
|
-
id: "bingbot",
|
|
1354
|
-
pattern: /\bbingbot\b/i,
|
|
1355
|
-
class: "search_crawler",
|
|
1356
|
-
purpose: "search",
|
|
1357
|
-
alsoFor: ["ai_search", "ai_training"],
|
|
1358
|
-
operator: "Microsoft",
|
|
1359
|
-
product: "Bingbot",
|
|
1360
|
-
verify: { reverseDnsSuffixes: ["search.msn.com"], ipRangeSources: ["bing"] },
|
|
1361
|
-
docs: "https://blogs.bing.com/webmaster/August-2012/How-to-Verify-that-Bingbot-is-Bingbot/"
|
|
1362
|
-
},
|
|
1363
|
-
{
|
|
1364
|
-
id: "msnbot",
|
|
1365
|
-
pattern: /\bmsnbot\b/i,
|
|
1366
|
-
class: "search_crawler",
|
|
1367
|
-
purpose: "search",
|
|
1368
|
-
operator: "Microsoft",
|
|
1369
|
-
product: "msnbot",
|
|
1370
|
-
verify: BING
|
|
1371
|
-
},
|
|
1372
|
-
// Apple. Applebot-Extended is a robots.txt token only and never appears in a user agent.
|
|
1373
|
-
// Applebot powers Spotlight, Siri and Safari search; Apple documents that what it crawls may
|
|
1374
|
-
// also train Apple's foundation models and give its AI answers context.
|
|
1375
|
-
{
|
|
1376
|
-
id: "applebot",
|
|
1377
|
-
pattern: /\bApplebot\b/,
|
|
1378
|
-
class: "search_crawler",
|
|
1379
|
-
purpose: "search",
|
|
1380
|
-
alsoFor: ["ai_search", "ai_training"],
|
|
1381
1631
|
operator: "Apple",
|
|
1382
|
-
product: "
|
|
1383
|
-
verify: { reverseDnsSuffixes: ["applebot.apple.com"], ipRangeSources: ["apple"] },
|
|
1384
|
-
docs: "https://support.apple.com/en-us/119829"
|
|
1632
|
+
product: "iMessage link preview"
|
|
1385
1633
|
},
|
|
1386
|
-
//
|
|
1387
|
-
// used for training, so it is an assistant fetching for a user, not an indexer.
|
|
1634
|
+
// Google: user-triggered agents and fetchers.
|
|
1388
1635
|
{
|
|
1389
|
-
id: "
|
|
1390
|
-
pattern: /\
|
|
1636
|
+
id: "google-agent",
|
|
1637
|
+
pattern: /\bGoogle-Agent\b/,
|
|
1391
1638
|
class: "ai_agent",
|
|
1392
|
-
purpose: "
|
|
1393
|
-
operator: "
|
|
1394
|
-
product: "
|
|
1395
|
-
verify: { ipRangeSources: ["
|
|
1396
|
-
docs:
|
|
1397
|
-
|
|
1398
|
-
{
|
|
1399
|
-
id: "duckduckbot",
|
|
1400
|
-
pattern: /\bDuckDuckBot\b/,
|
|
1401
|
-
class: "search_crawler",
|
|
1402
|
-
purpose: "search",
|
|
1403
|
-
operator: "DuckDuckGo",
|
|
1404
|
-
product: "DuckDuckBot",
|
|
1405
|
-
verify: { ipRangeSources: ["duckduckbot"] },
|
|
1406
|
-
docs: "https://duckduckgo.com/duckduckgo-help-pages/results/duckduckbot"
|
|
1407
|
-
},
|
|
1408
|
-
// Yandex robots share one verification domain set.
|
|
1409
|
-
{
|
|
1410
|
-
id: "yandexbot",
|
|
1411
|
-
pattern: /\bYandexBot\/\d/,
|
|
1412
|
-
class: "search_crawler",
|
|
1413
|
-
purpose: "search",
|
|
1414
|
-
operator: "Yandex",
|
|
1415
|
-
product: "YandexBot",
|
|
1416
|
-
verify: YANDEX,
|
|
1417
|
-
docs: "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots"
|
|
1418
|
-
},
|
|
1419
|
-
{
|
|
1420
|
-
id: "yandex-robot",
|
|
1421
|
-
pattern: /\bYandex[A-Za-z]*\/\d/,
|
|
1422
|
-
class: "search_crawler",
|
|
1423
|
-
purpose: "search",
|
|
1424
|
-
operator: "Yandex",
|
|
1425
|
-
product: "Yandex robot",
|
|
1426
|
-
verify: YANDEX,
|
|
1427
|
-
docs: "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots"
|
|
1428
|
-
},
|
|
1429
|
-
// Other search engines. Baidu's own verification page could not be retrieved, so it stays self-declared.
|
|
1430
|
-
{
|
|
1431
|
-
id: "baiduspider",
|
|
1432
|
-
pattern: /\bBaiduspider\b/i,
|
|
1433
|
-
class: "search_crawler",
|
|
1434
|
-
purpose: "search",
|
|
1435
|
-
operator: "Baidu",
|
|
1436
|
-
product: "Baiduspider"
|
|
1437
|
-
},
|
|
1438
|
-
{
|
|
1439
|
-
id: "seznambot",
|
|
1440
|
-
pattern: /\bSeznamBot\b/,
|
|
1441
|
-
class: "search_crawler",
|
|
1442
|
-
purpose: "search",
|
|
1443
|
-
operator: "Seznam",
|
|
1444
|
-
product: "SeznamBot"
|
|
1639
|
+
purpose: "ai_agent",
|
|
1640
|
+
operator: "Google",
|
|
1641
|
+
product: "Google-Agent",
|
|
1642
|
+
verify: { ipRangeSources: ["google-user-agents"] },
|
|
1643
|
+
docs: GOOGLE_USER_DOCS,
|
|
1644
|
+
userInitiated: true
|
|
1445
1645
|
},
|
|
1446
1646
|
{
|
|
1447
|
-
id: "
|
|
1448
|
-
pattern: /\
|
|
1449
|
-
class: "
|
|
1450
|
-
purpose: "
|
|
1451
|
-
operator: "
|
|
1452
|
-
product: "
|
|
1647
|
+
id: "google-gemininotebook",
|
|
1648
|
+
pattern: /\bGoogle-(?:GeminiNotebook|NotebookLM)\b/,
|
|
1649
|
+
class: "ai_agent",
|
|
1650
|
+
purpose: "ai_assistant",
|
|
1651
|
+
operator: "Google",
|
|
1652
|
+
product: "Google-GeminiNotebook",
|
|
1653
|
+
verify: GOOGLE_USER,
|
|
1654
|
+
docs: GOOGLE_USER_DOCS,
|
|
1655
|
+
userInitiated: true
|
|
1453
1656
|
},
|
|
1454
1657
|
{
|
|
1455
|
-
id: "
|
|
1456
|
-
pattern: /\
|
|
1457
|
-
class: "
|
|
1458
|
-
purpose: "
|
|
1459
|
-
operator: "
|
|
1460
|
-
product: "
|
|
1658
|
+
id: "google-read-aloud",
|
|
1659
|
+
pattern: /\bGoogle-Read-Aloud\b|\bgoogle-speakr\b/,
|
|
1660
|
+
class: "preview",
|
|
1661
|
+
purpose: "preview",
|
|
1662
|
+
operator: "Google",
|
|
1663
|
+
product: "Google-Read-Aloud",
|
|
1664
|
+
verify: GOOGLE_USER,
|
|
1665
|
+
docs: GOOGLE_USER_DOCS,
|
|
1666
|
+
userInitiated: true
|
|
1461
1667
|
},
|
|
1462
1668
|
{
|
|
1463
|
-
id: "
|
|
1464
|
-
pattern: /\
|
|
1465
|
-
class: "
|
|
1466
|
-
purpose: "
|
|
1467
|
-
operator: "
|
|
1468
|
-
product: "
|
|
1669
|
+
id: "google-site-verification",
|
|
1670
|
+
pattern: /\bGoogle-Site-Verification\b/,
|
|
1671
|
+
class: "monitor",
|
|
1672
|
+
purpose: "monitor",
|
|
1673
|
+
operator: "Google",
|
|
1674
|
+
product: "Google Site Verifier",
|
|
1675
|
+
verify: GOOGLE_USER,
|
|
1676
|
+
docs: GOOGLE_USER_DOCS,
|
|
1677
|
+
userInitiated: true
|
|
1469
1678
|
},
|
|
1470
1679
|
{
|
|
1471
|
-
id: "
|
|
1472
|
-
pattern: /\
|
|
1680
|
+
id: "google-feedfetcher",
|
|
1681
|
+
pattern: /\bFeedFetcher-Google\b/,
|
|
1473
1682
|
class: "search_crawler",
|
|
1474
1683
|
purpose: "search",
|
|
1475
|
-
operator: "
|
|
1476
|
-
product: "
|
|
1684
|
+
operator: "Google",
|
|
1685
|
+
product: "FeedFetcher-Google",
|
|
1686
|
+
verify: GOOGLE_USER,
|
|
1687
|
+
docs: GOOGLE_USER_DOCS
|
|
1477
1688
|
},
|
|
1478
1689
|
{
|
|
1479
|
-
id: "
|
|
1480
|
-
pattern: /\
|
|
1690
|
+
id: "google-producer",
|
|
1691
|
+
pattern: /\bGoogleProducer\b/,
|
|
1481
1692
|
class: "search_crawler",
|
|
1482
1693
|
purpose: "search",
|
|
1483
|
-
operator: "
|
|
1484
|
-
product: "
|
|
1485
|
-
|
|
1486
|
-
|
|
1487
|
-
{
|
|
1488
|
-
id: "chatgpt-user",
|
|
1489
|
-
pattern: /\bChatGPT-User\b/,
|
|
1490
|
-
class: "ai_agent",
|
|
1491
|
-
purpose: "ai_assistant",
|
|
1492
|
-
operator: "OpenAI",
|
|
1493
|
-
product: "ChatGPT-User",
|
|
1494
|
-
verify: { ipRangeSources: ["openai-chatgpt-user"] },
|
|
1495
|
-
docs: OPENAI_DOCS,
|
|
1496
|
-
userInitiated: true
|
|
1497
|
-
},
|
|
1498
|
-
{
|
|
1499
|
-
id: "oai-searchbot",
|
|
1500
|
-
pattern: /\bOAI-SearchBot\b/,
|
|
1501
|
-
class: "ai_crawler",
|
|
1502
|
-
purpose: "ai_search",
|
|
1503
|
-
operator: "OpenAI",
|
|
1504
|
-
product: "OAI-SearchBot",
|
|
1505
|
-
verify: { ipRangeSources: ["openai-searchbot"] },
|
|
1506
|
-
docs: OPENAI_DOCS
|
|
1694
|
+
operator: "Google",
|
|
1695
|
+
product: "GoogleProducer",
|
|
1696
|
+
verify: GOOGLE_USER,
|
|
1697
|
+
docs: GOOGLE_USER_DOCS
|
|
1507
1698
|
},
|
|
1508
1699
|
{
|
|
1509
|
-
id: "
|
|
1510
|
-
pattern: /\
|
|
1511
|
-
class: "
|
|
1512
|
-
purpose: "
|
|
1513
|
-
operator: "
|
|
1514
|
-
product: "
|
|
1515
|
-
verify:
|
|
1516
|
-
docs:
|
|
1700
|
+
id: "google-messages",
|
|
1701
|
+
pattern: /\bGoogleMessages\b/,
|
|
1702
|
+
class: "preview",
|
|
1703
|
+
purpose: "preview",
|
|
1704
|
+
operator: "Google",
|
|
1705
|
+
product: "Google Messages",
|
|
1706
|
+
verify: GOOGLE_USER,
|
|
1707
|
+
docs: GOOGLE_USER_DOCS
|
|
1517
1708
|
},
|
|
1518
1709
|
{
|
|
1519
|
-
id: "
|
|
1520
|
-
pattern: /\
|
|
1521
|
-
class: "
|
|
1522
|
-
purpose: "
|
|
1523
|
-
operator: "
|
|
1524
|
-
product: "
|
|
1525
|
-
verify:
|
|
1526
|
-
docs:
|
|
1710
|
+
id: "google-cws",
|
|
1711
|
+
pattern: /\bGoogle-CWS\b/,
|
|
1712
|
+
class: "automation",
|
|
1713
|
+
purpose: "monitor",
|
|
1714
|
+
operator: "Google",
|
|
1715
|
+
product: "Google-CWS",
|
|
1716
|
+
verify: GOOGLE_USER,
|
|
1717
|
+
docs: GOOGLE_USER_DOCS
|
|
1527
1718
|
},
|
|
1528
|
-
//
|
|
1719
|
+
// Fetches the URLs a user adds to a Pinpoint collection, like GeminiNotebook does for its sources.
|
|
1529
1720
|
{
|
|
1530
|
-
id: "
|
|
1531
|
-
pattern: /\
|
|
1721
|
+
id: "google-pinpoint",
|
|
1722
|
+
pattern: /\bGoogle-Pinpoint\b/,
|
|
1532
1723
|
class: "ai_agent",
|
|
1533
1724
|
purpose: "ai_assistant",
|
|
1534
|
-
operator: "
|
|
1535
|
-
product: "
|
|
1536
|
-
verify:
|
|
1537
|
-
docs:
|
|
1725
|
+
operator: "Google",
|
|
1726
|
+
product: "Google-Pinpoint",
|
|
1727
|
+
verify: GOOGLE_USER,
|
|
1728
|
+
docs: GOOGLE_USER_DOCS,
|
|
1538
1729
|
userInitiated: true
|
|
1539
1730
|
},
|
|
1731
|
+
// Google: special-case crawlers.
|
|
1540
1732
|
{
|
|
1541
|
-
id: "
|
|
1542
|
-
pattern: /\
|
|
1543
|
-
class: "
|
|
1544
|
-
purpose: "
|
|
1545
|
-
operator: "
|
|
1546
|
-
product: "
|
|
1547
|
-
verify:
|
|
1548
|
-
docs:
|
|
1549
|
-
},
|
|
1550
|
-
{
|
|
1551
|
-
id: "claudebot",
|
|
1552
|
-
pattern: /\bClaudeBot\b/,
|
|
1553
|
-
class: "ai_crawler",
|
|
1554
|
-
purpose: "ai_training",
|
|
1555
|
-
operator: "Anthropic",
|
|
1556
|
-
product: "ClaudeBot",
|
|
1557
|
-
verify: ANTHROPIC,
|
|
1558
|
-
docs: ANTHROPIC_DOCS
|
|
1733
|
+
id: "adsbot-google-mobile",
|
|
1734
|
+
pattern: /\bAdsBot-Google-Mobile\b/,
|
|
1735
|
+
class: "search_crawler",
|
|
1736
|
+
purpose: "ads",
|
|
1737
|
+
operator: "Google",
|
|
1738
|
+
product: "AdsBot-Google-Mobile",
|
|
1739
|
+
verify: GOOGLE_SPECIAL,
|
|
1740
|
+
docs: GOOGLE_SPECIAL_DOCS
|
|
1559
1741
|
},
|
|
1560
1742
|
{
|
|
1561
|
-
id: "
|
|
1562
|
-
pattern: /\
|
|
1563
|
-
class: "
|
|
1564
|
-
purpose: "
|
|
1565
|
-
operator: "
|
|
1566
|
-
product: "
|
|
1567
|
-
verify:
|
|
1568
|
-
docs:
|
|
1743
|
+
id: "adsbot-google",
|
|
1744
|
+
pattern: /\bAdsBot-Google\b/,
|
|
1745
|
+
class: "search_crawler",
|
|
1746
|
+
purpose: "ads",
|
|
1747
|
+
operator: "Google",
|
|
1748
|
+
product: "AdsBot-Google",
|
|
1749
|
+
verify: GOOGLE_SPECIAL,
|
|
1750
|
+
docs: GOOGLE_SPECIAL_DOCS
|
|
1569
1751
|
},
|
|
1570
1752
|
{
|
|
1571
|
-
id: "
|
|
1572
|
-
pattern: /\
|
|
1573
|
-
class: "
|
|
1574
|
-
purpose: "
|
|
1575
|
-
operator: "
|
|
1576
|
-
product: "
|
|
1577
|
-
verify:
|
|
1578
|
-
docs:
|
|
1753
|
+
id: "mediapartners-google",
|
|
1754
|
+
pattern: /\bMediapartners-Google\b/,
|
|
1755
|
+
class: "search_crawler",
|
|
1756
|
+
purpose: "ads",
|
|
1757
|
+
operator: "Google",
|
|
1758
|
+
product: "Mediapartners-Google",
|
|
1759
|
+
verify: GOOGLE_SPECIAL,
|
|
1760
|
+
docs: GOOGLE_SPECIAL_DOCS
|
|
1579
1761
|
},
|
|
1580
|
-
// Perplexity.
|
|
1581
1762
|
{
|
|
1582
|
-
id: "
|
|
1583
|
-
pattern: /\
|
|
1584
|
-
class: "
|
|
1585
|
-
purpose: "
|
|
1586
|
-
operator: "
|
|
1587
|
-
product: "
|
|
1588
|
-
verify:
|
|
1589
|
-
docs:
|
|
1590
|
-
userInitiated: true
|
|
1763
|
+
id: "apis-google",
|
|
1764
|
+
pattern: /\bAPIs-Google\b/,
|
|
1765
|
+
class: "automation",
|
|
1766
|
+
purpose: "monitor",
|
|
1767
|
+
operator: "Google",
|
|
1768
|
+
product: "APIs-Google",
|
|
1769
|
+
verify: GOOGLE_SPECIAL,
|
|
1770
|
+
docs: GOOGLE_SPECIAL_DOCS
|
|
1591
1771
|
},
|
|
1592
1772
|
{
|
|
1593
|
-
id: "
|
|
1594
|
-
pattern: /\
|
|
1595
|
-
class: "
|
|
1596
|
-
purpose: "
|
|
1597
|
-
operator: "
|
|
1598
|
-
product: "
|
|
1599
|
-
verify:
|
|
1600
|
-
docs:
|
|
1773
|
+
id: "google-safety",
|
|
1774
|
+
pattern: /\bGoogle-Safety\b/,
|
|
1775
|
+
class: "monitor",
|
|
1776
|
+
purpose: "monitor",
|
|
1777
|
+
operator: "Google",
|
|
1778
|
+
product: "Google-Safety",
|
|
1779
|
+
verify: GOOGLE_SPECIAL,
|
|
1780
|
+
docs: GOOGLE_SPECIAL_DOCS
|
|
1601
1781
|
},
|
|
1602
|
-
//
|
|
1782
|
+
// Google: common crawlers.
|
|
1603
1783
|
{
|
|
1604
|
-
id: "
|
|
1605
|
-
pattern: /\
|
|
1606
|
-
class: "
|
|
1607
|
-
purpose: "
|
|
1608
|
-
operator: "
|
|
1609
|
-
product: "
|
|
1610
|
-
verify:
|
|
1611
|
-
docs:
|
|
1612
|
-
userInitiated: true
|
|
1784
|
+
id: "google-inspectiontool",
|
|
1785
|
+
pattern: /\bGoogle-InspectionTool\b/,
|
|
1786
|
+
class: "search_crawler",
|
|
1787
|
+
purpose: "search",
|
|
1788
|
+
operator: "Google",
|
|
1789
|
+
product: "Google-InspectionTool",
|
|
1790
|
+
verify: GOOGLE_COMMON,
|
|
1791
|
+
docs: GOOGLE_COMMON_DOCS
|
|
1613
1792
|
},
|
|
1614
1793
|
{
|
|
1615
|
-
id: "
|
|
1616
|
-
pattern: /\
|
|
1794
|
+
id: "google-cloudvertexbot",
|
|
1795
|
+
pattern: /\bGoogle-CloudVertexBot\b/,
|
|
1617
1796
|
class: "ai_crawler",
|
|
1618
1797
|
purpose: "ai_search",
|
|
1619
|
-
operator: "
|
|
1620
|
-
product: "
|
|
1621
|
-
verify:
|
|
1622
|
-
docs:
|
|
1798
|
+
operator: "Google",
|
|
1799
|
+
product: "Google-CloudVertexBot",
|
|
1800
|
+
verify: GOOGLE_COMMON,
|
|
1801
|
+
docs: GOOGLE_COMMON_DOCS
|
|
1623
1802
|
},
|
|
1624
1803
|
{
|
|
1625
|
-
id: "
|
|
1626
|
-
pattern: /\
|
|
1627
|
-
class: "
|
|
1628
|
-
purpose: "
|
|
1629
|
-
operator: "
|
|
1630
|
-
product: "
|
|
1631
|
-
|
|
1804
|
+
id: "storebot-google",
|
|
1805
|
+
pattern: /\bStorebot-Google\b/,
|
|
1806
|
+
class: "search_crawler",
|
|
1807
|
+
purpose: "ads",
|
|
1808
|
+
operator: "Google",
|
|
1809
|
+
product: "Storebot-Google",
|
|
1810
|
+
verify: GOOGLE_COMMON,
|
|
1811
|
+
docs: GOOGLE_COMMON_DOCS
|
|
1632
1812
|
},
|
|
1633
|
-
// Amazon.
|
|
1634
1813
|
{
|
|
1635
|
-
id: "
|
|
1636
|
-
pattern: /\
|
|
1637
|
-
class: "
|
|
1638
|
-
purpose: "
|
|
1639
|
-
operator: "
|
|
1640
|
-
product: "
|
|
1641
|
-
verify:
|
|
1642
|
-
docs:
|
|
1643
|
-
userInitiated: true
|
|
1814
|
+
id: "googleother",
|
|
1815
|
+
pattern: /\bGoogleOther(?:-Image|-Video)?\b/,
|
|
1816
|
+
class: "search_crawler",
|
|
1817
|
+
purpose: "unknown",
|
|
1818
|
+
operator: "Google",
|
|
1819
|
+
product: "GoogleOther",
|
|
1820
|
+
verify: GOOGLE_COMMON,
|
|
1821
|
+
docs: GOOGLE_COMMON_DOCS
|
|
1644
1822
|
},
|
|
1645
|
-
// Amzn-SearchBot feeds Alexa and Amazon's AI search, not model training.
|
|
1646
1823
|
{
|
|
1647
|
-
id: "
|
|
1648
|
-
pattern: /\
|
|
1649
|
-
class: "
|
|
1650
|
-
purpose: "
|
|
1651
|
-
operator: "
|
|
1652
|
-
product: "
|
|
1653
|
-
verify:
|
|
1654
|
-
docs:
|
|
1824
|
+
id: "googlebot-image",
|
|
1825
|
+
pattern: /\bGooglebot-Image\b/,
|
|
1826
|
+
class: "search_crawler",
|
|
1827
|
+
purpose: "search",
|
|
1828
|
+
operator: "Google",
|
|
1829
|
+
product: "Googlebot-Image",
|
|
1830
|
+
verify: GOOGLE_COMMON,
|
|
1831
|
+
docs: GOOGLE_COMMON_DOCS
|
|
1655
1832
|
},
|
|
1656
1833
|
{
|
|
1657
|
-
id: "
|
|
1658
|
-
pattern: /\
|
|
1659
|
-
class: "
|
|
1660
|
-
purpose: "
|
|
1661
|
-
operator: "
|
|
1662
|
-
product: "
|
|
1663
|
-
verify:
|
|
1664
|
-
docs:
|
|
1834
|
+
id: "googlebot-video",
|
|
1835
|
+
pattern: /\bGooglebot-Video\b/,
|
|
1836
|
+
class: "search_crawler",
|
|
1837
|
+
purpose: "search",
|
|
1838
|
+
operator: "Google",
|
|
1839
|
+
product: "Googlebot-Video",
|
|
1840
|
+
verify: GOOGLE_COMMON,
|
|
1841
|
+
docs: GOOGLE_COMMON_DOCS
|
|
1665
1842
|
},
|
|
1666
|
-
// Common Crawl publishes ranges and forward-confirmed reverse DNS. Its open archive is the
|
|
1667
|
-
// most common source of AI training data; there is no separate archive purpose.
|
|
1668
1843
|
{
|
|
1669
|
-
id: "
|
|
1670
|
-
pattern: /\
|
|
1671
|
-
class: "
|
|
1672
|
-
purpose: "
|
|
1673
|
-
operator: "
|
|
1674
|
-
product: "
|
|
1675
|
-
verify:
|
|
1676
|
-
docs:
|
|
1844
|
+
id: "googlebot-news",
|
|
1845
|
+
pattern: /\bGooglebot-News\b/,
|
|
1846
|
+
class: "search_crawler",
|
|
1847
|
+
purpose: "search",
|
|
1848
|
+
operator: "Google",
|
|
1849
|
+
product: "Googlebot-News",
|
|
1850
|
+
verify: GOOGLE_COMMON,
|
|
1851
|
+
docs: GOOGLE_COMMON_DOCS
|
|
1677
1852
|
},
|
|
1678
|
-
//
|
|
1853
|
+
// Google-Extended is a robots.txt token only: what Googlebot crawls may also train Gemini
|
|
1854
|
+
// models and ground Gemini answers unless a site disallows it.
|
|
1679
1855
|
{
|
|
1680
|
-
id: "
|
|
1681
|
-
pattern: /\
|
|
1682
|
-
class: "
|
|
1683
|
-
purpose: "
|
|
1684
|
-
|
|
1685
|
-
|
|
1686
|
-
|
|
1687
|
-
|
|
1856
|
+
id: "googlebot",
|
|
1857
|
+
pattern: /\bGooglebot\b/i,
|
|
1858
|
+
class: "search_crawler",
|
|
1859
|
+
purpose: "search",
|
|
1860
|
+
alsoFor: ["ai_search", "ai_training"],
|
|
1861
|
+
operator: "Google",
|
|
1862
|
+
product: "Googlebot",
|
|
1863
|
+
verify: GOOGLE_COMMON,
|
|
1864
|
+
docs: GOOGLE_COMMON_DOCS
|
|
1688
1865
|
},
|
|
1689
1866
|
{
|
|
1690
|
-
id: "
|
|
1691
|
-
pattern: /\
|
|
1692
|
-
class: "
|
|
1693
|
-
purpose: "
|
|
1694
|
-
operator: "
|
|
1695
|
-
product: "
|
|
1696
|
-
docs: META_DOCS
|
|
1867
|
+
id: "google-uptime-checks",
|
|
1868
|
+
pattern: /\bGoogleStackdriverMonitoring-UptimeChecks\b/,
|
|
1869
|
+
class: "monitor",
|
|
1870
|
+
purpose: "monitor",
|
|
1871
|
+
operator: "Google",
|
|
1872
|
+
product: "Cloud Monitoring uptime check"
|
|
1697
1873
|
},
|
|
1698
|
-
//
|
|
1874
|
+
// Microsoft Bing: every Bing crawler reverse-resolves under search.msn.com.
|
|
1699
1875
|
{
|
|
1700
|
-
id: "
|
|
1701
|
-
pattern: /\
|
|
1702
|
-
class: "
|
|
1703
|
-
purpose: "
|
|
1704
|
-
operator: "
|
|
1705
|
-
product: "
|
|
1706
|
-
|
|
1876
|
+
id: "bingpreview",
|
|
1877
|
+
pattern: /\bBingPreview\b/,
|
|
1878
|
+
class: "preview",
|
|
1879
|
+
purpose: "preview",
|
|
1880
|
+
operator: "Microsoft",
|
|
1881
|
+
product: "BingPreview",
|
|
1882
|
+
verify: BING,
|
|
1883
|
+
docs: "https://blogs.bing.com/webmaster/August-2012/How-to-Verify-that-Bingbot-is-Bingbot/"
|
|
1707
1884
|
},
|
|
1708
1885
|
{
|
|
1709
|
-
id: "
|
|
1710
|
-
pattern: /\
|
|
1886
|
+
id: "microsoftpreview",
|
|
1887
|
+
pattern: /\bMicrosoftPreview\b/,
|
|
1888
|
+
class: "preview",
|
|
1889
|
+
purpose: "preview",
|
|
1890
|
+
operator: "Microsoft",
|
|
1891
|
+
product: "MicrosoftPreview",
|
|
1892
|
+
verify: BING
|
|
1893
|
+
},
|
|
1894
|
+
{
|
|
1895
|
+
id: "adidxbot",
|
|
1896
|
+
pattern: /\badidxbot\b/i,
|
|
1711
1897
|
class: "search_crawler",
|
|
1712
1898
|
purpose: "ads",
|
|
1713
|
-
operator: "
|
|
1714
|
-
product: "
|
|
1715
|
-
|
|
1899
|
+
operator: "Microsoft",
|
|
1900
|
+
product: "AdIdxBot",
|
|
1901
|
+
verify: BING
|
|
1716
1902
|
},
|
|
1903
|
+
// Bing's index also feeds Copilot answers and Microsoft's model training, unless a page
|
|
1904
|
+
// opts out with the nocache or noarchive robots tags.
|
|
1717
1905
|
{
|
|
1718
|
-
id: "
|
|
1719
|
-
pattern: /\
|
|
1720
|
-
class: "
|
|
1721
|
-
purpose: "
|
|
1722
|
-
|
|
1723
|
-
|
|
1906
|
+
id: "bingbot",
|
|
1907
|
+
pattern: /\bbingbot\b/i,
|
|
1908
|
+
class: "search_crawler",
|
|
1909
|
+
purpose: "search",
|
|
1910
|
+
alsoFor: ["ai_search", "ai_training"],
|
|
1911
|
+
operator: "Microsoft",
|
|
1912
|
+
product: "Bingbot",
|
|
1913
|
+
verify: { reverseDnsSuffixes: ["search.msn.com"], ipRangeSources: ["bing"] },
|
|
1914
|
+
docs: "https://blogs.bing.com/webmaster/August-2012/How-to-Verify-that-Bingbot-is-Bingbot/"
|
|
1724
1915
|
},
|
|
1725
|
-
// Other AI crawlers without published verification data.
|
|
1726
1916
|
{
|
|
1727
|
-
id: "
|
|
1728
|
-
pattern: /\
|
|
1729
|
-
class: "
|
|
1730
|
-
purpose: "
|
|
1731
|
-
operator: "
|
|
1732
|
-
product: "
|
|
1917
|
+
id: "msnbot",
|
|
1918
|
+
pattern: /\bmsnbot\b/i,
|
|
1919
|
+
class: "search_crawler",
|
|
1920
|
+
purpose: "search",
|
|
1921
|
+
operator: "Microsoft",
|
|
1922
|
+
product: "msnbot",
|
|
1923
|
+
verify: BING
|
|
1733
1924
|
},
|
|
1734
|
-
//
|
|
1925
|
+
// Apple. Applebot-Extended is a robots.txt token only and never appears in a user agent.
|
|
1926
|
+
// Applebot powers Spotlight, Siri and Safari search; Apple documents that what it crawls may
|
|
1927
|
+
// also train Apple's foundation models and give its AI answers context.
|
|
1735
1928
|
{
|
|
1736
|
-
id: "
|
|
1737
|
-
pattern: /\
|
|
1738
|
-
class: "
|
|
1739
|
-
purpose: "
|
|
1740
|
-
|
|
1741
|
-
|
|
1929
|
+
id: "applebot",
|
|
1930
|
+
pattern: /\bApplebot\b/,
|
|
1931
|
+
class: "search_crawler",
|
|
1932
|
+
purpose: "search",
|
|
1933
|
+
alsoFor: ["ai_search", "ai_training"],
|
|
1934
|
+
operator: "Apple",
|
|
1935
|
+
product: "Applebot",
|
|
1936
|
+
verify: { reverseDnsSuffixes: ["applebot.apple.com"], ipRangeSources: ["apple"] },
|
|
1937
|
+
docs: "https://support.apple.com/en-us/119829"
|
|
1742
1938
|
},
|
|
1939
|
+
// DuckDuckGo. DuckAssistBot fetches pages in real time for DuckAssist answers and is not
|
|
1940
|
+
// used for training, so it is an assistant fetching for a user, not an indexer.
|
|
1743
1941
|
{
|
|
1744
|
-
id: "
|
|
1745
|
-
pattern: /\
|
|
1942
|
+
id: "duckassistbot",
|
|
1943
|
+
pattern: /\bDuckAssistBot\b/,
|
|
1746
1944
|
class: "ai_agent",
|
|
1747
1945
|
purpose: "ai_assistant",
|
|
1748
|
-
operator: "
|
|
1749
|
-
product: "
|
|
1946
|
+
operator: "DuckDuckGo",
|
|
1947
|
+
product: "DuckAssistBot",
|
|
1948
|
+
verify: { ipRangeSources: ["duckassistbot"] },
|
|
1949
|
+
docs: "https://duckduckgo.com/duckduckgo-help-pages/results/duckassistbot"
|
|
1750
1950
|
},
|
|
1751
1951
|
{
|
|
1752
|
-
id: "
|
|
1753
|
-
pattern: /\
|
|
1754
|
-
class: "
|
|
1755
|
-
purpose: "
|
|
1756
|
-
operator: "
|
|
1757
|
-
product: "
|
|
1952
|
+
id: "duckduckbot",
|
|
1953
|
+
pattern: /\bDuckDuckBot\b/,
|
|
1954
|
+
class: "search_crawler",
|
|
1955
|
+
purpose: "search",
|
|
1956
|
+
operator: "DuckDuckGo",
|
|
1957
|
+
product: "DuckDuckBot",
|
|
1958
|
+
verify: { ipRangeSources: ["duckduckbot"] },
|
|
1959
|
+
docs: "https://duckduckgo.com/duckduckgo-help-pages/results/duckduckbot"
|
|
1758
1960
|
},
|
|
1961
|
+
// Yandex robots share one verification domain set.
|
|
1759
1962
|
{
|
|
1760
|
-
id: "
|
|
1761
|
-
pattern: /\
|
|
1762
|
-
class: "
|
|
1763
|
-
purpose: "
|
|
1764
|
-
operator: "
|
|
1765
|
-
product: "
|
|
1963
|
+
id: "yandexbot",
|
|
1964
|
+
pattern: /\bYandexBot\/\d/,
|
|
1965
|
+
class: "search_crawler",
|
|
1966
|
+
purpose: "search",
|
|
1967
|
+
operator: "Yandex",
|
|
1968
|
+
product: "YandexBot",
|
|
1969
|
+
verify: YANDEX,
|
|
1970
|
+
docs: "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots"
|
|
1766
1971
|
},
|
|
1767
1972
|
{
|
|
1768
|
-
id: "
|
|
1769
|
-
pattern: /\
|
|
1770
|
-
class: "
|
|
1771
|
-
purpose: "
|
|
1772
|
-
operator: "
|
|
1773
|
-
product: "
|
|
1973
|
+
id: "yandex-robot",
|
|
1974
|
+
pattern: /\bYandex[A-Za-z]*\/\d/,
|
|
1975
|
+
class: "search_crawler",
|
|
1976
|
+
purpose: "search",
|
|
1977
|
+
operator: "Yandex",
|
|
1978
|
+
product: "Yandex robot",
|
|
1979
|
+
verify: YANDEX,
|
|
1980
|
+
docs: "https://yandex.com/support/webmaster/en/robot-workings/check-yandex-robots"
|
|
1774
1981
|
},
|
|
1982
|
+
// Other search engines. Baidu's own verification page could not be retrieved, so it stays self-declared.
|
|
1775
1983
|
{
|
|
1776
|
-
id: "
|
|
1777
|
-
pattern: /\
|
|
1778
|
-
class: "
|
|
1779
|
-
purpose: "
|
|
1780
|
-
operator: "
|
|
1781
|
-
product: "
|
|
1984
|
+
id: "baiduspider",
|
|
1985
|
+
pattern: /\bBaiduspider\b/i,
|
|
1986
|
+
class: "search_crawler",
|
|
1987
|
+
purpose: "search",
|
|
1988
|
+
operator: "Baidu",
|
|
1989
|
+
product: "Baiduspider"
|
|
1782
1990
|
},
|
|
1783
1991
|
{
|
|
1784
|
-
id: "
|
|
1785
|
-
pattern: /\
|
|
1786
|
-
class: "
|
|
1787
|
-
purpose: "
|
|
1788
|
-
operator: "
|
|
1789
|
-
product: "
|
|
1992
|
+
id: "seznambot",
|
|
1993
|
+
pattern: /\bSeznamBot\b/,
|
|
1994
|
+
class: "search_crawler",
|
|
1995
|
+
purpose: "search",
|
|
1996
|
+
operator: "Seznam",
|
|
1997
|
+
product: "SeznamBot"
|
|
1790
1998
|
},
|
|
1791
1999
|
{
|
|
1792
|
-
id: "
|
|
1793
|
-
pattern: /\
|
|
1794
|
-
class: "
|
|
1795
|
-
purpose: "
|
|
1796
|
-
operator: "
|
|
1797
|
-
product: "
|
|
2000
|
+
id: "naver-yeti",
|
|
2001
|
+
pattern: /\bYeti\/\d/,
|
|
2002
|
+
class: "search_crawler",
|
|
2003
|
+
purpose: "search",
|
|
2004
|
+
operator: "Naver",
|
|
2005
|
+
product: "Yeti"
|
|
1798
2006
|
},
|
|
1799
2007
|
{
|
|
1800
|
-
id: "
|
|
1801
|
-
pattern: /\
|
|
1802
|
-
class: "
|
|
1803
|
-
purpose: "
|
|
2008
|
+
id: "petalbot",
|
|
2009
|
+
pattern: /\bPetalBot\b/,
|
|
2010
|
+
class: "search_crawler",
|
|
2011
|
+
purpose: "search",
|
|
1804
2012
|
operator: "Huawei",
|
|
1805
|
-
product: "
|
|
1806
|
-
},
|
|
1807
|
-
// Link previews.
|
|
1808
|
-
{
|
|
1809
|
-
id: "facebookexternalhit",
|
|
1810
|
-
pattern: /\bfacebookexternalhit\b/i,
|
|
1811
|
-
class: "preview",
|
|
1812
|
-
purpose: "preview",
|
|
1813
|
-
operator: "Meta",
|
|
1814
|
-
product: "facebookexternalhit",
|
|
1815
|
-
docs: META_DOCS
|
|
2013
|
+
product: "PetalBot"
|
|
1816
2014
|
},
|
|
1817
2015
|
{
|
|
1818
|
-
id: "
|
|
1819
|
-
pattern: /\
|
|
1820
|
-
class: "
|
|
1821
|
-
purpose: "
|
|
1822
|
-
operator: "
|
|
1823
|
-
product: "
|
|
2016
|
+
id: "mojeekbot",
|
|
2017
|
+
pattern: /\bMojeekBot\b/,
|
|
2018
|
+
class: "search_crawler",
|
|
2019
|
+
purpose: "search",
|
|
2020
|
+
operator: "Mojeek",
|
|
2021
|
+
product: "MojeekBot"
|
|
1824
2022
|
},
|
|
1825
|
-
// Telegram's preview UA mentions TwitterBot, so it must match first.
|
|
1826
2023
|
{
|
|
1827
|
-
id: "
|
|
1828
|
-
pattern: /\
|
|
1829
|
-
class: "
|
|
1830
|
-
purpose: "
|
|
1831
|
-
operator: "
|
|
1832
|
-
product: "
|
|
2024
|
+
id: "qwantbot",
|
|
2025
|
+
pattern: /\bQwantbot\b/i,
|
|
2026
|
+
class: "search_crawler",
|
|
2027
|
+
purpose: "search",
|
|
2028
|
+
operator: "Qwant",
|
|
2029
|
+
product: "Qwantbot"
|
|
1833
2030
|
},
|
|
1834
2031
|
{
|
|
1835
|
-
id: "
|
|
1836
|
-
pattern: /\
|
|
1837
|
-
class: "
|
|
1838
|
-
purpose: "
|
|
1839
|
-
operator: "
|
|
1840
|
-
product: "
|
|
2032
|
+
id: "sogou",
|
|
2033
|
+
pattern: /\bSogou [\w ]*spider\b/i,
|
|
2034
|
+
class: "search_crawler",
|
|
2035
|
+
purpose: "search",
|
|
2036
|
+
operator: "Sogou",
|
|
2037
|
+
product: "Sogou spider"
|
|
1841
2038
|
},
|
|
2039
|
+
// OpenAI.
|
|
1842
2040
|
{
|
|
1843
|
-
id: "
|
|
1844
|
-
pattern: /\
|
|
1845
|
-
class: "
|
|
1846
|
-
purpose: "
|
|
1847
|
-
operator: "
|
|
1848
|
-
product: "
|
|
2041
|
+
id: "chatgpt-user",
|
|
2042
|
+
pattern: /\bChatGPT-User\b/,
|
|
2043
|
+
class: "ai_agent",
|
|
2044
|
+
purpose: "ai_assistant",
|
|
2045
|
+
operator: "OpenAI",
|
|
2046
|
+
product: "ChatGPT-User",
|
|
2047
|
+
verify: { ipRangeSources: ["openai-chatgpt-user"] },
|
|
2048
|
+
docs: OPENAI_DOCS,
|
|
2049
|
+
userInitiated: true
|
|
1849
2050
|
},
|
|
1850
2051
|
{
|
|
1851
|
-
id: "
|
|
1852
|
-
pattern: /\
|
|
1853
|
-
class: "
|
|
1854
|
-
purpose: "
|
|
1855
|
-
operator: "
|
|
1856
|
-
product: "
|
|
2052
|
+
id: "oai-searchbot",
|
|
2053
|
+
pattern: /\bOAI-SearchBot\b/,
|
|
2054
|
+
class: "ai_crawler",
|
|
2055
|
+
purpose: "ai_search",
|
|
2056
|
+
operator: "OpenAI",
|
|
2057
|
+
product: "OAI-SearchBot",
|
|
2058
|
+
verify: { ipRangeSources: ["openai-searchbot"] },
|
|
2059
|
+
docs: OPENAI_DOCS
|
|
1857
2060
|
},
|
|
1858
2061
|
{
|
|
1859
|
-
id: "
|
|
1860
|
-
pattern: /\
|
|
1861
|
-
class: "
|
|
1862
|
-
purpose: "
|
|
1863
|
-
operator: "
|
|
1864
|
-
product: "
|
|
2062
|
+
id: "oai-adsbot",
|
|
2063
|
+
pattern: /\bOAI-AdsBot\b/,
|
|
2064
|
+
class: "ai_crawler",
|
|
2065
|
+
purpose: "ads",
|
|
2066
|
+
operator: "OpenAI",
|
|
2067
|
+
product: "OAI-AdsBot",
|
|
2068
|
+
verify: { ipRangeSources: ["openai-adsbot"] },
|
|
2069
|
+
docs: OPENAI_DOCS
|
|
1865
2070
|
},
|
|
1866
2071
|
{
|
|
1867
|
-
id: "
|
|
1868
|
-
pattern: /\
|
|
1869
|
-
class: "
|
|
1870
|
-
purpose: "
|
|
1871
|
-
operator: "
|
|
1872
|
-
product: "
|
|
2072
|
+
id: "gptbot",
|
|
2073
|
+
pattern: /\bGPTBot\b/,
|
|
2074
|
+
class: "ai_crawler",
|
|
2075
|
+
purpose: "ai_training",
|
|
2076
|
+
operator: "OpenAI",
|
|
2077
|
+
product: "GPTBot",
|
|
2078
|
+
verify: { ipRangeSources: ["openai-gptbot"] },
|
|
2079
|
+
docs: OPENAI_DOCS
|
|
1873
2080
|
},
|
|
2081
|
+
// Anthropic: one published range file covers every Anthropic bot.
|
|
1874
2082
|
{
|
|
1875
|
-
id: "
|
|
1876
|
-
pattern: /\
|
|
1877
|
-
class: "
|
|
1878
|
-
purpose: "
|
|
1879
|
-
operator: "
|
|
1880
|
-
product: "
|
|
2083
|
+
id: "claude-user",
|
|
2084
|
+
pattern: /\bClaude-User\b/,
|
|
2085
|
+
class: "ai_agent",
|
|
2086
|
+
purpose: "ai_assistant",
|
|
2087
|
+
operator: "Anthropic",
|
|
2088
|
+
product: "Claude-User",
|
|
2089
|
+
verify: ANTHROPIC,
|
|
2090
|
+
docs: ANTHROPIC_DOCS,
|
|
2091
|
+
userInitiated: true
|
|
1881
2092
|
},
|
|
1882
2093
|
{
|
|
1883
|
-
id: "
|
|
1884
|
-
pattern: /\
|
|
1885
|
-
class: "
|
|
1886
|
-
purpose: "
|
|
1887
|
-
operator: "
|
|
1888
|
-
product: "
|
|
2094
|
+
id: "claude-searchbot",
|
|
2095
|
+
pattern: /\bClaude-SearchBot\b/,
|
|
2096
|
+
class: "ai_crawler",
|
|
2097
|
+
purpose: "ai_search",
|
|
2098
|
+
operator: "Anthropic",
|
|
2099
|
+
product: "Claude-SearchBot",
|
|
2100
|
+
verify: ANTHROPIC,
|
|
2101
|
+
docs: ANTHROPIC_DOCS
|
|
1889
2102
|
},
|
|
1890
2103
|
{
|
|
1891
|
-
id: "
|
|
1892
|
-
pattern: /\
|
|
1893
|
-
class: "
|
|
1894
|
-
purpose: "
|
|
1895
|
-
operator: "
|
|
1896
|
-
product: "
|
|
2104
|
+
id: "claudebot",
|
|
2105
|
+
pattern: /\bClaudeBot\b/,
|
|
2106
|
+
class: "ai_crawler",
|
|
2107
|
+
purpose: "ai_training",
|
|
2108
|
+
operator: "Anthropic",
|
|
2109
|
+
product: "ClaudeBot",
|
|
2110
|
+
verify: ANTHROPIC,
|
|
2111
|
+
docs: ANTHROPIC_DOCS
|
|
1897
2112
|
},
|
|
1898
2113
|
{
|
|
1899
|
-
id: "
|
|
1900
|
-
pattern: /\
|
|
1901
|
-
class: "
|
|
1902
|
-
purpose: "
|
|
1903
|
-
operator: "
|
|
1904
|
-
product: "
|
|
2114
|
+
id: "claude-web",
|
|
2115
|
+
pattern: /\bClaude-Web\b/,
|
|
2116
|
+
class: "ai_agent",
|
|
2117
|
+
purpose: "ai_assistant",
|
|
2118
|
+
operator: "Anthropic",
|
|
2119
|
+
product: "Claude-Web",
|
|
2120
|
+
verify: ANTHROPIC,
|
|
2121
|
+
docs: ANTHROPIC_DOCS
|
|
1905
2122
|
},
|
|
1906
2123
|
{
|
|
1907
|
-
id: "
|
|
1908
|
-
pattern: /\
|
|
1909
|
-
class: "
|
|
1910
|
-
purpose: "
|
|
1911
|
-
operator: "
|
|
1912
|
-
product: "
|
|
2124
|
+
id: "anthropic-ai",
|
|
2125
|
+
pattern: /\banthropic-ai\b/i,
|
|
2126
|
+
class: "ai_crawler",
|
|
2127
|
+
purpose: "ai_training",
|
|
2128
|
+
operator: "Anthropic",
|
|
2129
|
+
product: "anthropic-ai",
|
|
2130
|
+
verify: ANTHROPIC,
|
|
2131
|
+
docs: ANTHROPIC_DOCS
|
|
1913
2132
|
},
|
|
2133
|
+
// Perplexity.
|
|
1914
2134
|
{
|
|
1915
|
-
id: "
|
|
1916
|
-
pattern: /\
|
|
1917
|
-
class: "
|
|
1918
|
-
purpose: "
|
|
1919
|
-
operator: "
|
|
1920
|
-
product: "
|
|
2135
|
+
id: "perplexity-user",
|
|
2136
|
+
pattern: /\bPerplexity-User\b/,
|
|
2137
|
+
class: "ai_agent",
|
|
2138
|
+
purpose: "ai_assistant",
|
|
2139
|
+
operator: "Perplexity",
|
|
2140
|
+
product: "Perplexity-User",
|
|
2141
|
+
verify: { ipRangeSources: ["perplexity-user"] },
|
|
2142
|
+
docs: PERPLEXITY_DOCS,
|
|
2143
|
+
userInitiated: true
|
|
1921
2144
|
},
|
|
1922
2145
|
{
|
|
1923
|
-
id: "
|
|
1924
|
-
pattern: /\
|
|
1925
|
-
class: "
|
|
1926
|
-
purpose: "
|
|
1927
|
-
operator: "
|
|
1928
|
-
product: "
|
|
2146
|
+
id: "perplexitybot",
|
|
2147
|
+
pattern: /\bPerplexityBot\b/,
|
|
2148
|
+
class: "ai_crawler",
|
|
2149
|
+
purpose: "ai_search",
|
|
2150
|
+
operator: "Perplexity",
|
|
2151
|
+
product: "PerplexityBot",
|
|
2152
|
+
verify: { ipRangeSources: ["perplexity-bot"] },
|
|
2153
|
+
docs: PERPLEXITY_DOCS
|
|
1929
2154
|
},
|
|
2155
|
+
// Mistral AI.
|
|
1930
2156
|
{
|
|
1931
|
-
id: "
|
|
1932
|
-
pattern: /\
|
|
1933
|
-
class: "
|
|
1934
|
-
purpose: "
|
|
1935
|
-
operator:
|
|
1936
|
-
product: "
|
|
2157
|
+
id: "mistralai-user",
|
|
2158
|
+
pattern: /\bMistralAI-User\b/,
|
|
2159
|
+
class: "ai_agent",
|
|
2160
|
+
purpose: "ai_assistant",
|
|
2161
|
+
operator: "Mistral AI",
|
|
2162
|
+
product: "MistralAI-User",
|
|
2163
|
+
verify: { ipRangeSources: ["mistral-user"] },
|
|
2164
|
+
docs: MISTRAL_DOCS,
|
|
2165
|
+
userInitiated: true
|
|
1937
2166
|
},
|
|
1938
|
-
// Uptime and monitoring.
|
|
1939
2167
|
{
|
|
1940
|
-
id: "
|
|
1941
|
-
pattern: /\
|
|
1942
|
-
class: "
|
|
1943
|
-
purpose: "
|
|
1944
|
-
operator: "
|
|
1945
|
-
product: "
|
|
2168
|
+
id: "mistralai-index",
|
|
2169
|
+
pattern: /\bMistralAI-Index\b/,
|
|
2170
|
+
class: "ai_crawler",
|
|
2171
|
+
purpose: "ai_search",
|
|
2172
|
+
operator: "Mistral AI",
|
|
2173
|
+
product: "MistralAI-Index",
|
|
2174
|
+
verify: { ipRangeSources: ["mistral-index"] },
|
|
2175
|
+
docs: MISTRAL_DOCS
|
|
1946
2176
|
},
|
|
1947
2177
|
{
|
|
1948
|
-
id: "
|
|
1949
|
-
pattern: /\
|
|
1950
|
-
class: "
|
|
1951
|
-
purpose: "
|
|
1952
|
-
operator: "
|
|
1953
|
-
product: "
|
|
2178
|
+
id: "mistralai-training",
|
|
2179
|
+
pattern: /\bMistralAI-Training\b/,
|
|
2180
|
+
class: "ai_crawler",
|
|
2181
|
+
purpose: "ai_training",
|
|
2182
|
+
operator: "Mistral AI",
|
|
2183
|
+
product: "MistralAI-Training",
|
|
2184
|
+
docs: MISTRAL_DOCS
|
|
1954
2185
|
},
|
|
2186
|
+
// Amazon.
|
|
1955
2187
|
{
|
|
1956
|
-
id: "
|
|
1957
|
-
pattern: /\
|
|
1958
|
-
class: "
|
|
1959
|
-
purpose: "
|
|
1960
|
-
operator: "
|
|
1961
|
-
product: "
|
|
2188
|
+
id: "amzn-user",
|
|
2189
|
+
pattern: /\bAmzn-User\b/,
|
|
2190
|
+
class: "ai_agent",
|
|
2191
|
+
purpose: "ai_assistant",
|
|
2192
|
+
operator: "Amazon",
|
|
2193
|
+
product: "Amzn-User",
|
|
2194
|
+
verify: { ipRangeSources: ["amazon-user"] },
|
|
2195
|
+
docs: AMAZON_DOCS,
|
|
2196
|
+
userInitiated: true
|
|
1962
2197
|
},
|
|
2198
|
+
// Amzn-SearchBot feeds Alexa and Amazon's AI search, not model training.
|
|
1963
2199
|
{
|
|
1964
|
-
id: "
|
|
1965
|
-
pattern: /\
|
|
1966
|
-
class: "
|
|
1967
|
-
purpose: "
|
|
1968
|
-
operator: "
|
|
1969
|
-
product: "
|
|
2200
|
+
id: "amzn-searchbot",
|
|
2201
|
+
pattern: /\bAmzn-SearchBot\b/,
|
|
2202
|
+
class: "ai_crawler",
|
|
2203
|
+
purpose: "ai_search",
|
|
2204
|
+
operator: "Amazon",
|
|
2205
|
+
product: "Amzn-SearchBot",
|
|
2206
|
+
verify: { ipRangeSources: ["amazon-searchbot"] },
|
|
2207
|
+
docs: AMAZON_DOCS
|
|
1970
2208
|
},
|
|
1971
2209
|
{
|
|
1972
|
-
id: "
|
|
1973
|
-
pattern: /\
|
|
1974
|
-
class: "
|
|
1975
|
-
purpose: "
|
|
1976
|
-
operator: "
|
|
1977
|
-
product: "
|
|
2210
|
+
id: "amazonbot",
|
|
2211
|
+
pattern: /\bAmazonbot\b/,
|
|
2212
|
+
class: "ai_crawler",
|
|
2213
|
+
purpose: "ai_training",
|
|
2214
|
+
operator: "Amazon",
|
|
2215
|
+
product: "Amazonbot",
|
|
2216
|
+
verify: { ipRangeSources: ["amazon-amazonbot"] },
|
|
2217
|
+
docs: AMAZON_DOCS
|
|
1978
2218
|
},
|
|
2219
|
+
// Common Crawl publishes ranges and forward-confirmed reverse DNS. Its open archive is the
|
|
2220
|
+
// most common source of AI training data; there is no separate archive purpose.
|
|
1979
2221
|
{
|
|
1980
|
-
id: "
|
|
1981
|
-
pattern: /\
|
|
1982
|
-
class: "
|
|
1983
|
-
purpose: "
|
|
1984
|
-
operator: "
|
|
1985
|
-
product: "
|
|
2222
|
+
id: "ccbot",
|
|
2223
|
+
pattern: /\bCCBot\b/,
|
|
2224
|
+
class: "ai_crawler",
|
|
2225
|
+
purpose: "ai_training",
|
|
2226
|
+
operator: "Common Crawl",
|
|
2227
|
+
product: "CCBot",
|
|
2228
|
+
verify: { reverseDnsSuffixes: ["crawl.commoncrawl.org"], ipRangeSources: ["commoncrawl"] },
|
|
2229
|
+
docs: "https://commoncrawl.org/ccbot"
|
|
1986
2230
|
},
|
|
2231
|
+
// Meta publishes no ranges or DNS names for its crawlers, so all are self-declared.
|
|
1987
2232
|
{
|
|
1988
|
-
id: "
|
|
1989
|
-
pattern: /\
|
|
1990
|
-
class: "
|
|
1991
|
-
purpose: "
|
|
1992
|
-
operator: "
|
|
1993
|
-
product: "
|
|
2233
|
+
id: "meta-externalfetcher",
|
|
2234
|
+
pattern: /\bmeta-externalfetcher\b/i,
|
|
2235
|
+
class: "ai_agent",
|
|
2236
|
+
purpose: "ai_assistant",
|
|
2237
|
+
operator: "Meta",
|
|
2238
|
+
product: "Meta-ExternalFetcher",
|
|
2239
|
+
docs: META_DOCS,
|
|
2240
|
+
userInitiated: true
|
|
1994
2241
|
},
|
|
1995
2242
|
{
|
|
1996
|
-
id: "
|
|
1997
|
-
pattern: /\
|
|
1998
|
-
class: "
|
|
1999
|
-
purpose: "
|
|
2000
|
-
operator: "
|
|
2001
|
-
product: "
|
|
2243
|
+
id: "meta-externalagent",
|
|
2244
|
+
pattern: /\bmeta-externalagent\b/i,
|
|
2245
|
+
class: "ai_crawler",
|
|
2246
|
+
purpose: "ai_training",
|
|
2247
|
+
operator: "Meta",
|
|
2248
|
+
product: "Meta-ExternalAgent",
|
|
2249
|
+
docs: META_DOCS
|
|
2002
2250
|
},
|
|
2251
|
+
// Meta-WebIndexer feeds Meta AI's search results, so it is an AI crawler.
|
|
2003
2252
|
{
|
|
2004
|
-
id: "
|
|
2005
|
-
pattern: /\
|
|
2006
|
-
class: "
|
|
2007
|
-
purpose: "
|
|
2008
|
-
operator: "
|
|
2009
|
-
product: "
|
|
2253
|
+
id: "meta-webindexer",
|
|
2254
|
+
pattern: /\bmeta-webindexer\b/i,
|
|
2255
|
+
class: "ai_crawler",
|
|
2256
|
+
purpose: "ai_search",
|
|
2257
|
+
operator: "Meta",
|
|
2258
|
+
product: "Meta-WebIndexer",
|
|
2259
|
+
docs: META_DOCS
|
|
2010
2260
|
},
|
|
2011
2261
|
{
|
|
2012
|
-
id: "
|
|
2013
|
-
pattern: /\
|
|
2014
|
-
class: "
|
|
2015
|
-
purpose: "
|
|
2016
|
-
operator: "
|
|
2017
|
-
product: "
|
|
2262
|
+
id: "meta-externalads",
|
|
2263
|
+
pattern: /\bmeta-externalads\b/i,
|
|
2264
|
+
class: "search_crawler",
|
|
2265
|
+
purpose: "ads",
|
|
2266
|
+
operator: "Meta",
|
|
2267
|
+
product: "Meta-ExternalAds",
|
|
2268
|
+
docs: META_DOCS
|
|
2018
2269
|
},
|
|
2019
2270
|
{
|
|
2020
|
-
id: "
|
|
2021
|
-
pattern: /\
|
|
2022
|
-
class: "
|
|
2023
|
-
purpose: "
|
|
2024
|
-
operator: "
|
|
2025
|
-
product: "
|
|
2271
|
+
id: "facebookbot",
|
|
2272
|
+
pattern: /\bFacebookBot\b/,
|
|
2273
|
+
class: "ai_crawler",
|
|
2274
|
+
purpose: "ai_training",
|
|
2275
|
+
operator: "Meta",
|
|
2276
|
+
product: "FacebookBot"
|
|
2026
2277
|
},
|
|
2278
|
+
// Other AI crawlers without published verification data.
|
|
2027
2279
|
{
|
|
2028
|
-
id: "
|
|
2029
|
-
pattern: /\
|
|
2030
|
-
class: "
|
|
2031
|
-
purpose: "
|
|
2032
|
-
operator:
|
|
2033
|
-
product: "
|
|
2280
|
+
id: "bytespider",
|
|
2281
|
+
pattern: /\bBytespider\b/i,
|
|
2282
|
+
class: "ai_crawler",
|
|
2283
|
+
purpose: "ai_training",
|
|
2284
|
+
operator: "ByteDance",
|
|
2285
|
+
product: "Bytespider"
|
|
2034
2286
|
},
|
|
2287
|
+
// cohere-ai fetches for Cohere's assistant; the training crawler names itself separately.
|
|
2035
2288
|
{
|
|
2036
|
-
id: "
|
|
2037
|
-
pattern: /\
|
|
2038
|
-
class: "
|
|
2039
|
-
purpose: "
|
|
2040
|
-
operator:
|
|
2041
|
-
product: "
|
|
2289
|
+
id: "cohere-training",
|
|
2290
|
+
pattern: /\bcohere-training-data-crawler\b/i,
|
|
2291
|
+
class: "ai_crawler",
|
|
2292
|
+
purpose: "ai_training",
|
|
2293
|
+
operator: "Cohere",
|
|
2294
|
+
product: "cohere-training-data-crawler"
|
|
2042
2295
|
},
|
|
2043
|
-
// SEO crawlers: operated by a known company, but not visitors, search engines or AI.
|
|
2044
2296
|
{
|
|
2045
|
-
id: "
|
|
2046
|
-
pattern: /\
|
|
2047
|
-
class: "
|
|
2048
|
-
purpose: "
|
|
2049
|
-
operator: "
|
|
2050
|
-
product: "
|
|
2297
|
+
id: "cohere",
|
|
2298
|
+
pattern: /\bcohere-ai\b/i,
|
|
2299
|
+
class: "ai_agent",
|
|
2300
|
+
purpose: "ai_assistant",
|
|
2301
|
+
operator: "Cohere",
|
|
2302
|
+
product: "cohere-ai"
|
|
2051
2303
|
},
|
|
2052
2304
|
{
|
|
2053
|
-
id: "
|
|
2054
|
-
pattern: /\
|
|
2055
|
-
class: "
|
|
2056
|
-
purpose: "
|
|
2057
|
-
operator: "
|
|
2058
|
-
product: "
|
|
2305
|
+
id: "diffbot",
|
|
2306
|
+
pattern: /\bDiffbot\b/i,
|
|
2307
|
+
class: "ai_crawler",
|
|
2308
|
+
purpose: "ai_training",
|
|
2309
|
+
operator: "Diffbot",
|
|
2310
|
+
product: "Diffbot"
|
|
2059
2311
|
},
|
|
2060
2312
|
{
|
|
2061
|
-
id: "
|
|
2062
|
-
pattern: /\
|
|
2063
|
-
class: "
|
|
2064
|
-
purpose: "
|
|
2065
|
-
operator: "
|
|
2066
|
-
product: "
|
|
2313
|
+
id: "youbot",
|
|
2314
|
+
pattern: /\bYouBot\b/,
|
|
2315
|
+
class: "ai_crawler",
|
|
2316
|
+
purpose: "ai_search",
|
|
2317
|
+
operator: "You.com",
|
|
2318
|
+
product: "YouBot"
|
|
2067
2319
|
},
|
|
2068
2320
|
{
|
|
2069
|
-
id: "
|
|
2070
|
-
pattern: /\
|
|
2071
|
-
class: "
|
|
2072
|
-
purpose: "
|
|
2073
|
-
operator: "
|
|
2074
|
-
product: "
|
|
2321
|
+
id: "ai2bot",
|
|
2322
|
+
pattern: /\bAI2Bot\b/i,
|
|
2323
|
+
class: "ai_crawler",
|
|
2324
|
+
purpose: "ai_training",
|
|
2325
|
+
operator: "Allen Institute for AI",
|
|
2326
|
+
product: "AI2Bot"
|
|
2075
2327
|
},
|
|
2076
2328
|
{
|
|
2077
|
-
id: "
|
|
2078
|
-
pattern: /\
|
|
2079
|
-
class: "
|
|
2080
|
-
purpose: "
|
|
2081
|
-
operator: "
|
|
2082
|
-
product: "
|
|
2329
|
+
id: "timpibot",
|
|
2330
|
+
pattern: /\bTimpibot\b/i,
|
|
2331
|
+
class: "ai_crawler",
|
|
2332
|
+
purpose: "ai_training",
|
|
2333
|
+
operator: "Timpi",
|
|
2334
|
+
product: "Timpibot"
|
|
2083
2335
|
},
|
|
2084
2336
|
{
|
|
2085
|
-
id: "
|
|
2086
|
-
pattern: /\
|
|
2087
|
-
class: "
|
|
2088
|
-
purpose: "
|
|
2089
|
-
operator: "
|
|
2090
|
-
product: "
|
|
2337
|
+
id: "imagesiftbot",
|
|
2338
|
+
pattern: /\bImagesiftBot\b/i,
|
|
2339
|
+
class: "ai_crawler",
|
|
2340
|
+
purpose: "ai_training",
|
|
2341
|
+
operator: "ImageSift",
|
|
2342
|
+
product: "ImagesiftBot"
|
|
2091
2343
|
},
|
|
2092
2344
|
{
|
|
2093
|
-
id: "
|
|
2094
|
-
pattern: /\
|
|
2095
|
-
class: "
|
|
2096
|
-
purpose: "
|
|
2097
|
-
operator: "
|
|
2098
|
-
product: "
|
|
2345
|
+
id: "omgilibot",
|
|
2346
|
+
pattern: /\bomgili(?:bot)?\b/i,
|
|
2347
|
+
class: "ai_crawler",
|
|
2348
|
+
purpose: "ai_training",
|
|
2349
|
+
operator: "Webz.io",
|
|
2350
|
+
product: "Omgilibot"
|
|
2099
2351
|
},
|
|
2100
2352
|
{
|
|
2101
|
-
id: "
|
|
2102
|
-
pattern: /\
|
|
2103
|
-
class: "
|
|
2104
|
-
purpose: "
|
|
2105
|
-
operator: "
|
|
2106
|
-
product: "
|
|
2353
|
+
id: "pangubot",
|
|
2354
|
+
pattern: /\bPanguBot\b/,
|
|
2355
|
+
class: "ai_crawler",
|
|
2356
|
+
purpose: "ai_training",
|
|
2357
|
+
operator: "Huawei",
|
|
2358
|
+
product: "PanguBot"
|
|
2107
2359
|
},
|
|
2360
|
+
// Link previews.
|
|
2108
2361
|
{
|
|
2109
|
-
id: "
|
|
2110
|
-
pattern: /\
|
|
2111
|
-
class: "
|
|
2112
|
-
purpose: "
|
|
2113
|
-
operator: "
|
|
2114
|
-
product: "
|
|
2362
|
+
id: "facebookexternalhit",
|
|
2363
|
+
pattern: /\bfacebookexternalhit\b/i,
|
|
2364
|
+
class: "preview",
|
|
2365
|
+
purpose: "preview",
|
|
2366
|
+
operator: "Meta",
|
|
2367
|
+
product: "facebookexternalhit",
|
|
2368
|
+
docs: META_DOCS
|
|
2115
2369
|
},
|
|
2116
|
-
// Automation tools and HTTP libraries: no operator is claimed.
|
|
2117
2370
|
{
|
|
2118
|
-
id: "
|
|
2119
|
-
pattern: /\
|
|
2120
|
-
class: "
|
|
2121
|
-
purpose: "
|
|
2122
|
-
operator:
|
|
2123
|
-
product: "
|
|
2371
|
+
id: "facebot",
|
|
2372
|
+
pattern: /\bFacebot\b/,
|
|
2373
|
+
class: "preview",
|
|
2374
|
+
purpose: "preview",
|
|
2375
|
+
operator: "Meta",
|
|
2376
|
+
product: "Facebot"
|
|
2124
2377
|
},
|
|
2378
|
+
// Telegram's preview UA mentions TwitterBot, so it must match first.
|
|
2125
2379
|
{
|
|
2126
|
-
id: "
|
|
2127
|
-
pattern: /\
|
|
2128
|
-
class: "
|
|
2129
|
-
purpose: "
|
|
2130
|
-
operator:
|
|
2131
|
-
product: "
|
|
2380
|
+
id: "telegrambot",
|
|
2381
|
+
pattern: /\bTelegramBot\b/,
|
|
2382
|
+
class: "preview",
|
|
2383
|
+
purpose: "preview",
|
|
2384
|
+
operator: "Telegram",
|
|
2385
|
+
product: "TelegramBot"
|
|
2132
2386
|
},
|
|
2133
2387
|
{
|
|
2134
|
-
id: "
|
|
2135
|
-
pattern: /\
|
|
2136
|
-
class: "
|
|
2137
|
-
purpose: "
|
|
2138
|
-
operator:
|
|
2139
|
-
product: "
|
|
2388
|
+
id: "twitterbot",
|
|
2389
|
+
pattern: /\bTwitterbot\b/i,
|
|
2390
|
+
class: "preview",
|
|
2391
|
+
purpose: "preview",
|
|
2392
|
+
operator: "X",
|
|
2393
|
+
product: "Twitterbot"
|
|
2140
2394
|
},
|
|
2141
2395
|
{
|
|
2142
|
-
id: "
|
|
2143
|
-
pattern: /\
|
|
2144
|
-
class: "
|
|
2145
|
-
purpose: "
|
|
2146
|
-
operator:
|
|
2147
|
-
product: "
|
|
2396
|
+
id: "slackbot",
|
|
2397
|
+
pattern: /\bSlackbot(?:-LinkExpanding)?\b|\bSlack-ImgProxy\b/,
|
|
2398
|
+
class: "preview",
|
|
2399
|
+
purpose: "preview",
|
|
2400
|
+
operator: "Slack",
|
|
2401
|
+
product: "Slackbot"
|
|
2148
2402
|
},
|
|
2149
2403
|
{
|
|
2150
|
-
id: "
|
|
2151
|
-
pattern: /\
|
|
2152
|
-
class: "
|
|
2153
|
-
purpose: "
|
|
2154
|
-
operator:
|
|
2155
|
-
product: "
|
|
2404
|
+
id: "discordbot",
|
|
2405
|
+
pattern: /\bDiscordbot\b/i,
|
|
2406
|
+
class: "preview",
|
|
2407
|
+
purpose: "preview",
|
|
2408
|
+
operator: "Discord",
|
|
2409
|
+
product: "Discordbot"
|
|
2156
2410
|
},
|
|
2157
2411
|
{
|
|
2158
|
-
id: "
|
|
2159
|
-
pattern:
|
|
2160
|
-
class: "
|
|
2161
|
-
purpose: "
|
|
2162
|
-
operator:
|
|
2163
|
-
product: "
|
|
2412
|
+
id: "linkedinbot",
|
|
2413
|
+
pattern: /\bLinkedInBot\b/i,
|
|
2414
|
+
class: "preview",
|
|
2415
|
+
purpose: "preview",
|
|
2416
|
+
operator: "LinkedIn",
|
|
2417
|
+
product: "LinkedInBot"
|
|
2164
2418
|
},
|
|
2165
2419
|
{
|
|
2166
|
-
id: "
|
|
2167
|
-
pattern: /\
|
|
2168
|
-
class: "
|
|
2169
|
-
purpose: "
|
|
2170
|
-
operator:
|
|
2171
|
-
product: "
|
|
2420
|
+
id: "whatsapp",
|
|
2421
|
+
pattern: /\bWhatsApp\/\d/,
|
|
2422
|
+
class: "preview",
|
|
2423
|
+
purpose: "preview",
|
|
2424
|
+
operator: "Meta",
|
|
2425
|
+
product: "WhatsApp"
|
|
2172
2426
|
},
|
|
2173
2427
|
{
|
|
2174
|
-
id: "
|
|
2175
|
-
pattern: /\
|
|
2176
|
-
class: "
|
|
2177
|
-
purpose: "
|
|
2178
|
-
operator:
|
|
2179
|
-
product: "
|
|
2428
|
+
id: "pinterestbot",
|
|
2429
|
+
pattern: /\bPinterestbot\b|\bPinterest\/0\.\d/i,
|
|
2430
|
+
class: "preview",
|
|
2431
|
+
purpose: "preview",
|
|
2432
|
+
operator: "Pinterest",
|
|
2433
|
+
product: "Pinterestbot"
|
|
2180
2434
|
},
|
|
2181
2435
|
{
|
|
2182
|
-
id: "
|
|
2183
|
-
pattern: /\
|
|
2184
|
-
class: "
|
|
2185
|
-
purpose: "
|
|
2186
|
-
operator:
|
|
2187
|
-
product: "
|
|
2436
|
+
id: "redditbot",
|
|
2437
|
+
pattern: /\bredditbot\b/i,
|
|
2438
|
+
class: "preview",
|
|
2439
|
+
purpose: "preview",
|
|
2440
|
+
operator: "Reddit",
|
|
2441
|
+
product: "redditbot"
|
|
2188
2442
|
},
|
|
2189
2443
|
{
|
|
2190
|
-
id: "
|
|
2191
|
-
pattern: /\
|
|
2192
|
-
class: "
|
|
2193
|
-
purpose: "
|
|
2194
|
-
operator:
|
|
2195
|
-
product: "
|
|
2444
|
+
id: "embedly",
|
|
2445
|
+
pattern: /\bEmbedly\b/i,
|
|
2446
|
+
class: "preview",
|
|
2447
|
+
purpose: "preview",
|
|
2448
|
+
operator: "Embedly",
|
|
2449
|
+
product: "Embedly"
|
|
2196
2450
|
},
|
|
2197
2451
|
{
|
|
2198
|
-
id: "
|
|
2199
|
-
pattern: /\
|
|
2200
|
-
class: "
|
|
2201
|
-
purpose: "
|
|
2202
|
-
operator:
|
|
2203
|
-
product: "
|
|
2452
|
+
id: "skypeuripreview",
|
|
2453
|
+
pattern: /\bSkypeUriPreview\b/,
|
|
2454
|
+
class: "preview",
|
|
2455
|
+
purpose: "preview",
|
|
2456
|
+
operator: "Microsoft",
|
|
2457
|
+
product: "SkypeUriPreview"
|
|
2204
2458
|
},
|
|
2205
2459
|
{
|
|
2206
|
-
id: "
|
|
2207
|
-
pattern: /\
|
|
2208
|
-
class: "
|
|
2209
|
-
purpose: "
|
|
2210
|
-
operator:
|
|
2211
|
-
product: "
|
|
2460
|
+
id: "iframely",
|
|
2461
|
+
pattern: /\bIframely\b/i,
|
|
2462
|
+
class: "preview",
|
|
2463
|
+
purpose: "preview",
|
|
2464
|
+
operator: "Iframely",
|
|
2465
|
+
product: "Iframely"
|
|
2212
2466
|
},
|
|
2213
2467
|
{
|
|
2214
|
-
id: "
|
|
2215
|
-
pattern: /\
|
|
2216
|
-
class: "
|
|
2217
|
-
purpose: "
|
|
2218
|
-
operator:
|
|
2219
|
-
product: "
|
|
2468
|
+
id: "bluesky-cardyb",
|
|
2469
|
+
pattern: /\bCardyb\b/i,
|
|
2470
|
+
class: "preview",
|
|
2471
|
+
purpose: "preview",
|
|
2472
|
+
operator: "Bluesky",
|
|
2473
|
+
product: "Cardyb"
|
|
2220
2474
|
},
|
|
2221
2475
|
{
|
|
2222
|
-
id: "
|
|
2223
|
-
pattern: /\
|
|
2224
|
-
class: "
|
|
2225
|
-
purpose: "
|
|
2226
|
-
operator:
|
|
2227
|
-
product: "
|
|
2476
|
+
id: "vkshare",
|
|
2477
|
+
pattern: /\bvkShare\b/,
|
|
2478
|
+
class: "preview",
|
|
2479
|
+
purpose: "preview",
|
|
2480
|
+
operator: "VK",
|
|
2481
|
+
product: "vkShare"
|
|
2228
2482
|
},
|
|
2229
2483
|
{
|
|
2230
|
-
id: "
|
|
2231
|
-
pattern: /\
|
|
2232
|
-
class: "
|
|
2233
|
-
purpose: "
|
|
2484
|
+
id: "mastodon",
|
|
2485
|
+
pattern: /\bMastodon\/\d/,
|
|
2486
|
+
class: "preview",
|
|
2487
|
+
purpose: "preview",
|
|
2234
2488
|
operator: null,
|
|
2235
|
-
product: "
|
|
2489
|
+
product: "Mastodon"
|
|
2236
2490
|
},
|
|
2491
|
+
// Uptime and monitoring.
|
|
2237
2492
|
{
|
|
2238
|
-
id: "
|
|
2239
|
-
pattern:
|
|
2240
|
-
class: "
|
|
2241
|
-
purpose: "
|
|
2242
|
-
operator:
|
|
2243
|
-
product: "
|
|
2493
|
+
id: "uptimerobot",
|
|
2494
|
+
pattern: /\bUptimeRobot\b/i,
|
|
2495
|
+
class: "monitor",
|
|
2496
|
+
purpose: "monitor",
|
|
2497
|
+
operator: "UptimeRobot",
|
|
2498
|
+
product: "UptimeRobot"
|
|
2244
2499
|
},
|
|
2245
2500
|
{
|
|
2246
|
-
id: "
|
|
2247
|
-
pattern: /\
|
|
2248
|
-
class: "
|
|
2249
|
-
purpose: "
|
|
2250
|
-
operator:
|
|
2251
|
-
product: "
|
|
2501
|
+
id: "pingdom",
|
|
2502
|
+
pattern: /\bPingdom/i,
|
|
2503
|
+
class: "monitor",
|
|
2504
|
+
purpose: "monitor",
|
|
2505
|
+
operator: "Pingdom",
|
|
2506
|
+
product: "Pingdom"
|
|
2252
2507
|
},
|
|
2253
2508
|
{
|
|
2254
|
-
id: "
|
|
2255
|
-
pattern: /\
|
|
2256
|
-
class: "
|
|
2257
|
-
purpose: "
|
|
2258
|
-
operator:
|
|
2259
|
-
product: "
|
|
2509
|
+
id: "statuscake",
|
|
2510
|
+
pattern: /\bStatusCake\b/i,
|
|
2511
|
+
class: "monitor",
|
|
2512
|
+
purpose: "monitor",
|
|
2513
|
+
operator: "StatusCake",
|
|
2514
|
+
product: "StatusCake"
|
|
2260
2515
|
},
|
|
2261
2516
|
{
|
|
2262
|
-
id: "
|
|
2263
|
-
pattern:
|
|
2264
|
-
class: "
|
|
2265
|
-
purpose: "
|
|
2266
|
-
operator:
|
|
2267
|
-
product: "
|
|
2517
|
+
id: "better-stack",
|
|
2518
|
+
pattern: /\bBetter ?(?:Uptime|Stack)\b/i,
|
|
2519
|
+
class: "monitor",
|
|
2520
|
+
purpose: "monitor",
|
|
2521
|
+
operator: "Better Stack",
|
|
2522
|
+
product: "Better Stack Uptime"
|
|
2268
2523
|
},
|
|
2269
2524
|
{
|
|
2270
|
-
id: "
|
|
2271
|
-
pattern: /\
|
|
2272
|
-
class: "
|
|
2273
|
-
purpose: "
|
|
2274
|
-
operator:
|
|
2275
|
-
product: "
|
|
2525
|
+
id: "datadog-synthetics",
|
|
2526
|
+
pattern: /\bDatadogSynthetics\b/i,
|
|
2527
|
+
class: "monitor",
|
|
2528
|
+
purpose: "monitor",
|
|
2529
|
+
operator: "Datadog",
|
|
2530
|
+
product: "Datadog Synthetics"
|
|
2276
2531
|
},
|
|
2277
2532
|
{
|
|
2278
|
-
id: "
|
|
2279
|
-
pattern: /\
|
|
2280
|
-
class: "
|
|
2281
|
-
purpose: "
|
|
2282
|
-
operator:
|
|
2283
|
-
product: "
|
|
2533
|
+
id: "site24x7",
|
|
2534
|
+
pattern: /\bSite24x7\b/i,
|
|
2535
|
+
class: "monitor",
|
|
2536
|
+
purpose: "monitor",
|
|
2537
|
+
operator: "Site24x7",
|
|
2538
|
+
product: "Site24x7"
|
|
2284
2539
|
},
|
|
2285
2540
|
{
|
|
2286
|
-
id: "
|
|
2287
|
-
pattern: /\
|
|
2288
|
-
class: "
|
|
2289
|
-
purpose: "
|
|
2290
|
-
operator:
|
|
2291
|
-
product: "
|
|
2541
|
+
id: "checkly",
|
|
2542
|
+
pattern: /\bCheckly\b/i,
|
|
2543
|
+
class: "monitor",
|
|
2544
|
+
purpose: "monitor",
|
|
2545
|
+
operator: "Checkly",
|
|
2546
|
+
product: "Checkly"
|
|
2292
2547
|
},
|
|
2293
2548
|
{
|
|
2294
|
-
id: "
|
|
2295
|
-
pattern: /\
|
|
2296
|
-
class: "
|
|
2297
|
-
purpose: "
|
|
2298
|
-
operator:
|
|
2299
|
-
product: "
|
|
2549
|
+
id: "newrelic",
|
|
2550
|
+
pattern: /\bNewRelicPinger\b|\bNew ?Relic ?Synthetics\b/i,
|
|
2551
|
+
class: "monitor",
|
|
2552
|
+
purpose: "monitor",
|
|
2553
|
+
operator: "New Relic",
|
|
2554
|
+
product: "New Relic Synthetics"
|
|
2300
2555
|
},
|
|
2301
2556
|
{
|
|
2302
|
-
id: "
|
|
2303
|
-
pattern: /\
|
|
2304
|
-
class: "
|
|
2305
|
-
purpose: "
|
|
2306
|
-
operator:
|
|
2307
|
-
product: "
|
|
2557
|
+
id: "freshping",
|
|
2558
|
+
pattern: /\bFreshping\b/i,
|
|
2559
|
+
class: "monitor",
|
|
2560
|
+
purpose: "monitor",
|
|
2561
|
+
operator: "Freshworks",
|
|
2562
|
+
product: "Freshping"
|
|
2308
2563
|
},
|
|
2309
2564
|
{
|
|
2310
|
-
id: "
|
|
2311
|
-
pattern: /
|
|
2312
|
-
class: "
|
|
2313
|
-
purpose: "
|
|
2314
|
-
operator:
|
|
2315
|
-
product: "
|
|
2316
|
-
}
|
|
2317
|
-
|
|
2318
|
-
|
|
2319
|
-
|
|
2320
|
-
|
|
2321
|
-
|
|
2322
|
-
|
|
2323
|
-
|
|
2324
|
-
|
|
2325
|
-
|
|
2326
|
-
|
|
2327
|
-
|
|
2328
|
-
|
|
2329
|
-
|
|
2330
|
-
|
|
2331
|
-
|
|
2332
|
-
|
|
2333
|
-
|
|
2334
|
-
|
|
2335
|
-
|
|
2336
|
-
|
|
2337
|
-
|
|
2338
|
-
|
|
2339
|
-
|
|
2340
|
-
|
|
2341
|
-
|
|
2342
|
-
|
|
2343
|
-
|
|
2344
|
-
|
|
2345
|
-
|
|
2346
|
-
|
|
2347
|
-
|
|
2348
|
-
|
|
2349
|
-
|
|
2350
|
-
|
|
2351
|
-
|
|
2352
|
-
|
|
2353
|
-
|
|
2354
|
-
|
|
2355
|
-
|
|
2356
|
-
|
|
2357
|
-
|
|
2358
|
-
|
|
2359
|
-
|
|
2360
|
-
|
|
2361
|
-
|
|
2362
|
-
|
|
2363
|
-
|
|
2364
|
-
|
|
2365
|
-
|
|
2366
|
-
|
|
2367
|
-
|
|
2368
|
-
|
|
2369
|
-
|
|
2370
|
-
|
|
2371
|
-
|
|
2372
|
-
|
|
2373
|
-
|
|
2374
|
-
|
|
2375
|
-
|
|
2376
|
-
|
|
2377
|
-
|
|
2378
|
-
|
|
2379
|
-
|
|
2380
|
-
|
|
2381
|
-
|
|
2382
|
-
|
|
2383
|
-
|
|
2384
|
-
|
|
2385
|
-
|
|
2386
|
-
|
|
2387
|
-
|
|
2388
|
-
|
|
2389
|
-
|
|
2390
|
-
|
|
2391
|
-
|
|
2392
|
-
|
|
2393
|
-
|
|
2394
|
-
|
|
2395
|
-
|
|
2396
|
-
|
|
2397
|
-
|
|
2398
|
-
|
|
2399
|
-
|
|
2400
|
-
|
|
2401
|
-
|
|
2402
|
-
|
|
2403
|
-
|
|
2404
|
-
|
|
2405
|
-
|
|
2406
|
-
|
|
2407
|
-
|
|
2408
|
-
|
|
2409
|
-
|
|
2410
|
-
|
|
2411
|
-
|
|
2412
|
-
|
|
2413
|
-
|
|
2414
|
-
|
|
2415
|
-
|
|
2416
|
-
|
|
2417
|
-
|
|
2418
|
-
|
|
2419
|
-
|
|
2420
|
-
|
|
2421
|
-
|
|
2422
|
-
|
|
2423
|
-
|
|
2424
|
-
|
|
2425
|
-
|
|
2426
|
-
|
|
2427
|
-
|
|
2428
|
-
|
|
2429
|
-
|
|
2430
|
-
|
|
2431
|
-
|
|
2432
|
-
|
|
2433
|
-
|
|
2434
|
-
|
|
2435
|
-
|
|
2436
|
-
|
|
2437
|
-
|
|
2438
|
-
"169.254.0.0/16",
|
|
2439
|
-
"172.16.0.0/12",
|
|
2440
|
-
"192.0.0.0/24",
|
|
2441
|
-
"192.0.2.0/24",
|
|
2442
|
-
"192.88.99.0/24",
|
|
2443
|
-
"192.168.0.0/16",
|
|
2444
|
-
"198.18.0.0/15",
|
|
2445
|
-
"198.51.100.0/24",
|
|
2446
|
-
"203.0.113.0/24",
|
|
2447
|
-
"224.0.0.0/4",
|
|
2448
|
-
"240.0.0.0/4"
|
|
2449
|
-
].map((c) => parseCidr(c));
|
|
2450
|
-
var NON_PUBLIC_V6 = [
|
|
2451
|
-
"::/128",
|
|
2452
|
-
"::1/128",
|
|
2453
|
-
"64:ff9b::/96",
|
|
2454
|
-
"64:ff9b:1::/48",
|
|
2455
|
-
"100::/64",
|
|
2456
|
-
"2001::/32",
|
|
2457
|
-
"2001:db8::/32",
|
|
2458
|
-
"2002::/16",
|
|
2459
|
-
"fc00::/7",
|
|
2460
|
-
"fe80::/10",
|
|
2461
|
-
"fec0::/10",
|
|
2462
|
-
"ff00::/8"
|
|
2463
|
-
].map((c) => parseCidr(c));
|
|
2464
|
-
|
|
2465
|
-
// ../classify/dist/signature.js
|
|
2466
|
-
var KNOWN_SIGNATURE_AGENTS = {
|
|
2467
|
-
"https://chatgpt.com": {
|
|
2468
|
-
operator: "OpenAI",
|
|
2469
|
-
product: "ChatGPT agent",
|
|
2470
|
-
class: "ai_agent",
|
|
2471
|
-
purpose: "ai_agent"
|
|
2565
|
+
id: "hetrixtools",
|
|
2566
|
+
pattern: /\bHetrixTools\b/i,
|
|
2567
|
+
class: "monitor",
|
|
2568
|
+
purpose: "monitor",
|
|
2569
|
+
operator: "HetrixTools",
|
|
2570
|
+
product: "HetrixTools"
|
|
2571
|
+
},
|
|
2572
|
+
{
|
|
2573
|
+
id: "updown",
|
|
2574
|
+
pattern: /\bupdown\.io\b/i,
|
|
2575
|
+
class: "monitor",
|
|
2576
|
+
purpose: "monitor",
|
|
2577
|
+
operator: "updown.io",
|
|
2578
|
+
product: "updown.io"
|
|
2579
|
+
},
|
|
2580
|
+
{
|
|
2581
|
+
id: "uptime-kuma",
|
|
2582
|
+
pattern: /\bUptime-Kuma\b/i,
|
|
2583
|
+
class: "monitor",
|
|
2584
|
+
purpose: "monitor",
|
|
2585
|
+
operator: null,
|
|
2586
|
+
product: "Uptime Kuma"
|
|
2587
|
+
},
|
|
2588
|
+
{
|
|
2589
|
+
id: "lighthouse",
|
|
2590
|
+
pattern: /\bChrome-Lighthouse\b/,
|
|
2591
|
+
class: "monitor",
|
|
2592
|
+
purpose: "monitor",
|
|
2593
|
+
operator: null,
|
|
2594
|
+
product: "Lighthouse"
|
|
2595
|
+
},
|
|
2596
|
+
// SEO crawlers: operated by a known company, but not visitors, search engines or AI.
|
|
2597
|
+
{
|
|
2598
|
+
id: "ahrefs",
|
|
2599
|
+
pattern: /\bAhrefs(?:Bot|SiteAudit)\b/,
|
|
2600
|
+
class: "automation",
|
|
2601
|
+
purpose: "seo",
|
|
2602
|
+
operator: "Ahrefs",
|
|
2603
|
+
product: "AhrefsBot"
|
|
2604
|
+
},
|
|
2605
|
+
{
|
|
2606
|
+
id: "semrush",
|
|
2607
|
+
pattern: /\bSemrushBot\b/i,
|
|
2608
|
+
class: "automation",
|
|
2609
|
+
purpose: "seo",
|
|
2610
|
+
operator: "Semrush",
|
|
2611
|
+
product: "SemrushBot"
|
|
2612
|
+
},
|
|
2613
|
+
{
|
|
2614
|
+
id: "mj12bot",
|
|
2615
|
+
pattern: /\bMJ12bot\b/i,
|
|
2616
|
+
class: "automation",
|
|
2617
|
+
purpose: "seo",
|
|
2618
|
+
operator: "Majestic",
|
|
2619
|
+
product: "MJ12bot"
|
|
2620
|
+
},
|
|
2621
|
+
{
|
|
2622
|
+
id: "dotbot",
|
|
2623
|
+
pattern: /\bDotBot\b|\brogerbot\b/i,
|
|
2624
|
+
class: "automation",
|
|
2625
|
+
purpose: "seo",
|
|
2626
|
+
operator: "Moz",
|
|
2627
|
+
product: "DotBot"
|
|
2628
|
+
},
|
|
2629
|
+
{
|
|
2630
|
+
id: "screaming-frog",
|
|
2631
|
+
pattern: /\bScreaming Frog SEO Spider\b/i,
|
|
2632
|
+
class: "automation",
|
|
2633
|
+
purpose: "seo",
|
|
2634
|
+
operator: "Screaming Frog",
|
|
2635
|
+
product: "Screaming Frog SEO Spider"
|
|
2636
|
+
},
|
|
2637
|
+
{
|
|
2638
|
+
id: "dataforseo",
|
|
2639
|
+
pattern: /\bDataForSeoBot\b/i,
|
|
2640
|
+
class: "automation",
|
|
2641
|
+
purpose: "seo",
|
|
2642
|
+
operator: "DataForSEO",
|
|
2643
|
+
product: "DataForSeoBot"
|
|
2644
|
+
},
|
|
2645
|
+
{
|
|
2646
|
+
id: "blexbot",
|
|
2647
|
+
pattern: /\bBLEXBot\b/i,
|
|
2648
|
+
class: "automation",
|
|
2649
|
+
purpose: "seo",
|
|
2650
|
+
operator: "WebMeUp",
|
|
2651
|
+
product: "BLEXBot"
|
|
2652
|
+
},
|
|
2653
|
+
{
|
|
2654
|
+
id: "serpstatbot",
|
|
2655
|
+
pattern: /\bserpstatbot\b/i,
|
|
2656
|
+
class: "automation",
|
|
2657
|
+
purpose: "seo",
|
|
2658
|
+
operator: "Serpstat",
|
|
2659
|
+
product: "serpstatbot"
|
|
2660
|
+
},
|
|
2661
|
+
{
|
|
2662
|
+
id: "barkrowler",
|
|
2663
|
+
pattern: /\bBarkrowler\b/i,
|
|
2664
|
+
class: "automation",
|
|
2665
|
+
purpose: "seo",
|
|
2666
|
+
operator: "Babbar",
|
|
2667
|
+
product: "Barkrowler"
|
|
2668
|
+
},
|
|
2669
|
+
// Automation tools and HTTP libraries: no operator is claimed.
|
|
2670
|
+
{
|
|
2671
|
+
id: "headlesschrome",
|
|
2672
|
+
pattern: /\bHeadlessChrome\b/,
|
|
2673
|
+
class: "automation",
|
|
2674
|
+
purpose: "scraper",
|
|
2675
|
+
operator: null,
|
|
2676
|
+
product: "HeadlessChrome"
|
|
2677
|
+
},
|
|
2678
|
+
{
|
|
2679
|
+
id: "phantomjs",
|
|
2680
|
+
pattern: /\bPhantomJS\b/i,
|
|
2681
|
+
class: "automation",
|
|
2682
|
+
purpose: "scraper",
|
|
2683
|
+
operator: null,
|
|
2684
|
+
product: "PhantomJS"
|
|
2685
|
+
},
|
|
2686
|
+
{
|
|
2687
|
+
id: "playwright",
|
|
2688
|
+
pattern: /\bPlaywright\b/i,
|
|
2689
|
+
class: "automation",
|
|
2690
|
+
purpose: "scraper",
|
|
2691
|
+
operator: null,
|
|
2692
|
+
product: "Playwright"
|
|
2472
2693
|
},
|
|
2473
|
-
"https://agent.bot.goog": {
|
|
2474
|
-
operator: "Google",
|
|
2475
|
-
product: "Google-Agent",
|
|
2476
|
-
class: "ai_agent",
|
|
2477
|
-
purpose: "ai_agent"
|
|
2478
|
-
}
|
|
2479
|
-
};
|
|
2480
|
-
|
|
2481
|
-
// ../classify/dist/version.js
|
|
2482
|
-
var RULES_VERSION = "2026-09-29.2";
|
|
2483
|
-
|
|
2484
|
-
// ../classify/dist/actor.js
|
|
2485
|
-
function verdict(cls, operator, product, evidence, reason, purpose, verification) {
|
|
2486
|
-
return {
|
|
2487
|
-
class: cls,
|
|
2488
|
-
operator,
|
|
2489
|
-
product,
|
|
2490
|
-
evidence,
|
|
2491
|
-
reason,
|
|
2492
|
-
rulesVersion: RULES_VERSION,
|
|
2493
|
-
purpose,
|
|
2494
|
-
verification
|
|
2495
|
-
};
|
|
2496
|
-
}
|
|
2497
|
-
function signatureNote(signature) {
|
|
2498
|
-
if (signature?.status === "invalid")
|
|
2499
|
-
return ` Its request signature failed verification: ${lower(signature.reason)}.`;
|
|
2500
|
-
return "";
|
|
2501
|
-
}
|
|
2502
|
-
function lower(s) {
|
|
2503
|
-
return s.length > 0 ? s[0]?.toLowerCase() + s.slice(1) : s;
|
|
2504
|
-
}
|
|
2505
|
-
function hasVerification(rule) {
|
|
2506
|
-
return (rule.verify?.ipRangeSources?.length ?? 0) + (rule.verify?.reverseDnsSuffixes?.length ?? 0) > 0;
|
|
2507
|
-
}
|
|
2508
|
-
function ruleVerdict(rule, network, signature) {
|
|
2509
|
-
const op = rule.operator;
|
|
2510
|
-
const badSignature = signature?.status === "invalid";
|
|
2511
|
-
if (network?.status === "verified" && op && !badSignature) {
|
|
2512
|
-
const how = network.method === "ip_ranges" ? `verified against ${op}'s published IP ranges` : "verified by reverse and forward DNS";
|
|
2513
|
-
return verdict(rule.class, op, rule.product, "network_verified", `${rule.product}, ${how}.`, rule.purpose, "verified");
|
|
2514
|
-
}
|
|
2515
|
-
if (!op) {
|
|
2516
|
-
return verdict(rule.class, null, rule.product, "none", `${rule.product} user agent.${signatureNote(signature)}`, rule.purpose, badSignature ? "suspected_spoof" : "no_claim");
|
|
2517
|
-
}
|
|
2518
|
-
if (network?.status === "mismatch") {
|
|
2519
|
-
const why = network.method === "ip_ranges" ? `the address is not in ${op}'s published ranges` : `the address does not verify with ${op}'s reverse DNS`;
|
|
2520
|
-
return verdict(rule.class, op, rule.product, "self_declared", `Claims to be ${rule.product}, but ${why}.${signatureNote(signature)}`, rule.purpose, "suspected_spoof");
|
|
2521
|
-
}
|
|
2522
|
-
const status = badSignature ? "suspected_spoof" : hasVerification(rule) ? "unchecked" : "unverifiable";
|
|
2523
|
-
return verdict(rule.class, op, rule.product, "self_declared", `Self-declared ${rule.product} (user agent only).${signatureNote(signature)}`, rule.purpose, status);
|
|
2524
|
-
}
|
|
2525
|
-
function classifyActor(input) {
|
|
2526
|
-
const ua = typeof input.ua === "string" ? input.ua.trim() : "";
|
|
2527
|
-
const rule = ua.length > 0 ? matchBotRule(ua) : null;
|
|
2528
|
-
const signature = input.signature;
|
|
2529
|
-
if (signature?.status === "verified") {
|
|
2530
|
-
const known = KNOWN_SIGNATURE_AGENTS[signature.agent];
|
|
2531
|
-
const host = new URL(signature.agent).host;
|
|
2532
|
-
if (known) {
|
|
2533
|
-
if (rule && rule.operator === known.operator) {
|
|
2534
|
-
return verdict(rule.class, known.operator, rule.product, "signature_verified", `${rule.product}, verified request signature from ${host}.`, rule.purpose, "verified");
|
|
2535
|
-
}
|
|
2536
|
-
const claim = rule ? ` Its user agent claims ${rule.product}.` : "";
|
|
2537
|
-
return verdict(known.class, known.operator, known.product, "signature_verified", `${known.product}, verified request signature from ${host}.${claim}`, known.purpose, "verified");
|
|
2538
|
-
}
|
|
2539
|
-
const cls = rule ? rule.class : isBrowserLike(ua) ? "browser" : "unknown";
|
|
2540
|
-
const purpose = rule ? rule.purpose : "unknown";
|
|
2541
|
-
if (rule?.operator) {
|
|
2542
|
-
return verdict(cls, host, null, "signature_verified", `Request signed by ${host}. Its user agent claims ${rule.product}.`, purpose, "verified");
|
|
2543
|
-
}
|
|
2544
|
-
return verdict(cls, host, rule?.product ?? null, "signature_verified", `Request signed by ${host}.`, purpose, "verified");
|
|
2545
|
-
}
|
|
2546
|
-
const spoofed = signature?.status === "invalid" ? "suspected_spoof" : "no_claim";
|
|
2547
|
-
if (ua.length === 0)
|
|
2548
|
-
return verdict("unknown", null, null, "none", `No user agent.${signatureNote(signature)}`, "unknown", spoofed);
|
|
2549
|
-
if (rule)
|
|
2550
|
-
return ruleVerdict(rule, input.network, signature);
|
|
2551
|
-
if (isBrowserLike(ua))
|
|
2552
|
-
return verdict("browser", null, null, "none", `Browser user agent.${signatureNote(signature)}`, "human", spoofed);
|
|
2553
|
-
return verdict("unknown", null, null, "none", `Unrecognized user agent.${signatureNote(signature)}`, "unknown", spoofed);
|
|
2554
|
-
}
|
|
2555
|
-
var RULE_BY_PRODUCT = /* @__PURE__ */ new Map();
|
|
2556
|
-
for (const rule of [...BOT_RULES, GENERIC_BOT_RULE]) {
|
|
2557
|
-
if (!RULE_BY_PRODUCT.has(rule.product))
|
|
2558
|
-
RULE_BY_PRODUCT.set(rule.product, rule);
|
|
2559
|
-
}
|
|
2560
|
-
|
|
2561
|
-
// ../contract/dist/actors.js
|
|
2562
|
-
var ACTOR_CLASS_LABELS = {
|
|
2563
|
-
browser: "Interactive browser",
|
|
2564
|
-
search_crawler: "Search crawler",
|
|
2565
|
-
ai_crawler: "AI crawler",
|
|
2566
|
-
ai_agent: "User-directed AI agent",
|
|
2567
|
-
monitor: "Uptime and monitoring",
|
|
2568
|
-
preview: "Link preview",
|
|
2569
|
-
automation: "Suspected automation",
|
|
2570
|
-
unknown: "Unknown"
|
|
2571
|
-
};
|
|
2572
|
-
|
|
2573
|
-
// ../contract/dist/channels.js
|
|
2574
|
-
var AI_ASSISTANT_HOSTS = [
|
|
2575
|
-
{ host: "chatgpt.com", label: "ChatGPT", operator: "OpenAI" },
|
|
2576
|
-
{ host: "chat.openai.com", label: "ChatGPT", operator: "OpenAI" },
|
|
2577
|
-
{ host: "perplexity.ai", label: "Perplexity", operator: "Perplexity" },
|
|
2578
|
-
{ host: "claude.ai", label: "Claude", operator: "Anthropic" },
|
|
2579
|
-
{ host: "gemini.google.com", label: "Gemini", operator: "Google" },
|
|
2580
|
-
{ host: "copilot.microsoft.com", label: "Microsoft Copilot", operator: "Microsoft" },
|
|
2581
|
-
{ host: "you.com", label: "You.com", operator: "You.com" },
|
|
2582
|
-
{ host: "phind.com", label: "Phind", operator: "Phind" }
|
|
2583
|
-
];
|
|
2584
|
-
|
|
2585
|
-
// ../contract/dist/limits.js
|
|
2586
|
-
var LIMITS = {
|
|
2587
|
-
/** Max request body accepted by the collector, in bytes (before decompression). */
|
|
2588
|
-
maxBodyBytes: 16 * 1024,
|
|
2589
|
-
/** Max events in one batch. */
|
|
2590
|
-
maxEventsPerBatch: 50,
|
|
2591
|
-
/** Max custom properties on one event. */
|
|
2592
|
-
maxProps: 8,
|
|
2593
|
-
maxPropKeyLength: 32,
|
|
2594
|
-
maxPropStringLength: 64,
|
|
2595
|
-
maxEventNameLength: 64,
|
|
2596
|
-
maxRouteLength: 256,
|
|
2597
|
-
maxHostLength: 253,
|
|
2598
|
-
maxCampaignValueLength: 64,
|
|
2599
|
-
/** Events observed further in the past than this are rejected as stale. */
|
|
2600
|
-
maxPastSkewMs: 24 * 60 * 60 * 1e3,
|
|
2601
|
-
/** Events observed further in the future than this are rejected as clock skew. */
|
|
2602
|
-
maxFutureSkewMs: 10 * 60 * 1e3,
|
|
2603
|
-
/** Browser SDK flush triggers. */
|
|
2604
|
-
browserFlushEvents: 10,
|
|
2605
|
-
browserFlushMs: 5e3,
|
|
2606
|
-
/** Browser SDK bounded pending queue, in bytes of serialized events. */
|
|
2607
|
-
browserMaxPendingBytes: 64 * 1024,
|
|
2608
|
-
/** Journey sessions: inactivity timeout and hard lifetime. */
|
|
2609
|
-
sessionIdleMs: 30 * 60 * 1e3,
|
|
2610
|
-
sessionMaxMs: 24 * 60 * 60 * 1e3
|
|
2611
|
-
};
|
|
2612
|
-
|
|
2613
|
-
// ../contract/dist/flows.js
|
|
2614
|
-
var FLOW_START_DEFINITIONS = [
|
|
2615
2694
|
{
|
|
2616
|
-
|
|
2617
|
-
|
|
2618
|
-
|
|
2695
|
+
id: "puppeteer",
|
|
2696
|
+
pattern: /\bPuppeteer\b/i,
|
|
2697
|
+
class: "automation",
|
|
2698
|
+
purpose: "scraper",
|
|
2699
|
+
operator: null,
|
|
2700
|
+
product: "Puppeteer"
|
|
2619
2701
|
},
|
|
2620
2702
|
{
|
|
2621
|
-
|
|
2622
|
-
|
|
2623
|
-
|
|
2703
|
+
id: "selenium",
|
|
2704
|
+
pattern: /\bSelenium\b/i,
|
|
2705
|
+
class: "automation",
|
|
2706
|
+
purpose: "scraper",
|
|
2707
|
+
operator: null,
|
|
2708
|
+
product: "Selenium"
|
|
2624
2709
|
},
|
|
2625
2710
|
{
|
|
2626
|
-
|
|
2627
|
-
|
|
2628
|
-
|
|
2711
|
+
id: "curl",
|
|
2712
|
+
pattern: /(?:^|[\s(])curl\/\d/i,
|
|
2713
|
+
class: "automation",
|
|
2714
|
+
purpose: "scraper",
|
|
2715
|
+
operator: null,
|
|
2716
|
+
product: "curl"
|
|
2629
2717
|
},
|
|
2630
|
-
{ key: "flow_went_on", label: "Went on", definition: "Of these sessions, the share with a second step." },
|
|
2631
2718
|
{
|
|
2632
|
-
|
|
2633
|
-
|
|
2634
|
-
|
|
2635
|
-
|
|
2636
|
-
|
|
2637
|
-
|
|
2638
|
-
|
|
2719
|
+
id: "wget",
|
|
2720
|
+
pattern: /\bWget\/\d/i,
|
|
2721
|
+
class: "automation",
|
|
2722
|
+
purpose: "scraper",
|
|
2723
|
+
operator: null,
|
|
2724
|
+
product: "Wget"
|
|
2725
|
+
},
|
|
2639
2726
|
{
|
|
2640
|
-
|
|
2641
|
-
|
|
2642
|
-
|
|
2727
|
+
id: "python-requests",
|
|
2728
|
+
pattern: /\bpython-requests\/\d/i,
|
|
2729
|
+
class: "automation",
|
|
2730
|
+
purpose: "scraper",
|
|
2731
|
+
operator: null,
|
|
2732
|
+
product: "python-requests"
|
|
2643
2733
|
},
|
|
2644
|
-
{ key: "flow_left", label: "Left", definition: "The session had no more steps and has ended." },
|
|
2645
2734
|
{
|
|
2646
|
-
|
|
2647
|
-
|
|
2648
|
-
|
|
2735
|
+
id: "python-httpx",
|
|
2736
|
+
pattern: /\bpython-httpx\/\d/i,
|
|
2737
|
+
class: "automation",
|
|
2738
|
+
purpose: "scraper",
|
|
2739
|
+
operator: null,
|
|
2740
|
+
product: "httpx"
|
|
2649
2741
|
},
|
|
2650
2742
|
{
|
|
2651
|
-
|
|
2652
|
-
|
|
2653
|
-
|
|
2743
|
+
id: "aiohttp",
|
|
2744
|
+
pattern: /\baiohttp\/\d/i,
|
|
2745
|
+
class: "automation",
|
|
2746
|
+
purpose: "scraper",
|
|
2747
|
+
operator: null,
|
|
2748
|
+
product: "aiohttp"
|
|
2654
2749
|
},
|
|
2655
2750
|
{
|
|
2656
|
-
|
|
2657
|
-
|
|
2658
|
-
|
|
2751
|
+
id: "python-urllib",
|
|
2752
|
+
pattern: /\bPython-urllib\/\d/i,
|
|
2753
|
+
class: "automation",
|
|
2754
|
+
purpose: "scraper",
|
|
2755
|
+
operator: null,
|
|
2756
|
+
product: "urllib"
|
|
2757
|
+
},
|
|
2758
|
+
{
|
|
2759
|
+
id: "scrapy",
|
|
2760
|
+
pattern: /\bScrapy\/\d/i,
|
|
2761
|
+
class: "automation",
|
|
2762
|
+
purpose: "scraper",
|
|
2763
|
+
operator: null,
|
|
2764
|
+
product: "Scrapy"
|
|
2765
|
+
},
|
|
2766
|
+
{
|
|
2767
|
+
id: "go-http-client",
|
|
2768
|
+
pattern: /\bGo-http-client\/\d/,
|
|
2769
|
+
class: "automation",
|
|
2770
|
+
purpose: "scraper",
|
|
2771
|
+
operator: null,
|
|
2772
|
+
product: "Go-http-client"
|
|
2773
|
+
},
|
|
2774
|
+
{
|
|
2775
|
+
id: "axios",
|
|
2776
|
+
pattern: /\baxios\/\d/i,
|
|
2777
|
+
class: "automation",
|
|
2778
|
+
purpose: "scraper",
|
|
2779
|
+
operator: null,
|
|
2780
|
+
product: "axios"
|
|
2781
|
+
},
|
|
2782
|
+
{
|
|
2783
|
+
id: "node-fetch",
|
|
2784
|
+
pattern: /\bnode-fetch\b/i,
|
|
2785
|
+
class: "automation",
|
|
2786
|
+
purpose: "scraper",
|
|
2787
|
+
operator: null,
|
|
2788
|
+
product: "node-fetch"
|
|
2789
|
+
},
|
|
2790
|
+
{
|
|
2791
|
+
id: "undici",
|
|
2792
|
+
pattern: /^(?:undici|node)$/i,
|
|
2793
|
+
class: "automation",
|
|
2794
|
+
purpose: "scraper",
|
|
2795
|
+
operator: null,
|
|
2796
|
+
product: "undici"
|
|
2797
|
+
},
|
|
2798
|
+
{
|
|
2799
|
+
id: "okhttp",
|
|
2800
|
+
pattern: /\bokhttp\/\d/i,
|
|
2801
|
+
class: "automation",
|
|
2802
|
+
purpose: "scraper",
|
|
2803
|
+
operator: null,
|
|
2804
|
+
product: "OkHttp"
|
|
2805
|
+
},
|
|
2806
|
+
{
|
|
2807
|
+
id: "apache-httpclient",
|
|
2808
|
+
pattern: /\bApache-HttpClient\/\d/,
|
|
2809
|
+
class: "automation",
|
|
2810
|
+
purpose: "scraper",
|
|
2811
|
+
operator: null,
|
|
2812
|
+
product: "Apache-HttpClient"
|
|
2813
|
+
},
|
|
2814
|
+
{
|
|
2815
|
+
id: "java",
|
|
2816
|
+
pattern: /(?:^|\s)Java\/\d/,
|
|
2817
|
+
class: "automation",
|
|
2818
|
+
purpose: "scraper",
|
|
2819
|
+
operator: null,
|
|
2820
|
+
product: "Java"
|
|
2821
|
+
},
|
|
2822
|
+
{
|
|
2823
|
+
id: "libwww-perl",
|
|
2824
|
+
pattern: /\blibwww-perl\/\d/i,
|
|
2825
|
+
class: "automation",
|
|
2826
|
+
purpose: "scraper",
|
|
2827
|
+
operator: null,
|
|
2828
|
+
product: "libwww-perl"
|
|
2829
|
+
},
|
|
2830
|
+
{
|
|
2831
|
+
id: "guzzle",
|
|
2832
|
+
pattern: /\bGuzzleHttp\/\d/i,
|
|
2833
|
+
class: "automation",
|
|
2834
|
+
purpose: "scraper",
|
|
2835
|
+
operator: null,
|
|
2836
|
+
product: "Guzzle"
|
|
2837
|
+
},
|
|
2838
|
+
{
|
|
2839
|
+
id: "httpie",
|
|
2840
|
+
pattern: /\bHTTPie\/\d/i,
|
|
2841
|
+
class: "automation",
|
|
2842
|
+
purpose: "scraper",
|
|
2843
|
+
operator: null,
|
|
2844
|
+
product: "HTTPie"
|
|
2845
|
+
},
|
|
2846
|
+
{
|
|
2847
|
+
id: "postman",
|
|
2848
|
+
pattern: /\bPostmanRuntime\/\d/,
|
|
2849
|
+
class: "automation",
|
|
2850
|
+
purpose: "scraper",
|
|
2851
|
+
operator: null,
|
|
2852
|
+
product: "PostmanRuntime"
|
|
2853
|
+
},
|
|
2854
|
+
{
|
|
2855
|
+
id: "insomnia",
|
|
2856
|
+
pattern: /\binsomnia\/\d/i,
|
|
2857
|
+
class: "automation",
|
|
2858
|
+
purpose: "scraper",
|
|
2859
|
+
operator: null,
|
|
2860
|
+
product: "Insomnia"
|
|
2861
|
+
},
|
|
2862
|
+
{
|
|
2863
|
+
id: "dart",
|
|
2864
|
+
pattern: /(?:^|\s)Dart\/\d/,
|
|
2865
|
+
class: "automation",
|
|
2866
|
+
purpose: "scraper",
|
|
2867
|
+
operator: null,
|
|
2868
|
+
product: "Dart"
|
|
2659
2869
|
}
|
|
2660
2870
|
];
|
|
2871
|
+
var GENERIC_BOT_RULE = {
|
|
2872
|
+
id: "generic-bot",
|
|
2873
|
+
pattern: /(?:bot|crawler|spider|scraper)\b/i,
|
|
2874
|
+
class: "automation",
|
|
2875
|
+
purpose: "unknown",
|
|
2876
|
+
operator: null,
|
|
2877
|
+
product: "Unidentified bot"
|
|
2878
|
+
};
|
|
2879
|
+
var MAX_UA_LENGTH2 = 512;
|
|
2880
|
+
function matchBotRule(input) {
|
|
2881
|
+
if (typeof input !== "string")
|
|
2882
|
+
return null;
|
|
2883
|
+
const ua = input.slice(0, MAX_UA_LENGTH2);
|
|
2884
|
+
if (ua.trim().length === 0)
|
|
2885
|
+
return null;
|
|
2886
|
+
for (const rule of BOT_RULES) {
|
|
2887
|
+
if (rule.pattern.test(ua))
|
|
2888
|
+
return rule;
|
|
2889
|
+
}
|
|
2890
|
+
if (GENERIC_BOT_RULE.pattern.test(ua) && (!isBrowserLike(ua) || /\+https?:|compatible;|@/i.test(ua))) {
|
|
2891
|
+
return GENERIC_BOT_RULE;
|
|
2892
|
+
}
|
|
2893
|
+
return null;
|
|
2894
|
+
}
|
|
2661
2895
|
|
|
2662
|
-
// ../
|
|
2663
|
-
|
|
2664
|
-
|
|
2665
|
-
|
|
2666
|
-
|
|
2667
|
-
|
|
2668
|
-
|
|
2669
|
-
|
|
2670
|
-
|
|
2671
|
-
|
|
2672
|
-
|
|
2896
|
+
// ../classify/dist/ip.js
|
|
2897
|
+
import { isIPv4, isIPv6 } from "node:net";
|
|
2898
|
+
function parseV4(ip) {
|
|
2899
|
+
if (!isIPv4(ip))
|
|
2900
|
+
return null;
|
|
2901
|
+
let n = 0n;
|
|
2902
|
+
for (const part of ip.split("."))
|
|
2903
|
+
n = n << 8n | BigInt(Number(part));
|
|
2904
|
+
return n;
|
|
2905
|
+
}
|
|
2906
|
+
function parseV6(input) {
|
|
2907
|
+
let ip = input;
|
|
2908
|
+
const zone = ip.indexOf("%");
|
|
2909
|
+
if (zone >= 0)
|
|
2910
|
+
ip = ip.slice(0, zone);
|
|
2911
|
+
if (!isIPv6(ip))
|
|
2912
|
+
return null;
|
|
2913
|
+
let tail = null;
|
|
2914
|
+
const lastColon = ip.lastIndexOf(":");
|
|
2915
|
+
const lastPart = ip.slice(lastColon + 1);
|
|
2916
|
+
if (lastPart.includes(".")) {
|
|
2917
|
+
tail = parseV4(lastPart);
|
|
2918
|
+
if (tail === null)
|
|
2919
|
+
return null;
|
|
2920
|
+
ip = `${ip.slice(0, lastColon + 1)}0:0`;
|
|
2673
2921
|
}
|
|
2674
|
-
|
|
2675
|
-
|
|
2676
|
-
|
|
2677
|
-
|
|
2678
|
-
|
|
2679
|
-
|
|
2680
|
-
|
|
2681
|
-
|
|
2682
|
-
|
|
2683
|
-
|
|
2684
|
-
|
|
2685
|
-
|
|
2922
|
+
const halves = ip.split("::");
|
|
2923
|
+
if (halves.length > 2)
|
|
2924
|
+
return null;
|
|
2925
|
+
const head = halves[0] ? halves[0].split(":") : [];
|
|
2926
|
+
const rest = halves.length === 2 && halves[1] ? halves[1].split(":") : [];
|
|
2927
|
+
const missing = 8 - head.length - rest.length;
|
|
2928
|
+
if (halves.length === 1 && missing !== 0)
|
|
2929
|
+
return null;
|
|
2930
|
+
if (missing < 0)
|
|
2931
|
+
return null;
|
|
2932
|
+
const groups = [...head, ...Array(halves.length === 2 ? missing : 0).fill("0"), ...rest];
|
|
2933
|
+
if (groups.length !== 8)
|
|
2934
|
+
return null;
|
|
2935
|
+
let n = 0n;
|
|
2936
|
+
for (const g of groups) {
|
|
2937
|
+
if (!/^[0-9a-f]{1,4}$/i.test(g))
|
|
2938
|
+
return null;
|
|
2939
|
+
n = n << 16n | BigInt(Number.parseInt(g, 16));
|
|
2686
2940
|
}
|
|
2941
|
+
if (tail !== null)
|
|
2942
|
+
n = n & ~0xffffffffn | tail;
|
|
2943
|
+
return n;
|
|
2687
2944
|
}
|
|
2688
|
-
var
|
|
2689
|
-
function
|
|
2690
|
-
if (
|
|
2691
|
-
return
|
|
2692
|
-
const
|
|
2693
|
-
|
|
2694
|
-
|
|
2695
|
-
|
|
2696
|
-
if (
|
|
2697
|
-
return
|
|
2698
|
-
|
|
2699
|
-
|
|
2700
|
-
|
|
2701
|
-
|
|
2702
|
-
|
|
2703
|
-
|
|
2704
|
-
if (match.evidenceBelow && !(evidenceRank(facts.evidence) < evidenceRank(match.evidenceBelow)))
|
|
2705
|
-
return false;
|
|
2706
|
-
if (match.principalBelow && !(principalRank(facts.principal) < principalRank(match.principalBelow)))
|
|
2707
|
-
return false;
|
|
2708
|
-
if (match.spoofed !== void 0 && match.spoofed !== facts.spoofed)
|
|
2709
|
-
return false;
|
|
2710
|
-
if (match.paths && !match.paths.some((p) => pathMatches(facts.path, p)))
|
|
2711
|
-
return false;
|
|
2712
|
-
if (match.methods && !match.methods.map(lower2).includes(lower2(facts.method)))
|
|
2713
|
-
return false;
|
|
2714
|
-
return true;
|
|
2945
|
+
var MAPPED_PREFIX = 0xffffn << 32n;
|
|
2946
|
+
function parseIp(input) {
|
|
2947
|
+
if (typeof input !== "string")
|
|
2948
|
+
return null;
|
|
2949
|
+
const ip = input.trim();
|
|
2950
|
+
if (ip.length === 0 || ip.length > 64)
|
|
2951
|
+
return null;
|
|
2952
|
+
const v4 = parseV4(ip);
|
|
2953
|
+
if (v4 !== null)
|
|
2954
|
+
return { v: 4, n: v4 };
|
|
2955
|
+
const v6 = parseV6(ip);
|
|
2956
|
+
if (v6 === null)
|
|
2957
|
+
return null;
|
|
2958
|
+
if (v6 >> 32n === MAPPED_PREFIX >> 32n)
|
|
2959
|
+
return { v: 4, n: v6 & 0xffffffffn };
|
|
2960
|
+
return { v: 6, n: v6 };
|
|
2715
2961
|
}
|
|
2716
|
-
function
|
|
2717
|
-
if (
|
|
2718
|
-
return
|
|
2719
|
-
|
|
2720
|
-
|
|
2721
|
-
|
|
2722
|
-
|
|
2962
|
+
function parseCidr(input) {
|
|
2963
|
+
if (typeof input !== "string")
|
|
2964
|
+
return null;
|
|
2965
|
+
const [addr, len, extra] = input.trim().split("/");
|
|
2966
|
+
if (addr === void 0 || extra !== void 0)
|
|
2967
|
+
return null;
|
|
2968
|
+
const p = parseIp(addr);
|
|
2969
|
+
if (!p)
|
|
2970
|
+
return null;
|
|
2971
|
+
const width = p.v === 4 ? 32 : 128;
|
|
2972
|
+
const mappedFromV6 = p.v === 4 && addr.includes(":");
|
|
2973
|
+
let bits = width;
|
|
2974
|
+
if (len !== void 0) {
|
|
2975
|
+
if (!/^\d{1,3}$/.test(len))
|
|
2976
|
+
return null;
|
|
2977
|
+
bits = Number(len);
|
|
2978
|
+
if (mappedFromV6)
|
|
2979
|
+
bits -= 96;
|
|
2980
|
+
if (bits < 0 || bits > width)
|
|
2981
|
+
return null;
|
|
2723
2982
|
}
|
|
2724
|
-
|
|
2983
|
+
const mask = bits === 0 ? 0n : (1n << BigInt(bits)) - 1n << BigInt(width - bits);
|
|
2984
|
+
return { v: p.v, base: p.n & mask, bits };
|
|
2725
2985
|
}
|
|
2726
|
-
var
|
|
2727
|
-
"
|
|
2728
|
-
"
|
|
2729
|
-
"
|
|
2730
|
-
"
|
|
2731
|
-
|
|
2732
|
-
|
|
2733
|
-
"
|
|
2734
|
-
"
|
|
2735
|
-
"
|
|
2736
|
-
"
|
|
2737
|
-
"
|
|
2738
|
-
"
|
|
2739
|
-
"
|
|
2740
|
-
|
|
2741
|
-
|
|
2742
|
-
|
|
2743
|
-
|
|
2744
|
-
|
|
2745
|
-
|
|
2746
|
-
|
|
2747
|
-
|
|
2748
|
-
|
|
2749
|
-
|
|
2750
|
-
|
|
2751
|
-
|
|
2752
|
-
|
|
2753
|
-
|
|
2754
|
-
|
|
2755
|
-
|
|
2756
|
-
|
|
2757
|
-
verified_only: [
|
|
2758
|
-
BLOCK_SPOOFED,
|
|
2759
|
-
{
|
|
2760
|
-
id: "unverified",
|
|
2761
|
-
label: "Slow down agents that are not verified",
|
|
2762
|
-
match: { classes: AGENT_CLASSES, evidenceBelow: "network_verified" },
|
|
2763
|
-
action: "limit",
|
|
2764
|
-
limit: { perMinute: 60, burst: 20 }
|
|
2765
|
-
}
|
|
2766
|
-
],
|
|
2767
|
-
commerce: [
|
|
2768
|
-
BLOCK_SPOOFED,
|
|
2769
|
-
{
|
|
2770
|
-
id: "checkout",
|
|
2771
|
-
label: "Checkout and accounts need an agent acting for a signed-in user",
|
|
2772
|
-
match: {
|
|
2773
|
-
// Not `unknown`: a person whose browser does not look like one must still reach checkout.
|
|
2774
|
-
classes: ["ai_agent", "automation"],
|
|
2775
|
-
paths: ["/checkout", "/cart", "/account"],
|
|
2776
|
-
principalBelow: "signed_user"
|
|
2777
|
-
},
|
|
2778
|
-
action: "require_signature"
|
|
2779
|
-
}
|
|
2780
|
-
]
|
|
2781
|
-
};
|
|
2782
|
-
|
|
2783
|
-
// ../contract/dist/origins.js
|
|
2784
|
-
var REFUSED_ORIGIN_SCHEMES = [
|
|
2785
|
-
// Script and inline content.
|
|
2786
|
-
"javascript",
|
|
2787
|
-
"vbscript",
|
|
2788
|
-
"data",
|
|
2789
|
-
"blob",
|
|
2790
|
-
"filesystem",
|
|
2791
|
-
// Files and blank pages, which send `Origin: null`.
|
|
2792
|
-
"file",
|
|
2793
|
-
"about",
|
|
2794
|
-
"content",
|
|
2795
|
-
// Browser pages and extensions.
|
|
2796
|
-
"chrome",
|
|
2797
|
-
"chrome-extension",
|
|
2798
|
-
"chrome-untrusted",
|
|
2799
|
-
"chrome-search",
|
|
2800
|
-
"devtools",
|
|
2801
|
-
"edge",
|
|
2802
|
-
"brave",
|
|
2803
|
-
"opera",
|
|
2804
|
-
"vivaldi",
|
|
2805
|
-
"view-source",
|
|
2806
|
-
"resource",
|
|
2807
|
-
"moz-extension",
|
|
2808
|
-
"safari-extension",
|
|
2809
|
-
"safari-web-extension",
|
|
2810
|
-
"ms-browser-extension",
|
|
2811
|
-
// Protocols that are never a page.
|
|
2812
|
-
"ftp",
|
|
2813
|
-
"ws",
|
|
2814
|
-
"wss",
|
|
2815
|
-
"mailto",
|
|
2816
|
-
"tel",
|
|
2817
|
-
"sms",
|
|
2818
|
-
"intent",
|
|
2819
|
-
"market"
|
|
2820
|
-
];
|
|
2821
|
-
var REFUSED = new Set(REFUSED_ORIGIN_SCHEMES);
|
|
2822
|
-
|
|
2823
|
-
// ../contract/dist/referrers.js
|
|
2824
|
-
var ANDROID_APP_HOSTS = Object.freeze({
|
|
2825
|
-
"com.google.android.gm": "mail.google.com",
|
|
2826
|
-
"com.microsoft.office.outlook": "outlook.live.com",
|
|
2827
|
-
"com.yahoo.mobile.client.android.mail": "mail.yahoo.com",
|
|
2828
|
-
"ch.protonmail.android": "mail.proton.me",
|
|
2829
|
-
"com.google.android.googlequicksearchbox": "google.com",
|
|
2830
|
-
"com.linkedin.android": "linkedin.com",
|
|
2831
|
-
"com.facebook.katana": "facebook.com",
|
|
2832
|
-
"com.facebook.orca": "facebook.com",
|
|
2833
|
-
"com.facebook.lite": "facebook.com",
|
|
2834
|
-
"com.instagram.android": "instagram.com",
|
|
2835
|
-
"com.instagram.barcelona": "threads.net",
|
|
2836
|
-
"com.twitter.android": "x.com",
|
|
2837
|
-
"com.reddit.frontpage": "reddit.com",
|
|
2838
|
-
"com.google.android.youtube": "youtube.com",
|
|
2839
|
-
"com.zhiliaoapp.musically": "tiktok.com",
|
|
2840
|
-
"xyz.blueskyweb.app": "bsky.app",
|
|
2841
|
-
"com.pinterest": "pinterest.com",
|
|
2842
|
-
"com.slack": "slack.com",
|
|
2843
|
-
"com.discord": "discord.com",
|
|
2844
|
-
"org.telegram.messenger": "t.me",
|
|
2845
|
-
"com.whatsapp": "whatsapp.com",
|
|
2846
|
-
"com.openai.chatgpt": "chatgpt.com",
|
|
2847
|
-
"ai.perplexity.app.android": "perplexity.ai",
|
|
2848
|
-
"com.anthropic.claude": "claude.ai",
|
|
2849
|
-
"com.google.android.apps.bard": "gemini.google.com",
|
|
2850
|
-
"com.microsoft.copilot": "copilot.microsoft.com"
|
|
2851
|
-
});
|
|
2986
|
+
var NON_PUBLIC_V4 = [
|
|
2987
|
+
"0.0.0.0/8",
|
|
2988
|
+
"10.0.0.0/8",
|
|
2989
|
+
"100.64.0.0/10",
|
|
2990
|
+
"127.0.0.0/8",
|
|
2991
|
+
"169.254.0.0/16",
|
|
2992
|
+
"172.16.0.0/12",
|
|
2993
|
+
"192.0.0.0/24",
|
|
2994
|
+
"192.0.2.0/24",
|
|
2995
|
+
"192.88.99.0/24",
|
|
2996
|
+
"192.168.0.0/16",
|
|
2997
|
+
"198.18.0.0/15",
|
|
2998
|
+
"198.51.100.0/24",
|
|
2999
|
+
"203.0.113.0/24",
|
|
3000
|
+
"224.0.0.0/4",
|
|
3001
|
+
"240.0.0.0/4"
|
|
3002
|
+
].map((c) => parseCidr(c));
|
|
3003
|
+
var NON_PUBLIC_V6 = [
|
|
3004
|
+
"::/128",
|
|
3005
|
+
"::1/128",
|
|
3006
|
+
"64:ff9b::/96",
|
|
3007
|
+
"64:ff9b:1::/48",
|
|
3008
|
+
"100::/64",
|
|
3009
|
+
"2001::/32",
|
|
3010
|
+
"2001:db8::/32",
|
|
3011
|
+
"2002::/16",
|
|
3012
|
+
"fc00::/7",
|
|
3013
|
+
"fe80::/10",
|
|
3014
|
+
"fec0::/10",
|
|
3015
|
+
"ff00::/8"
|
|
3016
|
+
].map((c) => parseCidr(c));
|
|
2852
3017
|
|
|
2853
|
-
// ../
|
|
2854
|
-
var
|
|
2855
|
-
|
|
2856
|
-
|
|
2857
|
-
|
|
2858
|
-
|
|
2859
|
-
|
|
2860
|
-
|
|
2861
|
-
"
|
|
2862
|
-
|
|
2863
|
-
|
|
2864
|
-
|
|
2865
|
-
|
|
2866
|
-
|
|
2867
|
-
"class",
|
|
2868
|
-
"id",
|
|
2869
|
-
"style",
|
|
2870
|
-
"role",
|
|
2871
|
-
"type",
|
|
2872
|
-
"name",
|
|
2873
|
-
"for",
|
|
2874
|
-
"rel",
|
|
2875
|
-
"media",
|
|
2876
|
-
"href",
|
|
2877
|
-
"width",
|
|
2878
|
-
"height",
|
|
2879
|
-
"colspan",
|
|
2880
|
-
"rowspan",
|
|
2881
|
-
"dir",
|
|
2882
|
-
"lang",
|
|
2883
|
-
"hidden",
|
|
2884
|
-
"open",
|
|
2885
|
-
"disabled",
|
|
2886
|
-
"readonly",
|
|
2887
|
-
"multiple",
|
|
2888
|
-
"viewbox",
|
|
2889
|
-
"d",
|
|
2890
|
-
"fill",
|
|
2891
|
-
"stroke",
|
|
2892
|
-
"stroke-width",
|
|
2893
|
-
"points",
|
|
2894
|
-
"x",
|
|
2895
|
-
"y",
|
|
2896
|
-
"cx",
|
|
2897
|
-
"cy",
|
|
2898
|
-
"r",
|
|
2899
|
-
"rx",
|
|
2900
|
-
"ry",
|
|
2901
|
-
"transform",
|
|
2902
|
-
"aria-hidden",
|
|
2903
|
-
"aria-expanded",
|
|
2904
|
-
"aria-current",
|
|
2905
|
-
"aria-disabled",
|
|
2906
|
-
"aria-pressed"
|
|
2907
|
-
];
|
|
2908
|
-
var REPLAY_TEXT_ATTRIBUTES = [
|
|
2909
|
-
"title",
|
|
2910
|
-
"alt",
|
|
2911
|
-
"placeholder",
|
|
2912
|
-
"aria-label",
|
|
2913
|
-
"aria-description"
|
|
2914
|
-
];
|
|
2915
|
-
var REPLAY_UNMASK_PRESET = [
|
|
2916
|
-
"nav",
|
|
2917
|
-
"header",
|
|
2918
|
-
"footer",
|
|
2919
|
-
"h1",
|
|
2920
|
-
"h2",
|
|
2921
|
-
"h3",
|
|
2922
|
-
"h4",
|
|
2923
|
-
"button",
|
|
2924
|
-
"label",
|
|
2925
|
-
"th",
|
|
2926
|
-
"legend",
|
|
2927
|
-
"[role=button]",
|
|
2928
|
-
"[role=tab]",
|
|
2929
|
-
"[role=menuitem]"
|
|
2930
|
-
];
|
|
2931
|
-
var REPLAY_LIMITS = {
|
|
2932
|
-
/** One POST body, compressed. Keepalive requests on page hide must stay under 64 KB. */
|
|
2933
|
-
maxChunkBytes: 256 * 1024,
|
|
2934
|
-
/** One chunk after gunzip. Anything larger is refused before it is parsed. */
|
|
2935
|
-
maxChunkRawBytes: 4 * 1024 * 1024,
|
|
2936
|
-
/**
|
|
2937
|
-
* A chunk that carries a full snapshot may be larger: a heavy page's first snapshot with its
|
|
2938
|
-
* inlined stylesheets can pass maxChunkBytes on its own. Every other chunk keeps the normal limits.
|
|
2939
|
-
*/
|
|
2940
|
-
maxSnapshotChunkBytes: 1024 * 1024,
|
|
2941
|
-
maxSnapshotChunkRawBytes: 10 * 1024 * 1024,
|
|
2942
|
-
maxChunkEvents: 5e3,
|
|
2943
|
-
/** Stored bytes per session, compressed. Past it the collector answers 202 and drops. */
|
|
2944
|
-
maxSessionBytes: 12 * 1024 * 1024,
|
|
2945
|
-
/** Recorded time per page load. The recorder stops after it. */
|
|
2946
|
-
maxPageMs: 60 * 60 * 1e3,
|
|
2947
|
-
/** The recorder flushes a chunk this often, or sooner when it reaches flushBytes raw. */
|
|
2948
|
-
flushMs: 1e4,
|
|
2949
|
-
flushBytes: 128 * 1024,
|
|
2950
|
-
/** Two interactions closer than this are one stretch of activity. Longer gaps are skippable. */
|
|
2951
|
-
idleGapMs: 5e3,
|
|
2952
|
-
/** A rage click: this many clicks within rageWindowMs inside rageRadiusPx. */
|
|
2953
|
-
rageClicks: 3,
|
|
2954
|
-
rageWindowMs: 1e3,
|
|
2955
|
-
rageRadiusPx: 30,
|
|
2956
|
-
maxSelectors: 50,
|
|
2957
|
-
maxSelectorLength: 200,
|
|
2958
|
-
maxExcludeRoutes: 50,
|
|
2959
|
-
/**
|
|
2960
|
-
* A recording with at least this many clicks is kept when its session ends, whatever its
|
|
2961
|
-
* active time: a short visit of a few taps on a phone is still worth watching.
|
|
2962
|
-
*/
|
|
2963
|
-
keepWithClicks: 3,
|
|
2964
|
-
/**
|
|
2965
|
-
* Recordings a workspace may start per UTC calendar month without a card on file. The
|
|
2966
|
-
* collector and the API read REPLAY_FREE_RECORDINGS_PER_MONTH first, so both agree.
|
|
2967
|
-
*/
|
|
2968
|
-
freeRecordingsPerMonth: 10
|
|
3018
|
+
// ../classify/dist/signature.js
|
|
3019
|
+
var KNOWN_SIGNATURE_AGENTS = {
|
|
3020
|
+
"https://chatgpt.com": {
|
|
3021
|
+
operator: "OpenAI",
|
|
3022
|
+
product: "ChatGPT agent",
|
|
3023
|
+
class: "ai_agent",
|
|
3024
|
+
purpose: "ai_agent"
|
|
3025
|
+
},
|
|
3026
|
+
"https://agent.bot.goog": {
|
|
3027
|
+
operator: "Google",
|
|
3028
|
+
product: "Google-Agent",
|
|
3029
|
+
class: "ai_agent",
|
|
3030
|
+
purpose: "ai_agent"
|
|
3031
|
+
}
|
|
2969
3032
|
};
|
|
2970
3033
|
|
|
2971
|
-
// ../
|
|
2972
|
-
var
|
|
2973
|
-
Mutation: 0,
|
|
2974
|
-
MouseMove: 1,
|
|
2975
|
-
MouseInteraction: 2,
|
|
2976
|
-
Scroll: 3,
|
|
2977
|
-
ViewportResize: 4,
|
|
2978
|
-
Input: 5,
|
|
2979
|
-
TouchMove: 6,
|
|
2980
|
-
MediaInteraction: 7,
|
|
2981
|
-
StyleSheetRule: 8,
|
|
2982
|
-
CanvasMutation: 9,
|
|
2983
|
-
Font: 10,
|
|
2984
|
-
Log: 11,
|
|
2985
|
-
Drag: 12,
|
|
2986
|
-
StyleDeclaration: 13,
|
|
2987
|
-
Selection: 14,
|
|
2988
|
-
AdoptedStyleSheet: 15,
|
|
2989
|
-
CustomElement: 16
|
|
2990
|
-
};
|
|
2991
|
-
var REPLAY_MAX_SCANNED_TEXT = 32 * 1024;
|
|
2992
|
-
var MASKED_RE = new RegExp(`^[\\s${REPLAY_MASK_CHAR}]*$`, "u");
|
|
2993
|
-
var BLOCKED_TAGS = /* @__PURE__ */ new Set([
|
|
2994
|
-
...REPLAY_ALWAYS_BLOCKED.map((s) => s.split(/\s+/).pop()),
|
|
2995
|
-
"frame",
|
|
2996
|
-
"frameset",
|
|
2997
|
-
"applet",
|
|
2998
|
-
"portal",
|
|
2999
|
-
"fencedframe"
|
|
3000
|
-
]);
|
|
3001
|
-
var KEPT = new Set(REPLAY_KEPT_ATTRIBUTES);
|
|
3002
|
-
var TEXT_ATTRS = new Set(REPLAY_TEXT_ATTRIBUTES);
|
|
3003
|
-
var INTERACTION_SOURCES = /* @__PURE__ */ new Set([
|
|
3004
|
-
REPLAY_SOURCE.MouseMove,
|
|
3005
|
-
REPLAY_SOURCE.MouseInteraction,
|
|
3006
|
-
REPLAY_SOURCE.Scroll,
|
|
3007
|
-
REPLAY_SOURCE.ViewportResize,
|
|
3008
|
-
REPLAY_SOURCE.Input,
|
|
3009
|
-
REPLAY_SOURCE.TouchMove,
|
|
3010
|
-
REPLAY_SOURCE.Drag,
|
|
3011
|
-
REPLAY_SOURCE.Selection
|
|
3012
|
-
]);
|
|
3034
|
+
// ../classify/dist/version.js
|
|
3035
|
+
var RULES_VERSION = "2026-09-29.2";
|
|
3013
3036
|
|
|
3014
|
-
// ../
|
|
3015
|
-
|
|
3037
|
+
// ../classify/dist/actor.js
|
|
3038
|
+
function verdict(cls, operator, product, evidence, reason, purpose, verification) {
|
|
3039
|
+
return {
|
|
3040
|
+
class: cls,
|
|
3041
|
+
operator,
|
|
3042
|
+
product,
|
|
3043
|
+
evidence,
|
|
3044
|
+
reason,
|
|
3045
|
+
rulesVersion: RULES_VERSION,
|
|
3046
|
+
purpose,
|
|
3047
|
+
verification
|
|
3048
|
+
};
|
|
3049
|
+
}
|
|
3050
|
+
function signatureNote(signature) {
|
|
3051
|
+
if (signature?.status === "invalid")
|
|
3052
|
+
return ` Its request signature failed verification: ${lower2(signature.reason)}.`;
|
|
3053
|
+
return "";
|
|
3054
|
+
}
|
|
3055
|
+
function lower2(s) {
|
|
3056
|
+
return s.length > 0 ? s[0]?.toLowerCase() + s.slice(1) : s;
|
|
3057
|
+
}
|
|
3058
|
+
function hasVerification(rule) {
|
|
3059
|
+
return (rule.verify?.ipRangeSources?.length ?? 0) + (rule.verify?.reverseDnsSuffixes?.length ?? 0) > 0;
|
|
3060
|
+
}
|
|
3061
|
+
function ruleVerdict(rule, network, signature) {
|
|
3062
|
+
const op = rule.operator;
|
|
3063
|
+
const badSignature = signature?.status === "invalid";
|
|
3064
|
+
if (network?.status === "verified" && op && !badSignature) {
|
|
3065
|
+
const how = network.method === "ip_ranges" ? `verified against ${op}'s published IP ranges` : "verified by reverse and forward DNS";
|
|
3066
|
+
return verdict(rule.class, op, rule.product, "network_verified", `${rule.product}, ${how}.`, rule.purpose, "verified");
|
|
3067
|
+
}
|
|
3068
|
+
if (!op) {
|
|
3069
|
+
return verdict(rule.class, null, rule.product, "none", `${rule.product} user agent.${signatureNote(signature)}`, rule.purpose, badSignature ? "suspected_spoof" : "no_claim");
|
|
3070
|
+
}
|
|
3071
|
+
if (network?.status === "mismatch") {
|
|
3072
|
+
const why = network.method === "ip_ranges" ? `the address is not in ${op}'s published ranges` : `the address does not verify with ${op}'s reverse DNS`;
|
|
3073
|
+
return verdict(rule.class, op, rule.product, "self_declared", `Claims to be ${rule.product}, but ${why}.${signatureNote(signature)}`, rule.purpose, "suspected_spoof");
|
|
3074
|
+
}
|
|
3075
|
+
const status = badSignature ? "suspected_spoof" : hasVerification(rule) ? "unchecked" : "unverifiable";
|
|
3076
|
+
return verdict(rule.class, op, rule.product, "self_declared", `Self-declared ${rule.product} (user agent only).${signatureNote(signature)}`, rule.purpose, status);
|
|
3077
|
+
}
|
|
3078
|
+
function classifyActor(input) {
|
|
3079
|
+
const ua = typeof input.ua === "string" ? input.ua.trim() : "";
|
|
3080
|
+
const rule = ua.length > 0 ? matchBotRule(ua) : null;
|
|
3081
|
+
const signature = input.signature;
|
|
3082
|
+
if (signature?.status === "verified") {
|
|
3083
|
+
const known = KNOWN_SIGNATURE_AGENTS[signature.agent];
|
|
3084
|
+
const host = new URL(signature.agent).host;
|
|
3085
|
+
if (known) {
|
|
3086
|
+
if (rule && rule.operator === known.operator) {
|
|
3087
|
+
return verdict(rule.class, known.operator, rule.product, "signature_verified", `${rule.product}, verified request signature from ${host}.`, rule.purpose, "verified");
|
|
3088
|
+
}
|
|
3089
|
+
const claim = rule ? ` Its user agent claims ${rule.product}.` : "";
|
|
3090
|
+
return verdict(known.class, known.operator, known.product, "signature_verified", `${known.product}, verified request signature from ${host}.${claim}`, known.purpose, "verified");
|
|
3091
|
+
}
|
|
3092
|
+
const cls = rule ? rule.class : isBrowserLike(ua) ? "browser" : "unknown";
|
|
3093
|
+
const purpose = rule ? rule.purpose : "unknown";
|
|
3094
|
+
if (rule?.operator) {
|
|
3095
|
+
return verdict(cls, host, null, "signature_verified", `Request signed by ${host}. Its user agent claims ${rule.product}.`, purpose, "verified");
|
|
3096
|
+
}
|
|
3097
|
+
return verdict(cls, host, rule?.product ?? null, "signature_verified", `Request signed by ${host}.`, purpose, "verified");
|
|
3098
|
+
}
|
|
3099
|
+
const spoofed = signature?.status === "invalid" ? "suspected_spoof" : "no_claim";
|
|
3100
|
+
if (ua.length === 0)
|
|
3101
|
+
return verdict("unknown", null, null, "none", `No user agent.${signatureNote(signature)}`, "unknown", spoofed);
|
|
3102
|
+
if (rule)
|
|
3103
|
+
return ruleVerdict(rule, input.network, signature);
|
|
3104
|
+
if (isBrowserLike(ua))
|
|
3105
|
+
return verdict("browser", null, null, "none", `Browser user agent.${signatureNote(signature)}`, "human", spoofed);
|
|
3106
|
+
return verdict("unknown", null, null, "none", `Unrecognized user agent.${signatureNote(signature)}`, "unknown", spoofed);
|
|
3107
|
+
}
|
|
3108
|
+
var RULE_BY_PRODUCT = /* @__PURE__ */ new Map();
|
|
3109
|
+
for (const rule of [...BOT_RULES, GENERIC_BOT_RULE]) {
|
|
3110
|
+
if (!RULE_BY_PRODUCT.has(rule.product))
|
|
3111
|
+
RULE_BY_PRODUCT.set(rule.product, rule);
|
|
3112
|
+
}
|
|
3016
3113
|
|
|
3017
3114
|
// ../classify/dist/channel.js
|
|
3018
3115
|
var AI_HOSTS = AI_ASSISTANT_HOSTS.map((a) => a.host);
|
|
@@ -3703,7 +3800,7 @@ This copy of the CLI was built without the setup guide.
|
|
|
3703
3800
|
Read it at ${DOCS_URL}
|
|
3704
3801
|
`;
|
|
3705
3802
|
function bundledGuide() {
|
|
3706
|
-
return true ? { text: "# Set up Little Friend in a product: guide for coding agents\n\nYou are a coding agent working in a product's repository. This guide takes you from nothing to a verified Little Friend install: the project, the script, events, goals, funnels, session replay, server events and AI agent visibility. It ends with a report for the person who asked.\n\nWork through the steps in order. Each decision has a default: use it unless the repo or the person who asked says otherwise. When a step does not apply, skip it and say why in the report.\n\n## Contents\n\n0. [Conventions](#0-conventions)\n1. [What Little Friend is, and the rules](#1-what-little-friend-is-and-the-rules)\n2. [Sign in once per machine](#2-sign-in-once-per-machine)\n3. [Read the repo and plan the journeys](#3-read-the-repo-and-plan-the-journeys)\n4. [Workspace, project, mode and consent](#4-workspace-project-mode-and-consent)\n5. [Install the script](#5-install-the-script)\n6. [Content Security Policy](#6-content-security-policy)\n7. [Custom events](#7-custom-events)\n8. [Goals](#8-goals)\n9. [Funnels](#9-funnels)\n10. [Identify signed-in people](#10-identify-signed-in-people)\n11. [Session replay](#11-session-replay)\n12. [Server events](#12-server-events)\n13. [Crawlers and AI agents: the door and a log drain](#13-crawlers-and-ai-agents-the-door-and-a-log-drain)\n14. [Hybrid and native apps](#14-hybrid-and-native-apps)\n15. [Verify](#15-verify)\n16. [Launch checklist](#16-launch-checklist)\n17. [Report back](#17-report-back)\n\n## 0. Conventions\n\n- **The CLI.** Run it as `npx --yes @littlefriend/cli <command>`. It needs Node 22 or later. This guide writes `littlefriend <command>` for short: type the full `npx` form, unless the CLI is installed globally.\n- **Scope flags.** Commands that work on a project take `--project <id>`, and commands that work in a workspace take `--workspace <id or name>`. Agent shells often drop environment variables between calls, so pass the flags on every command. The examples below leave them out to stay short. The CLI also reads `LITTLEFRIEND_PROJECT` and `LITTLEFRIEND_WORKSPACE`.\n- **Reading output.** Add `--json` when you need a value from the output, such as an id. Errors go to stderr with a non-zero exit code. With `--json`, they are also printed on stdout as `{ \"error\": { \"code\", \"message\" } }`.\n- **Secrets.** Server keys (`lfs_`), edge keys (`lfe_`), read keys (`lfr_`) and drain secrets are shown once. Never put them in your messages, logs, commits or the report. Write keys to an env file with `--write-env` (step 12). The public site key (`lf_`) is fine to commit.\n- **Credentials.** `~/.littlefriend/credentials.json` holds the sign-in. Never read it, print it or copy it into a repo.\n- **Deletes.** Never delete or revoke a project, key, goal, funnel or drain you did not create in this session.\n- **Repo rules win.** Follow the product repo's own rules for branches, commits, dependencies and deploys.\n- **The code wins.** `littlefriend <command> --help` shows a command's flags. If this guide and the CLI disagree, trust the CLI and say so in the report.\n\n## 1. What Little Friend is, and the rules\n\nLittle Friend is privacy-first web analytics (littlefriend.io). A small script, `lf.js`, counts page views, sources, clicks and named events without cookies. Server events count outcomes your backend confirms. Log drains and `@littlefriend/edge` show crawlers and AI agents, and the door decides which agents get in. Reports live at app.littlefriend.io.\n\nThe product must keep these rules. If a task would break one, stop and ask.\n\n1. **No personal data** in event names, property keys or values, routes, ids, or goal and funnel names. Personal data means emails, names, usernames, phone numbers, addresses, account or card numbers, text a person typed, and anything else that points to one person.\n2. **`identify` takes an opaque id** from the product's own database, such as `usr_8f3k2`. Never an email, a phone number, a name, or a hash of any of them.\n3. **No typed text.** Never send form values, search queries or error messages as properties. Report a failed form with a short category, such as `validation` or `server`.\n4. **Secret keys stay on the server.** Never put `lfs_`, `lfe_` or `lfr_` keys in client code, or in a variable a bundler exposes to the browser (`NEXT_PUBLIC_`, `VITE_`, `PUBLIC_`, `NUXT_PUBLIC_`, `EXPO_PUBLIC_`).\n5. **No extra tracking.** Do not add cookies, storage or fingerprinting for analytics. Do not work around Global Privacy Control or a visitor's opt-out: the script already honors both.\n6. **Unknown stays unknown.** Never combine an IP address with device traits to guess who someone is.\n\nLittle Friend also strips query strings and fragments (except allowlisted `utm_*` values), turns id-like and personal-looking path segments into `:id` or `:redacted`, drops property values that look like an email or a long number, and never reads form values or page text. Treat that as a safety net. Rule 1 still applies.\n\n## 2. Sign in once per machine\n\n```sh\nlittlefriend whoami\n```\n\nIf it names an account, go to step 3. Otherwise:\n\n```sh\nlittlefriend login\n```\n\nIt opens a browser, prints the URL too, and waits up to 5 minutes. The person who owns the Little Friend account approves the login in the browser. Tell them a sign-in page is waiting, then wait. If nobody approves in time, stop and ask. Do not retry in a loop.\n\n- The CLI acts with that person's role in each workspace. Creating projects, keys, goals, funnels and settings needs the editor or owner role.\n- In CI, set `LITTLEFRIEND_TOKEN` to an access token instead of signing in.\n- `littlefriend logout` revokes the sign-in and deletes the local credentials.\n\n## 3. Read the repo and plan the journeys\n\nCollect these facts before you create anything. They drive every later decision, and most of them go into the report.\n\n| Fact | Where to look |\n|---|---|\n| Product name and production domain | README, deploy config (`vercel.json`, `fly.toml`, `netlify.toml`, `wrangler.toml`), site URL env vars, canonical tags, sitemap config |\n| Every host the product serves pages from | `www.`, `app.`, docs subdomains, redirect config |\n| Web framework | `package.json` and config files. The table in step 5 maps them |\n| Where the HTML `<head>` is rendered | Root layout, `index.html`, document template |\n| Sign-in | Auth libraries (Auth.js, Clerk, Supabase, Lucia, Better Auth, Passport), `/login` and `/signup` routes, a current-user hook |\n| Money | Checkout, pricing, plans, trials, payment webhooks |\n| Onboarding and activation | First-run screens, \"create your first ...\" flows, invites |\n| Pages with private data | Account, settings, billing, admin, messages, documents, and any health, legal or financial records |\n| Consent tool | A cookie banner or consent manager (OneTrust, Cookiebot, Osano, Klaro, a custom one) |\n| Content Security Policy | Search code, config and headers files for `Content-Security-Policy` |\n| Server runtime and host | API routes, server framework (Express, Celsian, Hono, Next.js route handlers), Workers, Fly or Vercel config |\n| Env files | `.env`, `.env.local`, `.env.example`, and what `.gitignore` ignores |\n| App shells | `capacitor.config.*`, `ionic.config.json`, Cordova `config.xml`, `src-tauri/`, an Electron main process, React Native or Expo |\n| Existing analytics | Google Analytics, Plausible, PostHog, Segment or Vercel Analytics calls |\n\nExisting analytics: leave them in place unless asked to remove them. Their event lists are good candidates for your named events and goals.\n\n### Plan the journeys\n\nBefore you add a single event, decide what this product exists to get people to do, and which paths lead there. Steps 7 to 9 build exactly this plan, so every event, goal and funnel has a reason.\n\n1. **Say the product's job in one sentence:** who arrives, and what a good visit ends with. Read the home page, the pricing page, the main call to action, and the signup or checkout code.\n2. **List the outcomes, most valuable first.** Usually 1 to 3: money (purchase, upgrade, trial), an account (signup), a lead (contact, demo booked), or activation, the first time someone gets value (created a project, sent an invite, published a page).\n3. **Write one journey per top outcome, at most 3.** A journey is the path a person takes from arriving to the outcome, in 3 to 5 steps they would recognize: where it really starts (a landing page, pricing, a docs page), the moment they commit (opened signup, started checkout), and the outcome itself.\n4. **Decide how each step is seen.** A page is a route step (`route:/pricing`). An action inside a page, or something the server confirms, is a named event you add in step 7 (`event:signup.start`). The outcome becomes a goal in step 8, and the journey becomes a saved funnel in step 9.\n5. **Check the routes.** Read the router for each route step. Ids and personal segments are stored as `:id` or `:redacted`, so `/projects/42/settings` is stored as `/projects/:id/settings`.\n\nStart from the row that fits, then change it to match what the code really does:\n\n| Product | Outcome | Journey |\n|---|---|---|\n| App or SaaS | Signup, then activation | `route:/` \u2192 `route:/pricing` \u2192 `event:signup.start` \u2192 `goal:<Signup id>` \u2192 `goal:<Activated id>` |\n| Store | Purchase | `route:/products/:id` \u2192 `event:cart.add` \u2192 `event:checkout.start` \u2192 `goal:<Purchase id>` |\n| Services or lead generation | Contact or booking | `route:/` \u2192 `route:/services` \u2192 `event:contact.open` \u2192 `goal:<Contact sent id>` |\n| Docs or open source | Adoption | `route:/docs` \u2192 `route:/docs/getting-started` \u2192 `goal:<Install copied id>`, a goal on `install.copy` |\n| Marketing site for an app on another host | A visitor heads to sign up | `route:/` \u2192 `route:/pricing` \u2192 `goal:<Signup click id>`, a goal on `cta.signup` |\n\n**A journey ends at the edge of its host.** A journey-mode session lives in one tab's session storage, which browsers keep apart for each host. A visit that moves from `example.com` to `app.example.com` becomes two sessions. Give each host its own funnel: end the marketing site's funnel on a goal for a named event fired when the signup link is clicked (`lf('track', 'cta.signup')`), and start the app's funnel at its signup page.\n\nKeep the plan small and useful:\n\n- Pick what someone running the product would ask about on a Monday: \"Of the people who saw pricing, how many started a trial?\" If nobody would act on a number, do not build it.\n- At most 3 funnels and 5 goals per project. Every funnel ends on a goal.\n- A step the code cannot see yet (no route, no event) needs an event in step 7. Note it in the plan.\n- A server-confirmed outcome joins the journey only when the server event carries the browser's correlation id (step 12). Plan that, or end the funnel on the browser event just before it.\n- Aggregate mode has no sessions, so it has no funnels. Plan goals only, and say so in the report.\n\nWrite the plan into the report (step 17) as a table before you build anything:\n\n| Journey | Outcome goal | Steps, and how each is seen |\n|---|---|---|\n| Signup | Signup (server) | `/pricing` route, `signup.start` event on the button, `signup.completed` server event with the browser's correlation id |\n\n## 4. Workspace, project, mode and consent\n\n### Workspace\n\n```sh\nlittlefriend workspaces list\n```\n\n| Situation | Workspace |\n|---|---|\n| The person who asked named one | That one |\n| The list shows one workspace | That one |\n| Several, and nobody named one | Stop and ask. Never guess |\n| The one you need is missing | Stop and ask. The CLI does not create workspaces |\n\n### Reuse before you create\n\n```sh\nlittlefriend projects list --workspace <workspace>\n```\n\nIf a project already has the product's domain, reuse it: note its id and site key, and run `littlefriend projects get` to see its mode and allowed origins. Never create a second project for the same domain.\n\nOne product is one project, even when its marketing pages and app live on different hosts of the same domain. List every host in allowed origins.\n\n### Mode\n\n| The product has | Mode |\n|---|---|\n| Sign-in, signup, onboarding, checkout, or any flow of several steps | `journey` |\n| Only marketing or content pages, and no sign-in | `aggregate` |\n| It is a site you build for a client | `aggregate`, unless the client asked for journeys |\n\n- Aggregate mode keeps counts only, with no session or visitor key anywhere. It still gives views, sources, pages, devices, countries, events and goals.\n- Journey mode adds sessions, timelines, funnels, paths, entry and exit pages, session replay and identify. Its session id lives in the tab's session storage and ends after 30 idle minutes (24 hours at most).\n\n### Consent\n\n| Situation | Do this |\n|---|---|\n| The product already runs a consent tool | Use `--consent required` in step 5. Call `lf('consent', 'journey')` when the tool grants analytics consent, and `lf('consent', 'aggregate')` when it is withdrawn |\n| No consent tool, journey mode | Ask the person who asked whether to ship without a banner. Do not add a consent tool yourself |\n| Aggregate mode | Nothing to wire |\n\nUntil consent arrives, the script counts in aggregate mode. A browser that sends Global Privacy Control always stays in aggregate mode.\n\nIf the product has a privacy page, add this text for journey mode:\n\n> We use Little Friend for privacy-friendly analytics. It sets no cookies and does not store IP addresses. To connect the pages you view in one visit, it keeps a random id in your browser tab's session storage. The visit ends after 30 minutes of inactivity, and the id is gone when you close the tab. If your browser sends Global Privacy Control, your visit is only counted, never connected.\n\nIn aggregate mode, use only the first two sentences. With session replay on (step 11), add:\n\n> Some visits are recorded as a masked copy of the pages you view. Text stays hidden unless we chose to show it, form entries are never recorded, and recordings are deleted within 7 days.\n\nIf there is no privacy page, say so in the report.\n\n### Create the project\n\n```sh\nlittlefriend projects create --name \"Acme\" --domain acme.com --mode journey \\\n --origin https://acme.com --origin https://www.acme.com --origin http://localhost:3000 \\\n --timezone America/New_York\n```\n\nIt prints the project id and the public site key. Keep both for the rest of the guide and the report.\n\n- `--domain`: the production host. Little Friend stores it lowercase, without a scheme, path or `www.`.\n- `--origin`, once per origin: every production host that serves pages, plus the local dev server's origin so you can verify before deploy (step 15). Add app origins from step 14. Preview deploys and any origin not listed get `403` and count nowhere, which is intended. An empty list accepts every origin.\n- `--timezone`: the time zone the business reports in. Without it, reports use UTC.\n- To change origins later, run `littlefriend projects update` and pass the complete list: every origin already there plus the new ones.\n\n## 5. Install the script\n\nPrint the exact code for the stack:\n\n```sh\nlittlefriend snippet --framework next --mode journey\n```\n\nIt prints what to add, with the site key filled in, and the CSP lines the site needs. Add `--consent required` when step 4 said so. Add `--replay` only in step 11.\n\n| The repo has | `--framework` | Where the code goes |\n|---|---|---|\n| `next` | `next` | The root layout: `app/layout.tsx` (App Router) or `pages/_document.tsx` (Pages Router) |\n| `nuxt` | `nuxt` | `nuxt.config.ts` |\n| `@sveltejs/kit` | `sveltekit` | `src/app.html` |\n| `astro` | `astro` | The layout every page shares, often `src/layouts/Layout.astro` |\n| `@remix-run/*`, or React Router with `app/root.tsx` | `remix` | `app/root.tsx` |\n| `react` with Vite or another bundler, no framework | `react` | `index.html` |\n| `vue` with Vite | `vue` | `index.html` |\n| `@angular/core` | `angular` | `src/index.html` |\n| A WordPress theme or plugin | `wordpress` | Where the snippet says |\n| `what-framework`, a Shopify theme, plain HTML, anything else | `html` | The `<head>` every page shares: `index.html`, the document template, `layout/theme.liquid` |\n| A bundled app where you want typed functions | `npm` | The client entry, once at startup |\n\nThe snippet is the source of truth for the code. The last column is where to look first.\n\nRules:\n\n- **One tracker per page.** Use the tag or the npm package (`@littlefriend/tracker`), never both. Load it on every page, once.\n- **Single-page apps need nothing extra.** History API navigations count as page views. If the router does not use the History API, call `lf('page')` after each navigation. Call it for navigations only, never for a screen the load or a History API navigation already recorded.\n- **Outside production, mark traffic as test.** Add `data-test` to the tag, or pass `test: true` to `init`, when the build is not production (for example `process.env.NODE_ENV !== 'production'` or `import.meta.env.DEV`). Test traffic shows in the live install check and never in reports.\n- **Name dynamic routes.** Id-like segments already become `:id`. For readable groups, such as `/blog/:slug`, add `data-routes='[\"/blog/:slug\"]'` to the tag, pass `routes` to `init`, or call `lf('route', '/blog/:slug')`. Goals and funnel steps match these routes exactly.\n- **Calls before the script loads.** If the product's own code calls `lf(...)`, add this stub above the tag, so `lf` exists before the deferred script runs. Calls wait in a queue until it loads.\n\n```html\n<script>window.lf=window.lf||function(){(lf.q=lf.q||[]).push(arguments)}</script>\n```\n\n- **Secret scanners.** The site key (`lf_`) is public, but gitleaks and similar scanners may flag it as an API key. Add an allowlist entry for that exact key rather than hiding it. A constant named `LITTLE_FRIEND_SITE` rather than `..._KEY` trips fewer rules.\n- **Monorepo build caches.** If the tag only renders in production (for example on `VERCEL_ENV`), declare that variable for the build task in `turbo.json` or the cache's equivalent. Otherwise a preview build and a production build can share one cached output.\n\nWith the tag in a TypeScript app, declare the global once, for example in `src/lf.d.ts`:\n\n```ts\ndeclare global {\n function lf(command: string, ...args: unknown[]): void;\n}\nexport {};\n```\n\n## 6. Content Security Policy\n\nSkip this step if the product sets no CSP. Otherwise add:\n\n```text\nscript-src https://cdn.littlefriend.io\nconnect-src https://in.littlefriend.io\n```\n\n- With the npm package, the tracker is part of your bundle, so only `connect-src` is needed.\n- The replay script comes from the same host as `lf.js` and sends to the same collector. It needs no other entries.\n- A policy with a nonce or `'strict-dynamic'` ignores host entries for scripts. Give the tag the page's nonce, or use the npm package.\n- Update every copy of the policy: `Content-Security-Policy-Report-Only`, `<meta http-equiv>` tags, and the test files that assert headers.\n- Common places: `next.config.*` `headers()`, `middleware.ts` or `proxy.ts`, `vercel.json`, `netlify.toml`, `_headers`, Helmet options, server header code.\n\n## 7. Custom events\n\nThe script sends these on its own: page views, outbound links, downloads, engaged time and, with `data-scroll`, scroll depth. The `$` prefix is reserved for them.\n\nAdd the named events your journey plan (step 3) needs, and the few the product's key feature needs. Three ways to add more:\n\n```html\n<!-- A named click: sends $click with { \"id\": \"pricing.start_trial\", \"plan\": \"pro\" } -->\n<button data-lf=\"pricing.start_trial\" data-lf-plan=\"pro\">Start free trial</button>\n\n<!-- A tracked form: sends $form_start and $form_submit with { \"id\": \"signup\" } -->\n<form data-lf-form=\"signup\">...</form>\n```\n\n```ts\n// A named event, after the thing really happened\nlf('track', 'signup.completed', { plan: 'pro' });\n\n// A failed form: a short category, never the message\nlf('formError', 'signup', 'validation');\n```\n\nWith npm, import the same names: `track`, `formError`, `page`, `route`, `consent`, `optout`, `optin`.\n\n**Goals and funnel steps match an event name or a route, never a property.** Every `data-lf` click arrives as `$click`, and every tracked form as `$form_submit`. For anything that becomes a goal or a funnel step, fire its own named event with `track`.\n\n### Names\n\n- `area.action`, lowercase: dots between parts, underscores inside a part. `signup.start`, `signup.completed`, `onboarding.project_created`, `checkout.start`, `order.completed`, `plan.upgraded`, `invite.sent`.\n- Past tense for outcomes (`completed`, `created`, `sent`). `start` or `open` for intent.\n- The pattern is `^[a-z][a-z0-9_.:-]{0,63}$`. `data-lf` ids follow the same convention.\n- Fire outcome events after the server call succeeds, never on the button press. Fire intent events on the press.\n- Aim for 5 to 15 named events: the steps of signup, onboarding, activation and checkout, and the product's key feature. Do not name every click.\n\n| Good | Bad | Why the bad one fails |\n|---|---|---|\n| `signup.completed` | `Signup Completed` | Uppercase and spaces are refused |\n| `checkout.start` | `click_button_3` | Says nothing about the product |\n| `invite.sent` with `{ role: 'editor' }` | `invite.sent.jane@acme.com` | Personal data in the name |\n| `order.completed` with `{ plan: 'pro' }` | `order_8812_completed` | An id in the name makes a new event per order |\n| `search.submit` with `{ results: 12 }` | `search.submit` with `{ query: 'knee pain' }` | Typed text is personal data |\n| `project.created` | `$project_created` | `$` is reserved |\n\n### Properties\n\n- Up to 8 per event. Keys are lowercase snake case, up to 32 characters. Values are strings, numbers or booleans. Strings are cut to 64 characters.\n- Use small fixed sets (`plan`, `step`, `method`, `role`, `source`) and counts.\n- A value that looks like an email or a long run of digits is dropped, even when sent on purpose.\n\n### Page types (optional)\n\nFlows group pages by type. Little Friend suggests types from your paths, and owners and editors can change the rules in Settings, Page types. When a page's type can't be told from its path, set it on the page:\n\n```html\n<meta name=\"lf:type\" content=\"comparison\">\n```\n\nThe value is a lowercase key: letters, digits, `-` or `_`, starting with a letter, at most 32 characters. A tag beats the URL rules for that page. The tracker reads this one tag and nothing else.\n\nOn a client-side navigation the page view is recorded when the URL changes. Its type is read after the next animation frame, or at once when the tab is hidden, so head managers that write the tag on render are read correctly. On page load the tag is read when the script runs, so put it in the HTML the server sends.\n\n## 8. Goals\n\nA goal is a conversion. Create 2 to 5, one for each outcome in your journey plan (step 3), most valuable first.\n\n| Product | Goals to start with |\n|---|---|\n| App or SaaS with sign-in | Signup (server), activation: the first key action, upgrade or trial started (server) |\n| Store | Purchase (server, with value), checkout started |\n| Lead generation | Contact form sent, demo booked, the thank-you page |\n| Content or docs | Newsletter signup, a named event on the link to the product's signup |\n\n- Prefer server-confirmed goals for money and accounts: `--source server --server-confirmed`. `--source server` counts only events your server sends with a secret key, and `--server-confirmed` also marks the goal that way in reports, so a reader knows a browser cannot fake it. Use browser goals for intent, or when there is no backend.\n- A route goal matches the stored route exactly. Check the routes with `littlefriend report pages` first.\n- Create goals before launch. A goal counts from the moment it is saved.\n- Name goals in plain words: \"Signup\", \"Purchase\", \"Demo booked\".\n\n```sh\nlittlefriend goals create --name \"Signup\" --event signup.completed --source server --server-confirmed\nlittlefriend goals create --name \"Purchase\" --event order.completed --source server --server-confirmed\nlittlefriend goals create --name \"Activated\" --event project.created\nlittlefriend goals create --name \"Contact sent\" --route /contact/thanks\nlittlefriend goals list\n```\n\nKeep the goal ids (`goal_...`) for funnels and the report.\n\n## 9. Funnels\n\nJourney mode only. Save one funnel for each journey in your plan (step 3), 1 to 3 in all, named after the journey.\n\nA funnel has 1 to 8 steps in order. Each step is `route:/path`, `event:<name>` or `goal:<goalId>`. A session reaches a step when it has done the steps before it, in that order.\n\n| Flow | Steps |\n|---|---|\n| Signup | `route:/pricing` \u2192 `event:signup.start` \u2192 `goal:<Signup id>` |\n| Activation | `goal:<Signup id>` \u2192 `event:onboarding.profile_completed` \u2192 `goal:<Activated id>` |\n| Checkout | `route:/pricing` \u2192 `event:checkout.start` \u2192 `goal:<Purchase id>` |\n\n- Keep 3 to 5 steps. Start broad and end on a goal.\n- A server event joins a session only when it carries that session's correlation id or session id (step 12). Without one, a server-only goal as the last step is never reached. Pass a correlation id, or end on a browser event.\n\n```sh\nlittlefriend funnels create --name \"Signup\" \\\n --step route:/pricing --step event:signup.start --step goal:goal_XXXXXXXX\nlittlefriend funnels list\nlittlefriend funnels report fnl_XXXXXXXX --from 2026-10-01 --to 2026-10-07\n```\n\nDates are `YYYY-MM-DD` in the project's time zone.\n\n## 10. Identify signed-in people\n\nJourney mode only. `lf('identify', ...)` is handled by the replay script, so it needs step 11. Without replay, skip this step.\n\n```ts\n// Wherever the app knows the signed-in user: after sign-in and on each page load\nlf('identify', user.id); // your own opaque id, such as usr_8f3k2\n\n// On sign-out\nlf('identify', null);\n```\n\nWith npm: `import { identify } from '@littlefriend/replay'`, then `identify(user.id)` and `identify(null)`.\n\n- The ref is 1 to 64 letters, digits, `_`, `.`, `:` and `-`. UUIDs and prefixed ids work.\n- Prefix numeric ids, such as `usr_4821937`. A ref of seven or more bare digits is refused.\n- Never an email, a phone number, a name, a username, or a hash of any of them. Emails are refused.\n- The ref lasts for the journey session in that tab. Call it on each page load while signed in, so the next session carries it too.\n- Before consent it waits in memory, and a Global Privacy Control visitor sends nothing.\n- With a ref, a workspace owner can find one person's sessions and erase their sessions, events and recordings (Settings, Replay, Forget a person).\n\n## 11. Session replay\n\nReplay records a journey session as a masked copy of the page: layout, scrolling, clicks and page changes. Every word is masked until you choose to show it. Form values, checked boxes, chosen options, images, video, audio, canvas and iframes are never recorded.\n\n### Decide\n\n| Situation | Replay |\n|---|---|\n| Journey mode, and the person who asked wants replay | On |\n| Journey mode, and nobody said | Ask first. Each workspace records up to 10 sessions a month free. More needs a card on file (Settings, Billing). Past the allowance, recording stops until a card is added or the month ends. Your own test recordings count too |\n| Pages show health, legal or financial records, or other people's private messages | Off, unless the owner asks |\n| Aggregate mode | Not available |\n\n### Defaults\n\n| Setting | Default |\n|---|---|\n| Sample rate | `100`, unless the site expects more than about 1,000 recorded sessions a day. Then sample down, for example to `25`. Each project stores up to 1 GiB of recordings a day, about 2,000 typical recordings, and refuses more until midnight UTC |\n| Text shown | Nav, header, footer, headings, buttons, labels, table headers, legends, tabs, menu items |\n| Hidden elements | Support chat, third-party widgets, and any region that lists other people's data |\n| Pages never recorded | `/account`, `/settings`, `/billing`, `/checkout`, `/admin`, plus every private-data route from step 3 |\n| Minimum active time | 2 seconds, the built-in default. A shorter recording is dropped unless it has at least 3 clicks |\n\n### Turn it on\n\n```sh\nlittlefriend replay enable --rate 100\nlittlefriend replay set \\\n --unmask nav --unmask header --unmask footer \\\n --unmask h1 --unmask h2 --unmask h3 --unmask h4 \\\n --unmask button --unmask label --unmask th --unmask legend \\\n --unmask \"[role=button]\" --unmask \"[role=tab]\" --unmask \"[role=menuitem]\" \\\n --block .support-chat \\\n --exclude /account --exclude /settings --exclude /billing --exclude /checkout --exclude /admin\nlittlefriend snippet --framework next --mode journey --replay\n```\n\n- `replay set` replaces each list. Pass every item every time.\n- Add the replay code the snippet prints after `lf.js`, with the same site key. With npm, call `startReplay({ site })` from `@littlefriend/replay` after `init`.\n- An excluded route covers itself and every path below it: `/account` covers `/account/billing`. `*` stands for one segment: `/projects/*/settings` covers `/projects/acme/settings` and everything below it. Up to 50 routes, each at most 100 characters, with no query or fragment.\n- Selectors: tag names, classes, ids and attribute selectors with plain values, joined by spaces, `>` or commas. Pseudo-classes, sibling selectors and `*` are refused. Up to 50 per list, 200 characters each.\n\n### Mark the HTML\n\nEmails, phone numbers and card numbers stay masked even in shown text. Names do not. Search the shown regions (header, nav, account menu, buttons) for places that print the signed-in person's name, company, initials or avatar label, and mask them again:\n\n```html\n<header>\n <nav>...</nav>\n <button class=\"account-menu\" data-lf-mask>{user.name}</button>\n</header>\n\n<aside class=\"support-chat\" data-lf-block>...</aside>\n```\n\n- `data-lf-mask` masks text again inside a shown region.\n- `data-lf-block` leaves an element out, drawn as an empty box. Use it for regions whose shape alone says too much, and on sensitive parts of routes that draw their page late after navigation.\n- `data-lf-unmask` shows the text of one element, for plain product copy such as plan names and prices.\n\n`littlefriend replay get` shows the settings, and says the replay script is installed once the first recording arrives.\n\n## 12. Server events\n\nSend outcomes the backend confirms: account created, payment succeeded, plan changed. Default: yes for every product with sign-in or payments.\n\n### The key\n\n```sh\ngit check-ignore -q .env.local && echo ignored\nlittlefriend keys create --kind server --label \"acme server\" --write-env .env.local --env-name LF_SERVER_KEY\n```\n\n- Use the env file the framework loads (`.env.local` for Next.js, often `.env` elsewhere). It must be ignored by git. If `git check-ignore` prints nothing, add the file to `.gitignore` first.\n- With `--write-env`, the CLI appends `LF_SERVER_KEY=...` only if the name is not set yet, and never prints the secret.\n- Add `LF_SERVER_KEY=` with no value to `.env.example`, if the repo has one.\n- Production needs the same variable in the host's secret store. Copy it from the env file without printing it, if you have the host's CLI and the repo's rules allow it. Otherwise list it under \"Needs a person\". Check each host CLI's `--help` before you run these:\n\n```sh\n# Vercel\ngrep '^LF_SERVER_KEY=' .env.local | cut -d= -f2- | tr -d '\\n' | vercel env add LF_SERVER_KEY production\n\n# Fly\ngrep '^LF_SERVER_KEY=' .env | fly secrets import -a <app>\n\n# Cloudflare Workers (the edge key from step 13)\ngrep '^LF_EDGE_KEY=' .env | cut -d= -f2- | tr -d '\\n' | wrangler secret put LF_EDGE_KEY\n```\n\n### Send from Node.js\n\nInstall `@littlefriend/node` with the repo's package manager (Node 18.17 or later). Create one client at module scope:\n\n```ts\n// lib/little-friend.ts\nimport { LittleFriend } from '@littlefriend/node';\n\nconst key = process.env.LF_SERVER_KEY;\nexport const lf = key ? new LittleFriend({ key }) : null;\n```\n\nThen send each outcome where the backend confirms it: after the account row is written, in the payment webhook, after the plan change commits.\n\n```ts\nimport { createHash } from 'node:crypto';\nimport { lf } from './lib/little-friend';\n\nlf?.track({\n // 8 to 32 letters, digits, _ or -. The same id on a retry counts once.\n id: createHash('sha256').update(order.id).digest('base64url').slice(0, 32),\n name: 'order.completed',\n props: { plan: order.plan },\n value: { amount: order.totalCents, currency: 'USD' }, // minor units\n correlationId: order.checkoutRef, // optional, see below\n});\n\nawait lf?.flush(); // in a serverless function, before it returns\n```\n\n- `track` never throws. A bad event goes to `onError`, which logs a warning by default.\n- In serverless functions (Vercel, Next.js route handlers and server actions), `await lf.flush()` before returning. In a long-running server, `await lf.shutdown()` when the process exits.\n- Derive `id` from the record, as above. Raw UUIDs (36 characters) and short numeric ids fail the id rule.\n- Other languages: POST `{ \"v\": 1, \"events\": [ ... ] }` to `https://in.littlefriend.io/v1/server` with `Authorization: Bearer $LF_SERVER_KEY`. Retry `429` and `503` after `Retry-After`. Never retry a `400`. Details: https://littlefriend.io/docs/goals#http\n\n### Join server outcomes to journeys\n\nIn journey mode, a server event joins the visitor's session when it carries the same correlation id as a browser event:\n\n```ts\n// Browser, when checkout starts: a random id for this attempt, sent to the server with the form\nconst checkoutRef = crypto.randomUUID();\nlf('track', 'checkout.start', { plan: 'pro' }, checkoutRef);\n```\n\nThe server stores `checkoutRef` with the order and passes it as `correlationId`. A correlation id is 16 to 64 letters, digits, `_` or `-`, and never only digits and dashes. Aggregate mode drops it.\n\n## 13. Crawlers and AI agents: the door and a log drain\n\n`lf.js` sees only browsers. Crawlers and AI agents rarely run JavaScript, so Little Friend needs the requests themselves.\n\n| Where the product runs | See agents with | Door |\n|---|---|---|\n| Vercel, any framework, static too | A Vercel log drain | Next.js: `nextProxy` in `proxy.ts` (Next.js 16) or `middleware.ts` (Next.js 15). Other frameworks: skip it and say so |\n| Next.js on another host | `wrapFetch(edge, handler, { waitUntil })` in route handlers, or `nodeMiddleware(edge)` in a custom Node server | `nextProxy`, as on Vercel |\n| Cloudflare Workers | `withLittleFriend` from `@littlefriend/edge` | Add `door: createDoor` |\n| Node with Express, Connect or `node:http` | `nodeMiddleware(edge)` | `nodeDoor(createDoor(lf), createEdge(lf))` in its place |\n| Bun, Deno, Hono or another fetch handler | `wrapFetch(edge, handler)` | `wrapFetch(edge, handler, { door })` |\n| Celsian | `observeCelsian(app, edge)` | None built in. Skip it and say so |\n| Static hosting without functions, not on Vercel | Only with a Worker in front | Only with a Worker in front |\n\nA request both a drain and the edge package report is counted once.\n\nRequests to the project's ignored routes are counted and not kept. Every project starts with `/api`, because an app calling its own API (session checks, polling, presence) is not a crawler or an agent. Add the product's other API paths, such as `/trpc` or `/graphql`:\n\n```sh\nlittlefriend projects update --add-ignored-route /trpc\n```\n\nRemove `/api` only when the owner wants to watch who calls the API, and say so in the report.\n\n### A Vercel log drain\n\n```sh\nlittlefriend drains list\nlittlefriend drains create --provider vercel\n```\n\n- Check `drains list` first. Reuse an active drain.\n- `drains create` prints the endpoint, the header and the signing secret, once. Create it only when you can paste them into Vercel in the same sitting: in the team's settings, add a log drain for this project only, all sources, production, and 100% sampling. Drains need a Vercel Pro or Enterprise team.\n- If you cannot reach Vercel's settings, do not create the drain. List it under \"Needs a person\": they can do both halves on the dashboard's Agents page, with Add a Vercel log drain.\n- A lost secret: `littlefriend drains rotate <id>` issues a new pair and keeps the old one working for 24 hours.\n- Within minutes of traffic, the Coverage card on the Agents page shows the drain as Live.\n\n### `@littlefriend/edge` and the door\n\nUse this when the product has code that runs on every request. Install `@littlefriend/edge` 0.3.0 or later. The door came in 0.2.0, and 0.3.0 stops sending requests to ignored routes. If the repo already depends on `^0.2.0`, change it to `^0.3.0`: that range never picks up 0.3.0 on its own.\n\n1. Create an edge key into the env file, as in step 12:\n\n```sh\nlittlefriend keys create --kind edge --label \"acme edge\" --write-env .env.local --env-name LF_EDGE_KEY\n```\n\n2. Install `@littlefriend/edge` and wire it for the runtime in the table above. The exact code for each runtime is on https://littlefriend.io/docs/agents (sections Any host and The door). For Next.js 16:\n\n```ts\n// proxy.ts\nimport { createDoor, createEdge, nextProxy } from '@littlefriend/edge';\nimport { NextResponse } from 'next/server';\n\nconst lf = { key: process.env.LF_EDGE_KEY!, ipHeader: 'x-forwarded-for' };\nexport const proxy = nextProxy(createDoor(lf), NextResponse, createEdge(lf));\nexport const config = { matcher: ['/((?!_next/static|_next/image|favicon.ico).*)'] };\n```\n\n3. Set `ipHeader` to the header the host sets to the client address: `x-forwarded-for` on Vercel, `cf-connecting-ip` on Cloudflare, `fly-client-ip` on Fly. Without it, no address is sent and operator address checks cannot run.\n\n4. Pick a preset and keep dry run:\n\n```sh\nlittlefriend door set --preset verified_only --mode dry_run\nlittlefriend door simulate --ua \"Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; GPTBot/1.2; +https://openai.com/gptbot)\" --path /blog\n```\n\n| Product | Preset |\n|---|---|\n| A store with `/checkout`, `/cart` or `/account` | `commerce` |\n| An app or API with sign-in | `verified_only` |\n| Content, docs or marketing | `no_training` |\n| Unsure | `open` |\n\n- Dry run decides every request and records what it would have done. It never changes a response.\n- **Never switch the door to live.** The owner does that after reading 7 days of the door's report on the Agents page.\n- A browser always gets in, whatever the rules say.\n- `commerce` matches `/checkout`, `/cart` and `/account`. If the store uses other paths, say so in the report.\n\n## 14. Hybrid and native apps\n\n**Hybrid apps** (Capacitor, Ionic, Cordova, Tauri, Electron, React Native WebView, or the app's own WKWebView or Android WebView) show web pages. Install the script in the app's web code as in step 5, then add the origin the web view sends to the project's allowed origins, with the complete list:\n\n```sh\nlittlefriend projects update --origin https://acme.com --origin https://www.acme.com \\\n --origin http://localhost:3000 --origin capacitor://localhost --origin https://localhost\n```\n\n| App shell | Origin to add |\n|---|---|\n| Capacitor on iOS | `capacitor://localhost` |\n| Capacitor on Android | `https://localhost` |\n| Ionic with its Cordova web view | `ionic://localhost` on iOS, `http://localhost` on Android |\n| Cordova | `app://localhost` on iOS (scheme preference set to `app`), `https://localhost` on Android |\n| Tauri | `tauri://localhost` on macOS, iOS and Linux. `http://tauri.localhost` on Windows and Android, or `https://tauri.localhost` with `useHttpsScheme` |\n| Electron | The scheme and host the app serves from, such as `app://myapp` |\n| React Native WebView | The origin of the site it loads |\n| WKWebView with a custom scheme handler | The scheme and host you register, such as `app://localhost` |\n| WKWebView loading a file URL | No origin, see below |\n| Android WebView with WebViewAssetLoader | `https://appassets.androidplatform.net`, or `https://localhost` with `setDomain(\"localhost\")` |\n| Android WebView loading `file:///android_asset` | No origin, see below |\n\n- If the shell's config changes the scheme or host, add the ones it sets.\n- A page loaded from a file has no origin: an Android WebView on `file:///android_asset` sends `Origin: null`, a WKWebView that opened it with `loadFileURL` sends `Origin: null` from `fetch` and `Origin: file://` from `sendBeacon`, and an Electron app that loads its pages from files sends no `Origin` header at all. No list can name any of those, so leave the list empty, or serve the app from a scheme (a `WKURLSchemeHandler`, or `electron-serve`, which serves `app://-`) or from the asset loader. The script sends only the page's file name as the route, such as `/index.html`, so the folders above it, which can hold the user's name, never leave the device; the collector keeps the same rule for scripts older than 0.1.2, and keeps a route a current script names, such as `/settings/profile`, whole.\n- A file page cannot `pushState` to a new path: engines allow only a new query or fragment there. A hash-routed app names its routes itself. Name the starting route before the script loads, in the inline queue stub, as the template your router matches, such as `lf('route', '/orders/:id')`, or as the route part of the hash alone, `lf('route', '/' + location.hash.replace(/^#\\/?/, '').split(/[?&=]/)[0])`, so the load's own page view carries it. Never pass the whole fragment, since a fragment can carry a query or a sign-in token. The script cuts a named route at the first `?` or `#`, redacts a segment that holds `=` and replaces id-like segments, but a route should hold no values to begin with. After each later navigation, once the hash has changed, call `lf('route', '/settings')`, then `lf('page')`. Call `lf('page')` for navigations only: a router hook that also fires for the starting screen must skip that first call, or the screen counts twice. An app that routes with `pushState` gets its page views on its own and never calls `lf('page')`.\n- Apps on iOS send no referrer. A link opened from an Android app in Chrome arrives as `android-app://<package>/`, and a known app is stored under its web host, such as `mail.google.com` for Gmail, in that app's channel. The package table is under Webmail and email apps in `docs/METRICS.md`, published at https://littlefriend.io/docs/metrics#channels.\n- Events queued when the app goes to the background are sent as the page is hidden. An app killed while on screen (a crash, a force stop) loses what was queued in its last 5 seconds, and a kill and relaunch starts a new session.\n- UTM tags on a deep link are read like any landing: pass the link's query on to the page URL (`index.html?utm_source=newsletter&utm_medium=email`) and the visit lands in that channel with those values.\n- For local testing, an https asset-loader or Capacitor page cannot reach an http collector except at `localhost`: on the Android emulator, run `adb reverse tcp:<port> tcp:<port>` and point `data-api` at `http://localhost:<port>`. The real collector is https and needs neither.\n- If the repo deliberately skips analytics inside the shell, keep that and say so in the report.\n\n**Native screens** (SwiftUI, UIKit, Jetpack Compose, Flutter, React Native views) have no web page for the script. Send the outcomes they lead to from the backend with `@littlefriend/node` (step 12). The server key never goes inside the app.\n\n## 15. Verify\n\n1. Run the product: the local dev server (its origin is on the list from step 4), or the deployed site.\n2. Load the site in a real browser (Playwright, agent-browser, or a person). `curl` does not run the script. Headless Chrome says so in its user agent and is counted as automation, so its visits stay out of page views and funnels, which count people. For the journey walks in item 9, run the browser headed or give it a desktop Chrome user agent. Then run `littlefriend verify --source browser --since 10 --wait 120`. It exits 0 once an accepted event from the browser in the last 10 minutes shows up, and prints its time, route and source. `--source browser` matters on a project with a log drain or the edge SDK, whose events would otherwise pass for the tag's. It exits 1 when none arrives in time. On a project that already has traffic, keep `--since` short, so an older event cannot pass for yours.\n3. In the browser's network panel, `POST https://in.littlefriend.io/v1/e` answers `202` with `{\"accepted\": n, \"dropped\": 0}`. The console shows no CSP errors.\n4. Click through each flow with a named event and confirm each name in the request bodies. Any `dropped` above 0 means a name or property broke a rule. The live install check (Settings, Install) shows every field kept, every property dropped, and why.\n5. Replay: in the browser's network panel, `GET https://in.littlefriend.io/v1/r/config?k=<site key>` returns `\"on\": true`, and `POST /v1/r` answers `202`. From `curl`, send the page's origin with `-H 'Origin: https://<domain>'`: without an allowed origin the answer is `\"on\": false`. Then `littlefriend replay get` reads installed.\n6. Server events: run the flow that sends one. `verify` shows source `server`, or `await lf.flush()` resolves with `accepted: 1`.\n7. Agents: `littlefriend door get` shows the policy and its mode. On the dashboard's Agents page, the Coverage card shows the drain as Live, and the door page shows that the door has read its rules.\n8. A few minutes later: `littlefriend report overview` and `littlefriend report goals` for a sanity check.\n9. Walk each journey in your plan once, in one tab, step by step, on a page without `data-test` (test events stay out of reports). A few minutes later, `littlefriend funnels report fnl_XXXXXXXX` shows that session at every step. A step at 0 means its route or event name does not match what arrives: compare it with `littlefriend report pages` and the request bodies.\n\nVerify on production after the deploy too, without `data-test`.\n\n| Symptom | Cause |\n|---|---|\n| No request to `in.littlefriend.io` | CSP blocks the script, or the tag never rendered |\n| `403` `origin_not_allowed` | The page's origin is not in allowed origins |\n| `400` | Wrong or malformed site key, or a malformed batch |\n| Sessions unavailable, funnels empty | The project or the snippet is not in journey mode. Both must be |\n| Replay `403` `replay_off`, `session_in_aggregate_mode` or `route_excluded` | Replay is off, the project is in aggregate mode, or the page is excluded |\n| Replay `402` `replay_card_required` | The workspace used its free recordings this month and has no card on file |\n| Replay `429` `quota_exceeded` | The project stored its 1 GiB of recordings for the day, or one address used its share of that. It clears at midnight UTC. If it happens often, lower the sample rate |\n| Server events `rejected` in `onError` | Wrong kind of key, or a revoked one |\n\n## 16. Launch checklist\n\n- [ ] One project for the domain, in the right workspace, in the right mode.\n- [ ] Allowed origins list every production host, every app origin and the dev origin.\n- [ ] The script is on every page, once, with `data-test` outside production.\n- [ ] CSP updated, including report-only copies and header tests.\n- [ ] Every goal and funnel step has its own named event. No personal data in names, properties or routes.\n- [ ] The journey plan is written down, with one goal per outcome.\n- [ ] 2 to 5 goals and, in journey mode, 1 to 3 funnels saved, one per journey in the plan.\n- [ ] Each journey was walked once and shows at every step of its funnel report.\n- [ ] `identify` sends an opaque id, and `null` on sign-out (replay only).\n- [ ] Replay: shown regions checked for names, private routes excluded, widgets blocked.\n- [ ] Server key in an ignored env file and in the host's secrets. Serverless code flushes.\n- [ ] Door in dry run where the runtime allows it. Drain connected on Vercel.\n- [ ] Ignored routes cover the product's own API paths.\n- [ ] Privacy page text added.\n- [ ] `littlefriend verify --source browser` exits 0 on production, and batches show `\"dropped\": 0`.\n- [ ] No secret in your changes: `git diff <base branch> | grep -E 'lf[ser]_[A-Za-z0-9]'` prints nothing.\n- [ ] Report sent.\n\n## 17. Report back\n\nEnd with this report, filled in. Use \"skipped\" with a reason where a step did not apply. Never include a secret.\n\n```markdown\n## Little Friend setup: <product>\n\n- Repo and branch:\n- Domain and hosts:\n- Workspace:\n- Project id and site key (lf_):\n- Mode and consent:\n- Allowed origins:\n- Snippet: --framework <name>, in <file>\n- CSP: <files changed, or none needed>\n- Journey plan: <the table from step 3: journey, outcome goal, steps and how each is seen>\n- Named events: <name, where it fires>\n- Goals: <id, name, match>\n- Funnels: <id, name, steps>\n- Identify: <where it is called, what id it sends, or skipped>\n- Replay: <on or off, rate, shown, hidden, excluded routes>\n- Server events: <events, files, env var names, host secrets set or not>\n- Agents: <drain id and status, door runtime, preset and mode, or skipped>\n- Ignored routes: <the list, and any change from /api>\n- Hybrid or native: <origins added, backend events, or none>\n- Privacy page: <updated, or no privacy page>\n- Verify: <time, route and source that verify printed, local and production>\n- Skipped, and why:\n- Needs a person: <drains to connect, secrets to set, questions>\n- Guide or CLI problems found:\n```\n", bundled: true } : { text: GUIDE_PLACEHOLDER, bundled: false };
|
|
3803
|
+
return true ? { text: "# Set up Little Friend in a product: guide for coding agents\n\nYou are a coding agent working in a product's repository. This guide takes you from nothing to a verified Little Friend install: the project, the script, events, goals, funnels, session replay, server events and AI agent visibility. It ends with a report for the person who asked.\n\nWork through the steps in order. Each decision has a default: use it unless the repo or the person who asked says otherwise. When a step does not apply, skip it and say why in the report.\n\n## Contents\n\n0. [Conventions](#0-conventions)\n1. [What Little Friend is, and the rules](#1-what-little-friend-is-and-the-rules)\n2. [Sign in once per machine](#2-sign-in-once-per-machine)\n3. [Read the repo and plan the journeys](#3-read-the-repo-and-plan-the-journeys)\n4. [Workspace, project, mode and consent](#4-workspace-project-mode-and-consent)\n5. [Install the script](#5-install-the-script)\n6. [Content Security Policy](#6-content-security-policy)\n7. [Custom events](#7-custom-events)\n8. [Goals](#8-goals)\n9. [Funnels](#9-funnels)\n10. [Identify signed-in people](#10-identify-signed-in-people)\n11. [Session replay](#11-session-replay)\n12. [Server events](#12-server-events)\n13. [Crawlers and AI agents: the door and a log drain](#13-crawlers-and-ai-agents-the-door-and-a-log-drain)\n14. [Hybrid and native apps](#14-hybrid-and-native-apps)\n15. [Verify](#15-verify)\n16. [Launch checklist](#16-launch-checklist)\n17. [Report back](#17-report-back)\n\n## 0. Conventions\n\n- **The CLI.** Run it as `npx --yes @littlefriend/cli <command>`. It needs Node 22 or later. This guide writes `littlefriend <command>` for short: type the full `npx` form, unless the CLI is installed globally.\n- **Scope flags.** Commands that work on a project take `--project <id>`, and commands that work in a workspace take `--workspace <id or name>`. Agent shells often drop environment variables between calls, so pass the flags on every command. The examples below leave them out to stay short. The CLI also reads `LITTLEFRIEND_PROJECT` and `LITTLEFRIEND_WORKSPACE`.\n- **Reading output.** Add `--json` when you need a value from the output, such as an id. Errors go to stderr with a non-zero exit code. With `--json`, they are also printed on stdout as `{ \"error\": { \"code\", \"message\" } }`.\n- **Secrets.** Server keys (`lfs_`), edge keys (`lfe_`), read keys (`lfr_`) and drain secrets are shown once. Never put them in your messages, logs, commits or the report. Write keys to an env file with `--write-env` (step 12). The public site key (`lf_`) is fine to commit.\n- **Credentials.** `~/.littlefriend/credentials.json` holds the sign-in. Never read it, print it or copy it into a repo.\n- **Deletes.** Never delete or revoke a project, key, goal, funnel or drain you did not create in this session.\n- **Repo rules win.** Follow the product repo's own rules for branches, commits, dependencies and deploys.\n- **The code wins.** `littlefriend <command> --help` shows a command's flags. If this guide and the CLI disagree, trust the CLI and say so in the report.\n\n## 1. What Little Friend is, and the rules\n\nLittle Friend is privacy-first web analytics (littlefriend.io). A small script, `lf.js`, counts page views, sources, clicks and named events without cookies. Server events count outcomes your backend confirms. Log drains and `@littlefriend/edge` show crawlers and AI agents, and the door decides which agents get in. Reports live at app.littlefriend.io.\n\nThe product must keep these rules. If a task would break one, stop and ask.\n\n1. **No personal data** in event names, property keys or values, routes, ids, or goal and funnel names. Personal data means emails, names, usernames, phone numbers, addresses, account or card numbers, text a person typed, and anything else that points to one person.\n2. **`identify` takes an opaque id** from the product's own database, such as `usr_8f3k2`. Never an email, a phone number, a name, or a hash of any of them.\n3. **No typed text.** Never send form values, search queries or error messages as properties. Report a failed form with a short category, such as `validation` or `server`.\n4. **Secret keys stay on the server.** Never put `lfs_`, `lfe_` or `lfr_` keys in client code, or in a variable a bundler exposes to the browser (`NEXT_PUBLIC_`, `VITE_`, `PUBLIC_`, `NUXT_PUBLIC_`, `EXPO_PUBLIC_`).\n5. **No extra tracking.** Do not add cookies, storage or fingerprinting for analytics. Do not work around Global Privacy Control or a visitor's opt-out: the script already honors both.\n6. **Unknown stays unknown.** Never combine an IP address with device traits to guess who someone is.\n\nLittle Friend also strips query strings and fragments (except allowlisted `utm_*` values), turns id-like and personal-looking path segments into `:id` or `:redacted`, drops property values that look like an email or a long number, and never reads form values or page text. Treat that as a safety net. Rule 1 still applies.\n\n## 2. Sign in once per machine\n\n```sh\nlittlefriend whoami\n```\n\nIf it names an account, go to step 3. Otherwise:\n\n```sh\nlittlefriend login\n```\n\nIt opens a browser, prints the URL too, and waits up to 5 minutes. The person who owns the Little Friend account approves the login in the browser. Tell them a sign-in page is waiting, then wait. If nobody approves in time, stop and ask. Do not retry in a loop.\n\n- The CLI acts with that person's role in each workspace. Creating projects, keys, goals, funnels and settings needs the editor or owner role.\n- In CI, set `LITTLEFRIEND_TOKEN` to an access token instead of signing in.\n- `littlefriend logout` revokes the sign-in and deletes the local credentials.\n\n## 3. Read the repo and plan the journeys\n\nCollect these facts before you create anything. They drive every later decision, and most of them go into the report.\n\n| Fact | Where to look |\n|---|---|\n| Product name and production domain | README, deploy config (`vercel.json`, `fly.toml`, `netlify.toml`, `wrangler.toml`), site URL env vars, canonical tags, sitemap config |\n| Every host the product serves pages from | `www.`, `app.`, docs subdomains, redirect config |\n| Web framework | `package.json` and config files. The table in step 5 maps them |\n| Where the HTML `<head>` is rendered | Root layout, `index.html`, document template |\n| Sign-in | Auth libraries (Auth.js, Clerk, Supabase, Lucia, Better Auth, Passport), `/login` and `/signup` routes, a current-user hook |\n| Money | Checkout, pricing, plans, trials, payment webhooks |\n| Onboarding and activation | First-run screens, \"create your first ...\" flows, invites |\n| Pages with private data | Account, settings, billing, admin, messages, documents, and any health, legal or financial records |\n| Consent tool | A cookie banner or consent manager (OneTrust, Cookiebot, Osano, Klaro, a custom one) |\n| Content Security Policy | Search code, config and headers files for `Content-Security-Policy` |\n| Server runtime and host | API routes, server framework (Express, Celsian, Hono, Next.js route handlers), Workers, Fly or Vercel config |\n| Env files | `.env`, `.env.local`, `.env.example`, and what `.gitignore` ignores |\n| App shells | `capacitor.config.*`, `ionic.config.json`, Cordova `config.xml`, `src-tauri/`, an Electron main process, React Native or Expo |\n| Existing analytics | Google Analytics, Plausible, PostHog, Segment or Vercel Analytics calls |\n\nExisting analytics: leave them in place unless asked to remove them. Their event lists are good candidates for your named events and goals.\n\n### Plan the journeys\n\nBefore you add a single event, decide what this product exists to get people to do, and which paths lead there. Steps 7 to 9 build exactly this plan, so every event, goal and funnel has a reason.\n\n1. **Say the product's job in one sentence:** who arrives, and what a good visit ends with. Read the home page, the pricing page, the main call to action, and the signup or checkout code.\n2. **List the outcomes, most valuable first.** Usually 1 to 3: money (purchase, upgrade, trial), an account (signup), a lead (contact, demo booked), or activation, the first time someone gets value (created a project, sent an invite, published a page).\n3. **Write one journey per top outcome, at most 3.** A journey is the path a person takes from arriving to the outcome, in 3 to 5 steps they would recognize: where it really starts (a landing page, pricing, a docs page), the moment they commit (opened signup, started checkout), and the outcome itself.\n4. **Decide how each step is seen.** A page is a route step (`route:/pricing`). An action inside a page, or something the server confirms, is a named event you add in step 7 (`event:signup.start`). The outcome becomes a goal in step 8, and the journey becomes a saved funnel in step 9.\n5. **Check the routes.** Read the router for each route step. Ids and personal segments are stored as `:id` or `:redacted`, so `/projects/42/settings` is stored as `/projects/:id/settings`.\n\nStart from the row that fits, then change it to match what the code really does:\n\n| Product | Outcome | Journey |\n|---|---|---|\n| App or SaaS | Signup, then activation | `route:/` \u2192 `route:/pricing` \u2192 `event:signup.start` \u2192 `goal:<Signup id>` \u2192 `goal:<Activated id>` |\n| Store | Purchase | `route:/products/:id` \u2192 `event:cart.add` \u2192 `event:checkout.start` \u2192 `goal:<Purchase id>` |\n| Services or lead generation | Contact or booking | `route:/` \u2192 `route:/services` \u2192 `event:contact.open` \u2192 `goal:<Contact sent id>` |\n| Docs or open source | Adoption | `route:/docs` \u2192 `route:/docs/getting-started` \u2192 `goal:<Install copied id>`, a goal on `install.copy` |\n| Marketing site for an app on another host | A visitor heads to sign up | `route:/` \u2192 `route:/pricing` \u2192 `goal:<Signup click id>`, a goal on `cta.signup` |\n\n**A journey ends at the edge of its host.** A journey-mode session lives in one tab's session storage, which browsers keep apart for each host. A visit that moves from `example.com` to `app.example.com` becomes two sessions. Give each host its own funnel: end the marketing site's funnel on a goal for a named event fired when the signup link is clicked (`lf('track', 'cta.signup')`), and start the app's funnel at its signup page.\n\nKeep the plan small and useful:\n\n- Pick what someone running the product would ask about on a Monday: \"Of the people who saw pricing, how many started a trial?\" If nobody would act on a number, do not build it.\n- At most 3 funnels and 5 goals per project. Every funnel ends on a goal.\n- A step the code cannot see yet (no route, no event) needs an event in step 7. Note it in the plan.\n- A server-confirmed outcome joins the journey only when the server event carries the browser's correlation id (step 12). Plan that, or end the funnel on the browser event just before it.\n- Aggregate mode has no sessions, so it has no funnels. Plan goals only, and say so in the report.\n\nWrite the plan into the report (step 17) as a table before you build anything:\n\n| Journey | Outcome goal | Steps, and how each is seen |\n|---|---|---|\n| Signup | Signup (server) | `/pricing` route, `signup.start` event on the button, `signup.completed` server event with the browser's correlation id |\n\n## 4. Workspace, project, mode and consent\n\n### Workspace\n\n```sh\nlittlefriend workspaces list\n```\n\n| Situation | Workspace |\n|---|---|\n| The person who asked named one | That one |\n| The list shows one workspace | That one |\n| Several, and nobody named one | Stop and ask. Never guess |\n| The one you need is missing | Stop and ask. The CLI does not create workspaces |\n\n### Reuse before you create\n\n```sh\nlittlefriend projects list --workspace <workspace>\n```\n\nIf a project already has the product's domain, reuse it: note its id and site key, and run `littlefriend projects get` to see its mode and allowed origins. Never create a second project for the same domain.\n\nOne product is one project, even when its marketing pages and app live on different hosts of the same domain. List every host in allowed origins.\n\n### Mode\n\n| The product has | Mode |\n|---|---|\n| Sign-in, signup, onboarding, checkout, or any flow of several steps | `journey` |\n| Only marketing or content pages, and no sign-in | `aggregate` |\n| It is a site you build for a client | `aggregate`, unless the client asked for journeys |\n\n- Aggregate mode keeps counts only, with no session or visitor key anywhere. It still gives views, sources, pages, devices, countries, events and goals.\n- Journey mode adds sessions, timelines, funnels, paths, entry and exit pages, session replay and identify. Its session id lives in the tab's session storage and ends after 30 idle minutes (24 hours at most).\n\n### Consent\n\n| Situation | Do this |\n|---|---|\n| The product already runs a consent tool | Use `--consent required` in step 5. Call `lf('consent', 'journey')` when the tool grants analytics consent, and `lf('consent', 'aggregate')` when it is withdrawn |\n| No consent tool, journey mode | Ask the person who asked whether to ship without a banner. Do not add a consent tool yourself |\n| Aggregate mode | Nothing to wire |\n\nUntil consent arrives, the script counts in aggregate mode. A browser that sends Global Privacy Control always stays in aggregate mode.\n\nIf the product has a privacy page, add this text for journey mode:\n\n> We use Little Friend for privacy-friendly analytics. It sets no cookies and does not store IP addresses. To connect the pages you view in one visit, it keeps a random id in your browser tab's session storage. The visit ends after 30 minutes of inactivity, and the id is gone when you close the tab. If your browser sends Global Privacy Control, your visit is only counted, never connected.\n\nIn aggregate mode, use only the first two sentences. With session replay on (step 11), add:\n\n> Some visits are recorded as a masked copy of the pages you view. Text stays hidden unless we chose to show it, form entries are never recorded, and recordings are deleted within 7 days.\n\nIf there is no privacy page, say so in the report.\n\n### Create the project\n\n```sh\nlittlefriend projects create --name \"Acme\" --domain acme.com --mode journey \\\n --origin https://acme.com --origin https://www.acme.com --origin http://localhost:3000 \\\n --timezone America/New_York\n```\n\nIt prints the project id and the public site key. Keep both for the rest of the guide and the report.\n\n- `--domain`: the production host. Little Friend stores it lowercase, without a scheme, path or `www.`.\n- `--origin`, once per origin: every production host that serves pages, plus the local dev server's origin so you can verify before deploy (step 15). Add app origins from step 14. Preview deploys and any origin not listed get `403` and count nowhere, which is intended. An empty list accepts every origin.\n- `--timezone`: the time zone the business reports in. Without it, reports use UTC.\n- To add an origin later, run `littlefriend projects update --add-origin https://app.acme.com`. `--remove-origin` takes one out, and `--origin` replaces the whole list.\n- A native iOS or macOS app sends no origin. List its bundle id with `littlefriend apps add com.acme.shop`.\n\n## 5. Install the script\n\nPrint the exact code for the stack:\n\n```sh\nlittlefriend snippet --framework next --mode journey\n```\n\nIt prints what to add, with the site key filled in, and the CSP lines the site needs. Add `--consent required` when step 4 said so. Add `--replay` only in step 11.\n\n| The repo has | `--framework` | Where the code goes |\n|---|---|---|\n| `next` | `next` | The root layout: `app/layout.tsx` (App Router) or `pages/_document.tsx` (Pages Router) |\n| `nuxt` | `nuxt` | `nuxt.config.ts` |\n| `@sveltejs/kit` | `sveltekit` | `src/app.html` |\n| `astro` | `astro` | The layout every page shares, often `src/layouts/Layout.astro` |\n| `@remix-run/*`, or React Router with `app/root.tsx` | `remix` | `app/root.tsx` |\n| `react` with Vite or another bundler, no framework | `react` | `index.html` |\n| `vue` with Vite | `vue` | `index.html` |\n| `@angular/core` | `angular` | `src/index.html` |\n| A WordPress theme or plugin | `wordpress` | Where the snippet says |\n| `what-framework`, a Shopify theme, plain HTML, anything else | `html` | The `<head>` every page shares: `index.html`, the document template, `layout/theme.liquid` |\n| A bundled app where you want typed functions | `npm` | The client entry, once at startup |\n\nThe snippet is the source of truth for the code. The last column is where to look first.\n\nRules:\n\n- **One tracker per page.** Use the tag or the npm package (`@littlefriend/tracker`), never both. Load it on every page, once.\n- **Single-page apps need nothing extra.** History API navigations count as page views. If the router does not use the History API, call `lf('page')` after each navigation. Call it for navigations only, never for a screen the load or a History API navigation already recorded.\n- **Outside production, mark traffic as test.** Add `data-test` to the tag, or pass `test: true` to `init`, when the build is not production (for example `process.env.NODE_ENV !== 'production'` or `import.meta.env.DEV`). Test traffic shows in the live install check and never in reports.\n- **Name dynamic routes.** Id-like segments already become `:id`. For readable groups, such as `/blog/:slug`, add `data-routes='[\"/blog/:slug\"]'` to the tag, pass `routes` to `init`, or call `lf('route', '/blog/:slug')`. Goals and funnel steps match these routes exactly.\n- **Calls before the script loads.** If the product's own code calls `lf(...)`, add this stub above the tag, so `lf` exists before the deferred script runs. Calls wait in a queue until it loads.\n\n```html\n<script>window.lf=window.lf||function(){(lf.q=lf.q||[]).push(arguments)}</script>\n```\n\n- **Secret scanners.** The site key (`lf_`) is public, but gitleaks and similar scanners may flag it as an API key. Add an allowlist entry for that exact key rather than hiding it. A constant named `LITTLE_FRIEND_SITE` rather than `..._KEY` trips fewer rules.\n- **Monorepo build caches.** If the tag only renders in production (for example on `VERCEL_ENV`), declare that variable for the build task in `turbo.json` or the cache's equivalent. Otherwise a preview build and a production build can share one cached output.\n\nWith the tag in a TypeScript app, declare the global once, for example in `src/lf.d.ts`:\n\n```ts\ndeclare global {\n function lf(command: string, ...args: unknown[]): void;\n}\nexport {};\n```\n\n## 6. Content Security Policy\n\nSkip this step if the product sets no CSP. Otherwise add:\n\n```text\nscript-src https://cdn.littlefriend.io\nconnect-src https://in.littlefriend.io\n```\n\n- With the npm package, the tracker is part of your bundle, so only `connect-src` is needed.\n- The replay script comes from the same host as `lf.js` and sends to the same collector. It needs no other entries.\n- A policy with a nonce or `'strict-dynamic'` ignores host entries for scripts. Give the tag the page's nonce, or use the npm package.\n- Update every copy of the policy: `Content-Security-Policy-Report-Only`, `<meta http-equiv>` tags, and the test files that assert headers.\n- Common places: `next.config.*` `headers()`, `middleware.ts` or `proxy.ts`, `vercel.json`, `netlify.toml`, `_headers`, Helmet options, server header code.\n\n## 7. Custom events\n\nThe script sends these on its own: page views, outbound links, downloads, engaged time and, with `data-scroll`, scroll depth. The `$` prefix is reserved for them.\n\nAdd the named events your journey plan (step 3) needs, and the few the product's key feature needs. Three ways to add more:\n\n```html\n<!-- A named click: sends $click with { \"id\": \"pricing.start_trial\", \"plan\": \"pro\" } -->\n<button data-lf=\"pricing.start_trial\" data-lf-plan=\"pro\">Start free trial</button>\n\n<!-- A tracked form: sends $form_start and $form_submit with { \"id\": \"signup\" } -->\n<form data-lf-form=\"signup\">...</form>\n```\n\n```ts\n// A named event, after the thing really happened\nlf('track', 'signup.completed', { plan: 'pro' });\n\n// A failed form: a short category, never the message\nlf('formError', 'signup', 'validation');\n```\n\nWith npm, import the same names: `track`, `formError`, `page`, `route`, `consent`, `optout`, `optin`.\n\n**Goals and funnel steps match an event name or a route, never a property.** Every `data-lf` click arrives as `$click`, and every tracked form as `$form_submit`. For anything that becomes a goal or a funnel step, fire its own named event with `track`.\n\n### Names\n\n- `area.action`, lowercase: dots between parts, underscores inside a part. `signup.start`, `signup.completed`, `onboarding.project_created`, `checkout.start`, `order.completed`, `plan.upgraded`, `invite.sent`.\n- Past tense for outcomes (`completed`, `created`, `sent`). `start` or `open` for intent.\n- The pattern is `^[a-z][a-z0-9_.:-]{0,63}$`. `data-lf` ids follow the same convention.\n- Fire outcome events after the server call succeeds, never on the button press. Fire intent events on the press.\n- Aim for 5 to 15 named events: the steps of signup, onboarding, activation and checkout, and the product's key feature. Do not name every click.\n\n| Good | Bad | Why the bad one fails |\n|---|---|---|\n| `signup.completed` | `Signup Completed` | Uppercase and spaces are refused |\n| `checkout.start` | `click_button_3` | Says nothing about the product |\n| `invite.sent` with `{ role: 'editor' }` | `invite.sent.jane@acme.com` | Personal data in the name |\n| `order.completed` with `{ plan: 'pro' }` | `order_8812_completed` | An id in the name makes a new event per order |\n| `search.submit` with `{ results: 12 }` | `search.submit` with `{ query: 'knee pain' }` | Typed text is personal data |\n| `project.created` | `$project_created` | `$` is reserved |\n\n### Properties\n\n- Up to 8 per event. Keys are lowercase snake case, up to 32 characters. Values are strings, numbers or booleans. Strings are cut to 64 characters.\n- Use small fixed sets (`plan`, `step`, `method`, `role`, `source`) and counts.\n- A value that looks like an email or a long run of digits is dropped, even when sent on purpose.\n\n### Page types (optional)\n\nFlows group pages by type. Little Friend suggests types from your paths, and owners and editors can change the rules in Settings, Page types. When a page's type can't be told from its path, set it on the page:\n\n```html\n<meta name=\"lf:type\" content=\"comparison\">\n```\n\nThe value is a lowercase key: letters, digits, `-` or `_`, starting with a letter, at most 32 characters. A tag beats the URL rules for that page. The tracker reads this one tag and nothing else.\n\nOn a client-side navigation the page view is recorded when the URL changes. Its type is read after the next animation frame, or at once when the tab is hidden, so head managers that write the tag on render are read correctly. On page load the tag is read when the script runs, so put it in the HTML the server sends.\n\n## 8. Goals\n\nA goal is a conversion. Create 2 to 5, one for each outcome in your journey plan (step 3), most valuable first.\n\n| Product | Goals to start with |\n|---|---|\n| App or SaaS with sign-in | Signup (server), activation: the first key action, upgrade or trial started (server) |\n| Store | Purchase (server, with value), checkout started |\n| Lead generation | Contact form sent, demo booked, the thank-you page |\n| Content or docs | Newsletter signup, a named event on the link to the product's signup |\n\n- Prefer server-confirmed goals for money and accounts: `--source server --server-confirmed`. `--source server` counts only events your server sends with a secret key, and `--server-confirmed` also marks the goal that way in reports, so a reader knows a browser cannot fake it. Use browser goals for intent, or when there is no backend.\n- A route goal matches the stored route exactly. Check the routes with `littlefriend report pages` first.\n- Create goals before launch. A goal counts from the moment it is saved.\n- Name goals in plain words: \"Signup\", \"Purchase\", \"Demo booked\".\n\n```sh\nlittlefriend goals create --name \"Signup\" --event signup.completed --source server --server-confirmed\nlittlefriend goals create --name \"Purchase\" --event order.completed --source server --server-confirmed\nlittlefriend goals create --name \"Activated\" --event project.created\nlittlefriend goals create --name \"Contact sent\" --route /contact/thanks\nlittlefriend goals list\n```\n\nKeep the goal ids (`goal_...`) for funnels and the report.\n\n## 9. Funnels\n\nJourney mode only. Save one funnel for each journey in your plan (step 3), 1 to 3 in all, named after the journey.\n\nA funnel has 1 to 8 steps in order. Each step is `route:/path`, `event:<name>` or `goal:<goalId>`. A session reaches a step when it has done the steps before it, in that order.\n\n| Flow | Steps |\n|---|---|\n| Signup | `route:/pricing` \u2192 `event:signup.start` \u2192 `goal:<Signup id>` |\n| Activation | `goal:<Signup id>` \u2192 `event:onboarding.profile_completed` \u2192 `goal:<Activated id>` |\n| Checkout | `route:/pricing` \u2192 `event:checkout.start` \u2192 `goal:<Purchase id>` |\n\n- Keep 3 to 5 steps. Start broad and end on a goal.\n- A server event joins a session only when it carries that session's correlation id or session id (step 12). Without one, a server-only goal as the last step is never reached. Pass a correlation id, or end on a browser event.\n\n```sh\nlittlefriend funnels create --name \"Signup\" \\\n --step route:/pricing --step event:signup.start --step goal:goal_XXXXXXXX\nlittlefriend funnels list\nlittlefriend funnels report fnl_XXXXXXXX --from 2026-10-01 --to 2026-10-07\n```\n\nDates are `YYYY-MM-DD` in the project's time zone.\n\n## 10. Identify signed-in people\n\nJourney mode only. `lf('identify', ...)` is handled by the replay script, so it needs step 11. Without replay, skip this step.\n\n```ts\n// Wherever the app knows the signed-in user: after sign-in and on each page load\nlf('identify', user.id); // your own opaque id, such as usr_8f3k2\n\n// On sign-out\nlf('identify', null);\n```\n\nWith npm: `import { identify } from '@littlefriend/replay'`, then `identify(user.id)` and `identify(null)`.\n\n- The ref is 1 to 64 letters, digits, `_`, `.`, `:` and `-`. UUIDs and prefixed ids work.\n- Prefix numeric ids, such as `usr_4821937`. A ref of seven or more bare digits is refused.\n- Never an email, a phone number, a name, a username, or a hash of any of them. Emails are refused.\n- The ref lasts for the journey session in that tab. Call it on each page load while signed in, so the next session carries it too.\n- Before consent it waits in memory, and a Global Privacy Control visitor sends nothing.\n- With a ref, a workspace owner can find one person's sessions and erase their sessions, events and recordings (Settings, Replay, Forget a person).\n\n## 11. Session replay\n\nReplay records a journey session as a masked copy of the page: layout, scrolling, clicks and page changes. Every word is masked until you choose to show it. Form values, checked boxes, chosen options, images, video, audio, canvas and iframes are never recorded.\n\n### Decide\n\n| Situation | Replay |\n|---|---|\n| Journey mode, and the person who asked wants replay | On |\n| Journey mode, and nobody said | Ask first. Each workspace records up to 10 sessions a month free. More needs a card on file (Settings, Billing). Past the allowance, recording stops until a card is added or the month ends. Your own test recordings count too |\n| Pages show health, legal or financial records, or other people's private messages | Off, unless the owner asks |\n| Aggregate mode | Not available |\n\n### Defaults\n\n| Setting | Default |\n|---|---|\n| Sample rate | `100`, unless the site expects more than about 1,000 recorded sessions a day. Then sample down, for example to `25`. Each project stores up to 1 GiB of recordings a day, about 2,000 typical recordings, and refuses more until midnight UTC |\n| Text shown | Nav, header, footer, headings, buttons, labels, table headers, legends, tabs, menu items |\n| Hidden elements | Support chat, third-party widgets, and any region that lists other people's data |\n| Pages never recorded | `/account`, `/settings`, `/billing`, `/checkout`, `/admin`, plus every private-data route from step 3 |\n| Minimum active time | 2 seconds, the built-in default. A shorter recording is dropped unless it has at least 3 clicks |\n\n### Turn it on\n\n```sh\nlittlefriend replay enable --rate 100\nlittlefriend replay set \\\n --unmask nav --unmask header --unmask footer \\\n --unmask h1 --unmask h2 --unmask h3 --unmask h4 \\\n --unmask button --unmask label --unmask th --unmask legend \\\n --unmask \"[role=button]\" --unmask \"[role=tab]\" --unmask \"[role=menuitem]\" \\\n --block .support-chat \\\n --exclude /account --exclude /settings --exclude /billing --exclude /checkout --exclude /admin\nlittlefriend snippet --framework next --mode journey --replay\n```\n\n- `replay set` replaces each list. Pass every item every time.\n- Add the replay code the snippet prints after `lf.js`, with the same site key. With npm, call `startReplay({ site })` from `@littlefriend/replay` after `init`.\n- An excluded route covers itself and every path below it: `/account` covers `/account/billing`. `*` stands for one segment: `/projects/*/settings` covers `/projects/acme/settings` and everything below it. Up to 50 routes, each at most 100 characters, with no query or fragment.\n- Selectors: tag names, classes, ids and attribute selectors with plain values, joined by spaces, `>` or commas. Pseudo-classes, sibling selectors and `*` are refused. Up to 50 per list, 200 characters each.\n\n### Mark the HTML\n\nEmails, phone numbers and card numbers stay masked even in shown text. Names do not. Search the shown regions (header, nav, account menu, buttons) for places that print the signed-in person's name, company, initials or avatar label, and mask them again:\n\n```html\n<header>\n <nav>...</nav>\n <button class=\"account-menu\" data-lf-mask>{user.name}</button>\n</header>\n\n<aside class=\"support-chat\" data-lf-block>...</aside>\n```\n\n- `data-lf-mask` masks text again inside a shown region.\n- `data-lf-block` leaves an element out, drawn as an empty box. Use it for regions whose shape alone says too much, and on sensitive parts of routes that draw their page late after navigation.\n- `data-lf-unmask` shows the text of one element, for plain product copy such as plan names and prices.\n\n`littlefriend replay get` shows the settings, and says the replay script is installed once the first recording arrives.\n\n## 12. Server events\n\nSend outcomes the backend confirms: account created, payment succeeded, plan changed. Default: yes for every product with sign-in or payments.\n\n### The key\n\n```sh\ngit check-ignore -q .env.local && echo ignored\nlittlefriend keys create --kind server --label \"acme server\" --write-env .env.local --env-name LF_SERVER_KEY\n```\n\n- Use the env file the framework loads (`.env.local` for Next.js, often `.env` elsewhere). It must be ignored by git. If `git check-ignore` prints nothing, add the file to `.gitignore` first.\n- With `--write-env`, the CLI appends `LF_SERVER_KEY=...` only if the name is not set yet, and never prints the secret.\n- Add `LF_SERVER_KEY=` with no value to `.env.example`, if the repo has one.\n- Production needs the same variable in the host's secret store. Copy it from the env file without printing it, if you have the host's CLI and the repo's rules allow it. Otherwise list it under \"Needs a person\". Check each host CLI's `--help` before you run these:\n\n```sh\n# Vercel\ngrep '^LF_SERVER_KEY=' .env.local | cut -d= -f2- | tr -d '\\n' | vercel env add LF_SERVER_KEY production\n\n# Fly\ngrep '^LF_SERVER_KEY=' .env | fly secrets import -a <app>\n\n# Cloudflare Workers (the edge key from step 13)\ngrep '^LF_EDGE_KEY=' .env | cut -d= -f2- | tr -d '\\n' | wrangler secret put LF_EDGE_KEY\n```\n\n### Send from Node.js\n\nInstall `@littlefriend/node` with the repo's package manager (Node 18.17 or later). Create one client at module scope:\n\n```ts\n// lib/little-friend.ts\nimport { LittleFriend } from '@littlefriend/node';\n\nconst key = process.env.LF_SERVER_KEY;\nexport const lf = key ? new LittleFriend({ key }) : null;\n```\n\nThen send each outcome where the backend confirms it: after the account row is written, in the payment webhook, after the plan change commits.\n\n```ts\nimport { createHash } from 'node:crypto';\nimport { lf } from './lib/little-friend';\n\nlf?.track({\n // 8 to 32 letters, digits, _ or -. The same id on a retry counts once.\n id: createHash('sha256').update(order.id).digest('base64url').slice(0, 32),\n name: 'order.completed',\n props: { plan: order.plan },\n value: { amount: order.totalCents, currency: 'USD' }, // minor units\n correlationId: order.checkoutRef, // optional, see below\n});\n\nawait lf?.flush(); // in a serverless function, before it returns\n```\n\n- `track` never throws. A bad event goes to `onError`, which logs a warning by default.\n- In serverless functions (Vercel, Next.js route handlers and server actions), `await lf.flush()` before returning. In a long-running server, `await lf.shutdown()` when the process exits.\n- Derive `id` from the record, as above. Raw UUIDs (36 characters) and short numeric ids fail the id rule.\n- Other languages: POST `{ \"v\": 1, \"events\": [ ... ] }` to `https://in.littlefriend.io/v1/server` with `Authorization: Bearer $LF_SERVER_KEY`. Retry `429` and `503` after `Retry-After`. Never retry a `400`. Details: https://littlefriend.io/docs/goals#http\n\n### Join server outcomes to journeys\n\nIn journey mode, a server event joins the visitor's session when it carries the same correlation id as a browser event:\n\n```ts\n// Browser, when checkout starts: a random id for this attempt, sent to the server with the form\nconst checkoutRef = crypto.randomUUID();\nlf('track', 'checkout.start', { plan: 'pro' }, checkoutRef);\n```\n\nThe server stores `checkoutRef` with the order and passes it as `correlationId`. A correlation id is 16 to 64 letters, digits, `_` or `-`, and never only digits and dashes. Aggregate mode drops it.\n\n## 13. Crawlers and AI agents: the door and a log drain\n\n`lf.js` sees only browsers. Crawlers and AI agents rarely run JavaScript, so Little Friend needs the requests themselves.\n\n| Where the product runs | See agents with | Door |\n|---|---|---|\n| Vercel, any framework, static too | A Vercel log drain | Next.js: `nextProxy` in `proxy.ts` (Next.js 16) or `middleware.ts` (Next.js 15). Other frameworks: skip it and say so |\n| Next.js on another host | `wrapFetch(edge, handler, { waitUntil })` in route handlers, or `nodeMiddleware(edge)` in a custom Node server | `nextProxy`, as on Vercel |\n| Cloudflare Workers | `withLittleFriend` from `@littlefriend/edge` | Add `door: createDoor` |\n| Node with Express, Connect or `node:http` | `nodeMiddleware(edge)` | `nodeDoor(createDoor(lf), createEdge(lf))` in its place |\n| Bun, Deno, Hono or another fetch handler | `wrapFetch(edge, handler)` | `wrapFetch(edge, handler, { door })` |\n| Celsian | `observeCelsian(app, edge)` | None built in. Skip it and say so |\n| Static hosting without functions, not on Vercel | Only with a Worker in front | Only with a Worker in front |\n\nA request both a drain and the edge package report is counted once.\n\nRequests to the project's ignored routes are counted and not kept. Every project starts with `/api`, because an app calling its own API (session checks, polling, presence) is not a crawler or an agent. Add the product's other API paths, such as `/trpc` or `/graphql`:\n\n```sh\nlittlefriend projects update --add-ignored-route /trpc\n```\n\nRemove `/api` only when the owner wants to watch who calls the API, and say so in the report.\n\n### A Vercel log drain\n\n```sh\nlittlefriend drains list\nlittlefriend drains create --provider vercel\n```\n\n- Check `drains list` first. Reuse an active drain.\n- `drains create` prints the endpoint, the header and the signing secret, once. Create it only when you can paste them into Vercel in the same sitting: in the team's settings, add a log drain for this project only, all sources, production, and 100% sampling. Drains need a Vercel Pro or Enterprise team.\n- If you cannot reach Vercel's settings, do not create the drain. List it under \"Needs a person\": they can do both halves on the dashboard's Agents page, with Add a Vercel log drain.\n- A lost secret: `littlefriend drains rotate <id>` issues a new pair and keeps the old one working for 24 hours.\n- Within minutes of traffic, the Coverage card on the Agents page shows the drain as Live.\n\n### `@littlefriend/edge` and the door\n\nUse this when the product has code that runs on every request. Install `@littlefriend/edge` 0.3.0 or later. The door came in 0.2.0, and 0.3.0 stops sending requests to ignored routes. If the repo already depends on `^0.2.0`, change it to `^0.3.0`: that range never picks up 0.3.0 on its own.\n\n1. Create an edge key into the env file, as in step 12:\n\n```sh\nlittlefriend keys create --kind edge --label \"acme edge\" --write-env .env.local --env-name LF_EDGE_KEY\n```\n\n2. Install `@littlefriend/edge` and wire it for the runtime in the table above. The exact code for each runtime is on https://littlefriend.io/docs/agents (sections Any host and The door). For Next.js 16:\n\n```ts\n// proxy.ts\nimport { createDoor, createEdge, nextProxy } from '@littlefriend/edge';\nimport { NextResponse } from 'next/server';\n\nconst lf = { key: process.env.LF_EDGE_KEY!, ipHeader: 'x-forwarded-for' };\nexport const proxy = nextProxy(createDoor(lf), NextResponse, createEdge(lf));\nexport const config = { matcher: ['/((?!_next/static|_next/image|favicon.ico).*)'] };\n```\n\n3. Set `ipHeader` to the header the host sets to the client address: `x-forwarded-for` on Vercel, `cf-connecting-ip` on Cloudflare, `fly-client-ip` on Fly. Without it, no address is sent and operator address checks cannot run.\n\n4. Pick a preset and keep dry run:\n\n```sh\nlittlefriend door set --preset verified_only --mode dry_run\nlittlefriend door simulate --ua \"Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; GPTBot/1.2; +https://openai.com/gptbot)\" --path /blog\n```\n\n| Product | Preset |\n|---|---|\n| A store with `/checkout`, `/cart` or `/account` | `commerce` |\n| An app or API with sign-in | `verified_only` |\n| Content, docs or marketing | `no_training` |\n| Unsure | `open` |\n\n- Dry run decides every request and records what it would have done. It never changes a response.\n- **Never switch the door to live.** The owner does that after reading 7 days of the door's report on the Agents page.\n- A browser always gets in, whatever the rules say.\n- `commerce` matches `/checkout`, `/cart` and `/account`. If the store uses other paths, say so in the report.\n\n## 14. Hybrid and native apps\n\n**Hybrid apps** (Capacitor, Ionic, Cordova, Tauri, Electron, React Native WebView, or the app's own WKWebView or Android WebView) show web pages. Install the script in the app's web code as in step 5, then add the origin the web view sends to the project's allowed origins:\n\n```sh\nlittlefriend projects update --add-origin capacitor://localhost --add-origin https://localhost\n```\n\n| App shell | Origin to add |\n|---|---|\n| Capacitor on iOS | `capacitor://localhost` |\n| Capacitor on Android | `https://localhost` |\n| Ionic with its Cordova web view | `ionic://localhost` on iOS, `http://localhost` on Android |\n| Cordova | `app://localhost` on iOS (scheme preference set to `app`), `https://localhost` on Android |\n| Tauri | `tauri://localhost` on macOS, iOS and Linux. `http://tauri.localhost` on Windows and Android, or `https://tauri.localhost` with `useHttpsScheme` |\n| Electron | The scheme and host the app serves from, such as `app://myapp` |\n| React Native WebView | The origin of the site it loads |\n| WKWebView with a custom scheme handler | The scheme and host you register, such as `app://localhost` |\n| WKWebView loading a file URL | No origin, see below |\n| Android WebView with WebViewAssetLoader | `https://appassets.androidplatform.net`, or `https://localhost` with `setDomain(\"localhost\")` |\n| Android WebView loading `file:///android_asset` | No origin, see below |\n\n- If the shell's config changes the scheme or host, add the ones it sets.\n- A page loaded from a file has no origin: an Android WebView on `file:///android_asset` sends `Origin: null`, a WKWebView that opened it with `loadFileURL` sends `Origin: null` from `fetch` and `Origin: file://` from `sendBeacon`, and an Electron app that loads its pages from files sends no `Origin` header at all. No list can name any of those, so leave the list empty, or serve the app from a scheme (a `WKURLSchemeHandler`, or `electron-serve`, which serves `app://-`) or from the asset loader. The script sends only the page's file name as the route, such as `/index.html`, so the folders above it, which can hold the user's name, never leave the device; the collector keeps the same rule for scripts older than 0.1.2, and keeps a route a current script names, such as `/settings/profile`, whole.\n- A file page cannot `pushState` to a new path: engines allow only a new query or fragment there. A hash-routed app names its routes itself. Name the starting route before the script loads, in the inline queue stub, as the template your router matches, such as `lf('route', '/orders/:id')`, or as the route part of the hash alone, `lf('route', '/' + location.hash.replace(/^#\\/?/, '').split(/[?&=]/)[0])`, so the load's own page view carries it. Never pass the whole fragment, since a fragment can carry a query or a sign-in token. The script cuts a named route at the first `?` or `#`, redacts a segment that holds `=` and replaces id-like segments, but a route should hold no values to begin with. After each later navigation, once the hash has changed, call `lf('route', '/settings')`, then `lf('page')`. Call `lf('page')` for navigations only: a router hook that also fires for the starting screen must skip that first call, or the screen counts twice. An app that routes with `pushState` gets its page views on its own and never calls `lf('page')`.\n- Apps on iOS send no referrer. A link opened from an Android app in Chrome arrives as `android-app://<package>/`, and a known app is stored under its web host, such as `mail.google.com` for Gmail, in that app's channel. The package table is under Webmail and email apps in `docs/METRICS.md`, published at https://littlefriend.io/docs/metrics#channels.\n- Events queued when the app goes to the background are sent as the page is hidden. An app killed while on screen (a crash, a force stop) loses what was queued in its last 5 seconds, and a kill and relaunch starts a new session.\n- UTM tags on a deep link are read like any landing: pass the link's query on to the page URL (`index.html?utm_source=newsletter&utm_medium=email`) and the visit lands in that channel with those values.\n- For local testing, an https asset-loader or Capacitor page cannot reach an http collector except at `localhost`: on the Android emulator, run `adb reverse tcp:<port> tcp:<port>` and point `data-api` at `http://localhost:<port>`. The real collector is https and needs neither.\n- If the repo deliberately skips analytics inside the shell, keep that and say so in the report.\n\n### Native apps\n\nScreens drawn natively with SwiftUI, UIKit or AppKit send their own screens and events through the Swift SDK. The full guide is https://littlefriend.io/docs/ios.\n\n1. **Install the SDK.** Add the Swift package that https://littlefriend.io/docs/ios#install names, from `0.1.0`, and link its `LittleFriend` product.\n2. **Start it with the project's site key**, the public `lf_` key the script uses, once as the app launches, in the mode chosen in step 4. Leave the options out for aggregate mode.\n\n ```swift\n // The SwiftUI App's init, or application(_:didFinishLaunchingWithOptions:)\n var options = LittleFriend.Options()\n options.mode = .journey\n LittleFriend.start(key: \"<site key>\", options: options)\n ```\n\n With consent required, set the consent option to `.required`, and call `LittleFriend.consent(.journey)` when the person agrees, as step 4 describes for the script.\n3. **Add the app id to Allowed apps**: the app's bundle id. A project admits only the apps it lists. Until the app is listed, its batches are refused with `403 app_not_allowed`, and the dashboard's data health strip names it.\n\n ```sh\n littlefriend apps add com.acme.shop\n ```\n\n4. **Name the screens** with route templates, never values, as on the web. SwiftUI: `.lfScreen(\"/products/:id\", type: \"product\")` on each screen's view. Anywhere else: `LittleFriend.screen(\"/orders/:id\", type: \"order\")`. A UIKit app can set `autoScreens` and give each view controller a route with `LFScreen` and `lfRoute`.\n5. **Track the events the plan names**, with the same names and properties as the web, so the goals and funnels of steps 8 and 9 count both: `LittleFriend.track(\"signup\", props: [\"plan\": .string(\"pro\")])`.\n6. **Call `LittleFriend.flush()` after a correlated event.** In journey mode, give the app's event the correlation id the server sends with the confirmed outcome (step 12), `correlationId: hash`, and call `LittleFriend.flush()` right after it, so it goes out at once. When the app's event arrives first, the server's outcome counts under the app's platform.\n7. **Verify.** Open a screen in the app, then run `littlefriend verify --source browser --since 10 --wait 120`: app batches count under the browser source. The iOS Simulator sends test traffic, which verify shows as test traffic and reports never store. A release build on a device sends real traffic.\n\nNative Android screens, and Flutter and React Native views, send the outcomes they lead to from the backend with `@littlefriend/node` (step 12). The server key never goes inside the app.\n\n## 15. Verify\n\n1. Run the product: the local dev server (its origin is on the list from step 4), or the deployed site.\n2. Load the site in a real browser (Playwright, agent-browser, or a person). `curl` does not run the script. Headless Chrome says so in its user agent and is counted as automation, so its visits stay out of page views and funnels, which count people. For the journey walks in item 9, run the browser headed or give it a desktop Chrome user agent. Then run `littlefriend verify --source browser --since 10 --wait 120`. It exits 0 once an accepted event from the browser in the last 10 minutes shows up, and prints its time, route and source. `--source browser` matters on a project with a log drain or the edge SDK, whose events would otherwise pass for the tag's. It exits 1 when none arrives in time. On a project that already has traffic, keep `--since` short, so an older event cannot pass for yours.\n3. In the browser's network panel, `POST https://in.littlefriend.io/v1/e` answers `202` with `{\"accepted\": n, \"dropped\": 0}`. The console shows no CSP errors.\n4. Click through each flow with a named event and confirm each name in the request bodies. Any `dropped` above 0 means a name or property broke a rule. The live install check (Settings, Install) shows every field kept, every property dropped, and why.\n5. Replay: in the browser's network panel, `GET https://in.littlefriend.io/v1/r/config?k=<site key>` returns `\"on\": true`, and `POST /v1/r` answers `202`. From `curl`, send the page's origin with `-H 'Origin: https://<domain>'`: without an allowed origin the answer is `\"on\": false`. Then `littlefriend replay get` reads installed.\n6. Server events: run the flow that sends one. `verify` shows source `server`, or `await lf.flush()` resolves with `accepted: 1`.\n7. Agents: `littlefriend door get` shows the policy and its mode. On the dashboard's Agents page, the Coverage card shows the drain as Live, and the door page shows that the door has read its rules.\n8. A few minutes later: `littlefriend report overview` and `littlefriend report goals` for a sanity check.\n9. Walk each journey in your plan once, in one tab, step by step, on a page without `data-test` (test events stay out of reports). A few minutes later, `littlefriend funnels report fnl_XXXXXXXX` shows that session at every step. A step at 0 means its route or event name does not match what arrives: compare it with `littlefriend report pages` and the request bodies.\n\nVerify on production after the deploy too, without `data-test`.\n\n| Symptom | Cause |\n|---|---|\n| No request to `in.littlefriend.io` | CSP blocks the script, or the tag never rendered |\n| `403` `origin_not_allowed` | The page's origin is not in allowed origins |\n| `400` | Wrong or malformed site key, or a malformed batch |\n| Sessions unavailable, funnels empty | The project or the snippet is not in journey mode. Both must be |\n| Replay `403` `replay_off`, `session_in_aggregate_mode` or `route_excluded` | Replay is off, the project is in aggregate mode, or the page is excluded |\n| Replay `402` `replay_card_required` | The workspace used its free recordings this month and has no card on file |\n| Replay `429` `quota_exceeded` | The project stored its 1 GiB of recordings for the day, or one address used its share of that. It clears at midnight UTC. If it happens often, lower the sample rate |\n| Server events `rejected` in `onError` | Wrong kind of key, or a revoked one |\n\n## 16. Launch checklist\n\n- [ ] One project for the domain, in the right workspace, in the right mode.\n- [ ] Allowed origins list every production host, every app origin and the dev origin.\n- [ ] The script is on every page, once, with `data-test` outside production.\n- [ ] CSP updated, including report-only copies and header tests.\n- [ ] Every goal and funnel step has its own named event. No personal data in names, properties or routes.\n- [ ] The journey plan is written down, with one goal per outcome.\n- [ ] 2 to 5 goals and, in journey mode, 1 to 3 funnels saved, one per journey in the plan.\n- [ ] Each journey was walked once and shows at every step of its funnel report.\n- [ ] `identify` sends an opaque id, and `null` on sign-out (replay only).\n- [ ] Replay: shown regions checked for names, private routes excluded, widgets blocked.\n- [ ] Server key in an ignored env file and in the host's secrets. Serverless code flushes.\n- [ ] Door in dry run where the runtime allows it. Drain connected on Vercel.\n- [ ] Ignored routes cover the product's own API paths.\n- [ ] Privacy page text added.\n- [ ] `littlefriend verify --source browser` exits 0 on production, and batches show `\"dropped\": 0`.\n- [ ] No secret in your changes: `git diff <base branch> | grep -E 'lf[ser]_[A-Za-z0-9]'` prints nothing.\n- [ ] Report sent.\n\n## 17. Report back\n\nEnd with this report, filled in. Use \"skipped\" with a reason where a step did not apply. Never include a secret.\n\n```markdown\n## Little Friend setup: <product>\n\n- Repo and branch:\n- Domain and hosts:\n- Workspace:\n- Project id and site key (lf_):\n- Mode and consent:\n- Allowed origins:\n- Snippet: --framework <name>, in <file>\n- CSP: <files changed, or none needed>\n- Journey plan: <the table from step 3: journey, outcome goal, steps and how each is seen>\n- Named events: <name, where it fires>\n- Goals: <id, name, match>\n- Funnels: <id, name, steps>\n- Identify: <where it is called, what id it sends, or skipped>\n- Replay: <on or off, rate, shown, hidden, excluded routes>\n- Server events: <events, files, env var names, host secrets set or not>\n- Agents: <drain id and status, door runtime, preset and mode, or skipped>\n- Ignored routes: <the list, and any change from /api>\n- Hybrid or native: <origins added, apps listed and SDKs started, backend events, or none>\n- Privacy page: <updated, or no privacy page>\n- Verify: <time, route and source that verify printed, local and production>\n- Skipped, and why:\n- Needs a person: <drains to connect, secrets to set, questions>\n- Guide or CLI problems found:\n```\n", bundled: true } : { text: GUIDE_PLACEHOLDER, bundled: false };
|
|
3707
3804
|
}
|
|
3708
3805
|
var guide = {
|
|
3709
3806
|
path: ["guide"],
|
|
@@ -4193,7 +4290,7 @@ ${r.unavailableReason}` : null,
|
|
|
4193
4290
|
s.enabled && r.available && !r.installed ? "\nAdd lf-replay.js next to lf.js to start recording: littlefriend snippet --replay" : null
|
|
4194
4291
|
);
|
|
4195
4292
|
}
|
|
4196
|
-
async function
|
|
4293
|
+
async function save2(ctx, body, done) {
|
|
4197
4294
|
const project = await resolveProject(ctx);
|
|
4198
4295
|
const res = await ctx.api().put(projectPath(project, "/replay/settings"), body);
|
|
4199
4296
|
if (!res.available && res.unavailableReason && res.settings.enabled)
|
|
@@ -4219,7 +4316,7 @@ var replayEnable = {
|
|
|
4219
4316
|
options: { rate: { type: "string" } },
|
|
4220
4317
|
async run(ctx) {
|
|
4221
4318
|
const rate = intOpt(ctx, "rate", 1, 100);
|
|
4222
|
-
return
|
|
4319
|
+
return save2(ctx, { enabled: true, ...rate !== void 0 ? { sampleRate: rate } : {} }, "Replay is on.");
|
|
4223
4320
|
}
|
|
4224
4321
|
};
|
|
4225
4322
|
var replayDisable = {
|
|
@@ -4228,7 +4325,7 @@ var replayDisable = {
|
|
|
4228
4325
|
usage: "",
|
|
4229
4326
|
example: "littlefriend replay disable",
|
|
4230
4327
|
async run(ctx) {
|
|
4231
|
-
return
|
|
4328
|
+
return save2(ctx, { enabled: false }, "Replay is off. Pages with lf-replay.js record nothing.");
|
|
4232
4329
|
}
|
|
4233
4330
|
};
|
|
4234
4331
|
var LISTS = [
|
|
@@ -4300,6 +4397,12 @@ var replayCommands = [replayGet, replayEnable, replayDisable, replaySet];
|
|
|
4300
4397
|
// src/commands/report.ts
|
|
4301
4398
|
var REPORTS = ["overview", "pages", "sources", "goals", "countries"];
|
|
4302
4399
|
var TOP = 10;
|
|
4400
|
+
var PLATFORM_NAMES = {
|
|
4401
|
+
web: "Web",
|
|
4402
|
+
ios: "iOS",
|
|
4403
|
+
macos: "macOS",
|
|
4404
|
+
android: "Android"
|
|
4405
|
+
};
|
|
4303
4406
|
function metricValue(m) {
|
|
4304
4407
|
if (!m.available || m.value === null) return "unavailable";
|
|
4305
4408
|
if (m.unit === "percent") return percent(m.value);
|
|
@@ -4327,10 +4430,11 @@ function breakdownTable(title, b) {
|
|
|
4327
4430
|
])
|
|
4328
4431
|
);
|
|
4329
4432
|
}
|
|
4330
|
-
function header(name, meta) {
|
|
4433
|
+
function header(name, meta, platform) {
|
|
4331
4434
|
const warn = meta.caveats.filter((c) => c.severity === "warn").map((c) => `Note: ${c.message}`);
|
|
4435
|
+
const only = platform ? `, ${PLATFORM_NAMES[platform]} only` : "";
|
|
4332
4436
|
return lines(
|
|
4333
|
-
`${name}, ${meta.range.from} to ${meta.range.to} (${meta.timezone}), ${meta.mode} mode.`,
|
|
4437
|
+
`${name}, ${meta.range.from} to ${meta.range.to} (${meta.timezone}), ${meta.mode} mode${only}.`,
|
|
4334
4438
|
...warn,
|
|
4335
4439
|
meta.freshness.backlogBatches > 0 ? `Note: ${count(meta.freshness.backlogBatches)} batches are still being processed, so the newest events may be missing.` : null
|
|
4336
4440
|
);
|
|
@@ -4338,16 +4442,17 @@ function header(name, meta) {
|
|
|
4338
4442
|
var report = {
|
|
4339
4443
|
path: ["report"],
|
|
4340
4444
|
summary: "Quick numbers for a sanity check",
|
|
4341
|
-
usage: `[${REPORTS.join("|")}] [--from <date>] [--to <date>]`,
|
|
4445
|
+
usage: `[${REPORTS.join("|")}] [--from <date>] [--to <date>] [--platform ${PLATFORMS.join("|")}]`,
|
|
4342
4446
|
example: "littlefriend report sources --from 2026-09-01 --to 2026-09-28",
|
|
4343
4447
|
args: { min: 0, max: 1 },
|
|
4344
|
-
options: RANGE_OPTIONS,
|
|
4448
|
+
options: { ...RANGE_OPTIONS, platform: { type: "string" } },
|
|
4345
4449
|
async run(ctx) {
|
|
4346
4450
|
const which = ctx.args[0] ?? "overview";
|
|
4347
4451
|
if (!REPORTS.includes(which)) throw usageError(`The report is ${orList(REPORTS)}. Got ${ctx.args[0]}.`);
|
|
4452
|
+
const platform = oneOf(ctx, "platform", PLATFORMS);
|
|
4348
4453
|
const project = await resolveProject(ctx);
|
|
4349
4454
|
const range = dateRange(ctx, project.timezone);
|
|
4350
|
-
const query = { from: range.from, to: range.to, compare: "none" };
|
|
4455
|
+
const query = { from: range.from, to: range.to, compare: "none", ...platform ? { platform } : {} };
|
|
4351
4456
|
const api = ctx.api();
|
|
4352
4457
|
switch (which) {
|
|
4353
4458
|
case "overview": {
|
|
@@ -4355,7 +4460,7 @@ var report = {
|
|
|
4355
4460
|
return {
|
|
4356
4461
|
data: r,
|
|
4357
4462
|
text: lines(
|
|
4358
|
-
header(project.name, r.meta),
|
|
4463
|
+
header(project.name, r.meta, platform),
|
|
4359
4464
|
"",
|
|
4360
4465
|
metricsTable(r.metrics),
|
|
4361
4466
|
"",
|
|
@@ -4370,7 +4475,7 @@ var report = {
|
|
|
4370
4475
|
return {
|
|
4371
4476
|
data: r,
|
|
4372
4477
|
text: lines(
|
|
4373
|
-
header(project.name, r.meta),
|
|
4478
|
+
header(project.name, r.meta, platform),
|
|
4374
4479
|
"",
|
|
4375
4480
|
breakdownTable("Page", r.pages),
|
|
4376
4481
|
"",
|
|
@@ -4383,7 +4488,7 @@ var report = {
|
|
|
4383
4488
|
return {
|
|
4384
4489
|
data: r,
|
|
4385
4490
|
text: lines(
|
|
4386
|
-
header(project.name, r.meta),
|
|
4491
|
+
header(project.name, r.meta, platform),
|
|
4387
4492
|
"",
|
|
4388
4493
|
breakdownTable("Channel", r.channels),
|
|
4389
4494
|
"",
|
|
@@ -4398,7 +4503,7 @@ var report = {
|
|
|
4398
4503
|
return {
|
|
4399
4504
|
data: r,
|
|
4400
4505
|
text: lines(
|
|
4401
|
-
header(project.name, r.meta),
|
|
4506
|
+
header(project.name, r.meta, platform),
|
|
4402
4507
|
"",
|
|
4403
4508
|
r.goals.length > 0 ? table(
|
|
4404
4509
|
["GOAL", "COMPLETIONS", "CONVERSION", "SERVER-CONFIRMED"],
|
|
@@ -4419,7 +4524,7 @@ var report = {
|
|
|
4419
4524
|
});
|
|
4420
4525
|
return {
|
|
4421
4526
|
data: r,
|
|
4422
|
-
text: lines(header(project.name, r.meta), "", breakdownTable("Country", r.breakdown))
|
|
4527
|
+
text: lines(header(project.name, r.meta, platform), "", breakdownTable("Country", r.breakdown))
|
|
4423
4528
|
};
|
|
4424
4529
|
}
|
|
4425
4530
|
}
|
|
@@ -5032,6 +5137,7 @@ var COMMANDS = [
|
|
|
5032
5137
|
...doorCommands,
|
|
5033
5138
|
...drainCommands,
|
|
5034
5139
|
...verifyCommands,
|
|
5140
|
+
...appsCommands,
|
|
5035
5141
|
...reportCommands,
|
|
5036
5142
|
...guideCommands
|
|
5037
5143
|
];
|
|
@@ -5053,7 +5159,7 @@ var GLOBAL_HELP = `Global flags
|
|
|
5053
5159
|
Exit codes: 0 done, 1 failed or nothing arrived, 2 wrong flags, 3 sign in needed.`;
|
|
5054
5160
|
var SECTIONS = [
|
|
5055
5161
|
{ title: "Sign in", groups: ["login", "logout", "whoami"] },
|
|
5056
|
-
{ title: "Workspaces and projects", groups: ["workspaces", "projects"] },
|
|
5162
|
+
{ title: "Workspaces and projects", groups: ["workspaces", "projects", "apps"] },
|
|
5057
5163
|
{ title: "Install", groups: ["snippet", "keys", "verify"] },
|
|
5058
5164
|
{ title: "Measure", groups: ["goals", "funnels", "report"] },
|
|
5059
5165
|
{ title: "Session replay, the agent door and log drains", groups: ["replay", "door", "drains"] },
|