mcp-scraper 0.93.2 → 0.93.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +7 -0
- package/README.md +1 -1
- package/dist/bin/api-server.js +1 -1
- package/dist/bin/mcp-scraper-cli.js +1 -1
- package/dist/bin/mcp-scraper-core.js +1 -1
- package/dist/bin/mcp-scraper-install.js +1 -1
- package/dist/bin/mcp-stdio-server.js +1 -1
- package/dist/bin/paa-harvest.js +1 -1
- package/dist/{chunk-QWSWHUOK.js → chunk-4FGLXXZ2.js} +1 -1
- package/dist/{chunk-CG2FHIBA.js → chunk-BS4V4CVX.js} +1 -1
- package/dist/{chunk-5VB3I7UX.js → chunk-DFJT2YX6.js} +1 -1
- package/dist/{chunk-KPDNM7E7.js → chunk-IK5BG7MO.js} +1 -1
- package/dist/{chunk-MJ2OICVY.js → chunk-M5VZVMPC.js} +1 -1
- package/dist/chunk-SWCAKYBP.js +1 -0
- package/dist/{chunk-EWR7BPPD.js → chunk-XG6GCEUE.js} +1 -1
- package/dist/{chunk-73LAMUS5.js → chunk-ZG52SKGD.js} +4 -4
- package/dist/{extract-bundle-MQOAKQDV.js → extract-bundle-C3N6E6V7.js} +1 -1
- package/dist/index.js +1 -1
- package/dist/{server-GRB6VNT6.js → server-G2ZCSWXW.js} +3 -3
- package/dist/{worker-US6CTFYG.js → worker-P2BICG36.js} +1 -1
- package/package.json +1 -1
- package/dist/chunk-XJ6PJWT5.js +0 -1
- /package/dist/{chunk-VVG7LFFF.js → chunk-M5QHXNFZ.js} +0 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import{a as IE,b as Kh,c as RE,d as CE,e as TE,f as PE,g as xi,h as so,i as vp,j as ei,k as Ee,l as Ep,m as vy,n as Si,o as ao,p as kp,q as Ap,s as ST}from"./chunk-73LAMUS5.js";import{a as Yh,b as qE,d as UE,e as no,f as Zn,g as iy,h as Ak,k as pp,m as mp,n as Ik,o as Rk,p as Ck}from"./chunk-KPDNM7E7.js";import"./chunk-EWR7BPPD.js";import{A as wk,B as _k,D as vk,E as up,G as xk,I as Sk,J as Ek,K as ro,L as rc,M as kk,a as nk,d as ik,e as ok,f as sk,g as ak,h as ck,i as ny,j as Ws,k as lk,m as dk,n as uk,o as _i,q as pk,r as mk,s as gk,t as fk,u as dp,v as hk,w as yk,y as bk,z as Ol}from"./chunk-5JKBFNYF.js";import{d as dy,j as Hk,l as Wk,m as zk}from"./chunk-LSFDB5XE.js";import{A as dA,B as uA,C as pA,D as mA,E as gA,d as Xk,e as Zk,h as Qn,i as Qk,j as eA,k as py,l as tA,m as rA,n as nA,p as Ko,q as iA,r as oA,s as Go,t as sA,u as zs,v as $l,w as aA,x as my,y as cA,z as lA}from"./chunk-X7GZJU5P.js";import{a as BA,b as by,c as HA,d as WA,e as wy,f as _y,i as xr,l as Sp}from"./chunk-NLKA6SHC.js";import{$ as RC,$a as yT,A as dC,Aa as ZC,B as uC,Ba as Xy,C as pC,Ca as QC,D as mC,Da as eT,E as gC,Ea as tT,F as Gy,Fa as rT,G as fC,Ga as Zy,H as hC,Ha as nT,I as Vy,Ia as iT,J as yC,Ja as oT,K as bC,Ka as sT,L as Jy,La as aT,M as wC,Ma as cT,N as _C,Na as Qy,O as vC,Oa as eb,P as xC,Pa as lT,Q as SC,Qa as dT,R as EC,Ra as uT,S as kC,Sa as pT,T as AC,Ta as tb,U as Up,Ua as rb,V as IC,Va as Bp,W as Yo,Wa as mT,X as Vs,Xa as gT,Y as Xo,Ya as sc,Z as Hl,Za as fT,_ as Yy,_a as hT,a as Tk,aa as CC,ab as bT,b as Pk,ba as TC,bb as wT,c as Nk,ca as PC,cb as nb,d as oy,da as NC,db as _T,e as fp,ea as jp,eb as vT,f as sy,fa as LC,g as Y,ga as MC,gb as xT,h as io,ha as OC,i as hp,ia as DC,j as ay,ja as qC,k as Lk,ka as UC,l as Re,la as jC,m as zo,ma as $C,n as CA,na as $p,o as TA,oa as FC,p as _p,pa as BC,q as PA,qa as HC,r as nc,ra as Fp,s as hy,sa as WC,t as tR,ta as zC,u as Ny,ua as KC,v as rR,va as GC,w as Ei,wa as VC,x as nR,xa as JC,y as iR,ya as YC,z as lC,za as XC}from"./chunk-AFFMCO7R.js";import{b as yA,c as bA,d as wA,e as _A,f as vA,g as xA,h as xp,i as yy,j as MA,l as OA,m as DA,n as qA,o as UA,p as jA,q as $A,r as Rn,s as Vo,t as FA}from"./chunk-IJF4VXJ7.js";import{$ as Mp,$a as aC,A as yR,Aa as WR,B as bR,C as wR,Ca as Wy,D as _R,Da as zR,E as vR,Ea as KR,F as Lp,Fa as GR,G as xR,Ga as qp,H as Uy,Ha as VR,I as SR,Ia as JR,J as ER,Ja as YR,K as kR,Ka as zy,L as AR,Ma as XR,Na as ZR,Oa as QR,P as IR,Pa as eC,Q as RR,Sa as tC,Ta as rC,Ua as nC,V as CR,Va as iC,Wa as oC,Xa as oc,Ya as Ky,Z as TR,Za as sC,_ as jy,a as Ly,aa as Op,b as oR,ba as $y,bb as cC,ca as PR,d as sR,da as NR,e as aR,f as cR,fa as Fy,g as lR,ga as LR,h as dR,ha as By,i as Jo,ia as MR,j as My,ja as Bl,k as Oy,ka as OR,l as uR,la as DR,m as pR,ma as qR,n as Dy,na as UR,o as mR,oa as jR,p as gR,pa as $R,q as J,qa as Dp,r as z,ta as FR,ua as BR,v as ic,w as Gs,x as fR,y as qy,ya as HR,z as hR,za as Hy}from"./chunk-LI7WHOII.js";import{a as gp,b as Dl,c as ql,e as SA,g as EA,h as gy,k as kA,m as AA,n as IA,o as fy,p as RA}from"./chunk-QTDLZTQ7.js";import{a as ln,b as gE,c as Ji,d as Yi,e as tp,f as fE,g as Ve,h as An,i as Us,j as Ja,k as hE,n as kl,o as yE,p as bE,q as rp,r as wE,s as _E,t as vE,u as np,v as wi,w as Wo}from"./chunk-MJ2OICVY.js";import{f as Gr}from"./chunk-VVG7LFFF.js";import{$ as vI,Aa as KI,B as YA,Ba as Ks,C as XA,Ca as GI,D as ZA,Da as VI,E as QA,Ea as JI,F as eI,Fa as Np,G as tI,Ga as Py,H as rI,Ha as YI,I as nI,Ia as XI,J as iI,K as oI,Ka as ZI,L as sI,La as QI,M as aI,Ma as eR,N as cI,O as lI,P as dI,Q as uI,R as pI,S as mI,T as Ey,U as gI,V as fI,W as hI,X as yI,Y as bI,Z as wI,_ as _I,a as dn,aa as xI,b as dr,ba as SI,c as Jn,ca as EI,d as Gh,da as kI,e as yp,ea as AI,f as qk,fa as II,g as Uk,ga as RI,h as oo,ha as CI,i as bp,j as jl,ja as TI,k as vi,ka as PI,l as jk,la as Rp,m as $k,ma as NI,n as Kk,na as ky,o as Gk,oa as Cp,p as uy,pa as LI,q as Vk,qa as MI,r as Jk,ra as OI,s as Yk,sa as DI,t as wp,ta as Ay,u as Fl,ua as Iy,v as fA,va as Ty,w as hA,wa as WI,x as NA,xa as zI,y as LA,za as Pp}from"./chunk-QWSWHUOK.js";import{b as xE,g as Hh,i as NE,j as Zi,k as Al}from"./chunk-DT2FYN6N.js";import{a as OE,b as DE,d as Vh,f as Jh}from"./chunk-W2BVJ7S2.js";import{a as $t,b as vt,c as Pt,d as un,e as pn,f as sp,g as Xa}from"./chunk-RK2VCTZI.js";import{b as Ya,c as js,d as ip,e as op,f as $s,h as jt,i as SE,j as EE,k as At,l as Wh,m as kE,n as Xi,o as AE,p as zh,q as Fs}from"./chunk-ORB4RHCK.js";import{a as to,b as Vr,d as Ml,e as rk,f as Xn}from"./chunk-TMB56NCA.js";import{c as cy,e as Mk,g as Ok,h as Dk}from"./chunk-HUV2WTRW.js";import{a as qI,b as UI,c as ti,f as jI,h as $I,j as FI,k as Ry,l as Cy,m as BI,n as Tp,q as HI}from"./chunk-YGBTTW5D.js";import{$ as lp,A as Zh,C as BE,D as Qh,E as Za,F as HE,G as WE,H as zE,I as KE,J as ap,K as ey,L as ty,M as ry,N as eo,O as GE,P as VE,Q as JE,R as Qa,S as YE,T as XE,U as Nl,V as cp,X as ZE,Y as Ll,_ as ce,a as D,aa as ec,b as Il,ba as N,c as jE,ca as QE,d as Hs,da as ek,e as rt,ea as tk,fa as Yn,g as Rl,ga as In,ha as tc,m as Qi,n as Xh,o as Cl,p as vr,q as Tl,r as Pl,s as $E,v as FE}from"./chunk-UN6FVDJQ.js";import{a as zu,b as ae,c as Ku,d as Gu}from"./chunk-SH5KB4P7.js";import{a as tr,b as qt,d as Ut,e as ep,f as mE,g as Vn,h as kn}from"./chunk-2SP57VCG.js";import{b as ve,c as we}from"./chunk-6DTXIZY2.js";import{a as Ul,b as zA,f as KA,g as Ip,h as xy,i as Sy,j as GA,k as VA}from"./chunk-W2T4NTCE.js";import{a as Fk,b as ly,c as Bk}from"./chunk-KJQXUZ4Y.js";import{a as Kn,b as Wu,d as iS,e as oS,f as sS,g as aS,h as cS,i as lS}from"./chunk-Y6MKMSOC.js";import{a as ME}from"./chunk-WO3N5FH2.js";import{a as LE,b as fe,c as Bs}from"./chunk-HE45FFBU.js";import{a as JA}from"./chunk-XJ6PJWT5.js";import{$ as _S,$a as GS,$b as cE,A as uS,Aa as NS,Ab as Bh,B as pS,Ba as LS,C as mS,Ca as MS,D as gS,Da as OS,E as fS,Ea as DS,Eb as eE,Fa as Yu,Fb as tE,G as hS,Ga as Gn,Hb as rE,Ia as qS,Ja as Oh,Ka as Dh,La as qh,Lb as nE,Ma as US,Na as jS,Nb as ie,Oa as $S,Ob as iE,Pa as FS,Pb as Zu,Q as Ju,Qa as Xu,R as gt,Ra as BS,Rb as oe,S as Bo,Sa as Uh,Sb as _t,T as tt,Ta as HS,Tb as ze,Ub as Qu,Vb as oE,W as yS,Wa as WS,X as bS,Xb as Fr,Y as Ih,Ya as zS,Yb as sE,Z as bi,Za as KS,_ as wS,_a as jh,_b as aE,a as Vu,aa as Rh,ab as VS,ac as lE,ba as vS,bb as JS,bc as dE,ca as Ch,cb as YS,cc as uE,da as Th,db as XS,dc as pE,fa as Ph,fb as ZS,ga as xS,gb as QS,h as $r,ha as Nh,hb as Sl,i as xl,j as lr,jb as _r,ka as SS,l as Ds,lb as $h,ma as ES,na as kS,oa as Ho,pa as Vt,qa as AS,ra as IS,sa as Va,t as R,ta as Lh,ua as RS,va as CS,wa as Z,wb as qs,xa as TS,ya as PS,yb as El,z as dS,za as Mh,zb as Fh}from"./chunk-WJ4XFLS4.js";var Hp=[{slug:"ai-development-workflows",title:"AI Development Workflows: A Complete Guide",titleDisplay:"AI Development",titleDisplayItalic:"Workflows.",description:"What they are, how they're built, which tools actually matter, and the mistakes most teams make before they figure it out.",publishedAt:"2026-05-23",author:"MCP Scraper",authorInitials:"M",tags:["AI","workflows","development","automation"],category:"AI Workflows",badge:"49 questions answered",fieldGuideLabel:"Field guide",deck:"What they are, how they're built, which tools actually matter, and the mistakes most teams make before they figure it out.",readTimeMinutes:12,stats:[{value:"5",label:"sections"},{value:"49",label:"questions"},{value:"4",label:"core stages"},{value:"0",label:"fluff"}],ctaHeading:"Build faster with",ctaHeadingItalic:"real data.",ctaBody:"MCP Scraper gives your AI workflows the web intelligence they need \u2014 SERP data, People Also Ask harvests, page extraction, YouTube transcripts, and more. All via API or MCP server.",sections:[{id:"s1",num:"01",title:"What is an AI",titleItalic:"Workflow?",deck:"The definition most people skip, and why skipping it costs them three months of rework.",callout:{eyebrow:"Quick take",heading:"An AI workflow is not an AI tool.",body:"A tool does one thing. A workflow connects data, model, and action into a loop that runs without you. <strong>Most teams build tools. Winners build workflows.</strong>"},cards:[{id:"q1-1",num:"1.1",question:"What is an example of an AI workflow?",answer:"An AI workflow is a connected sequence where an AI model handles one or more steps in a larger process. A concrete example: a model reads an incoming invoice, extracts line items, cross-references them against a purchase order in your database, flags discrepancies, and routes it to the right approver \u2014 all automatically. <strong>The human only touches the exceptions.</strong> That's the leverage. Single-step automations (just a prompt, just a classification) are AI tools. Workflows string those steps together with state and decision logic."},{id:"q1-2",num:"1.2",question:"What is the basic workflow of AI?",answer:"The core components of any AI workflow are: <strong>agents</strong> that perform tasks or make decisions, <strong>data pipelines</strong> that feed those agents, <strong>tool integrations</strong> that let agents take action in external systems, and <strong>feedback loops</strong> that improve outputs over time. Strip away any one of those and you have a prototype, not a workflow. The feedback loop is where most teams cut corners \u2014 and it's the one that compounds."},{id:"q1-3",num:"1.3",question:"What are the four stages of an AI workflow?",answer:"The four stages are: <strong>(1) Data input</strong> \u2014 structured or unstructured data enters the system. <strong>(2) Processing and analysis</strong> \u2014 the model interprets, classifies, or extracts. <strong>(3) Decision-making</strong> \u2014 based on the model's output, the workflow branches. <strong>(4) Output with feedback</strong> \u2014 an action is taken and the result is logged so future runs improve. Most implementations nail stages 1\u20133 and forget 4. That's why they plateau."},{id:"q1-4",num:"1.4",question:"What are the four types of workflows?",answer:"<strong>Sequential</strong> \u2014 tasks run in a fixed order. Good for predictable processes. <strong>Parallel</strong> \u2014 multiple tasks run simultaneously. Good for speed. <strong>State machine</strong> \u2014 waits for an event to transition. Good for long-running or human-in-the-loop processes. <strong>Rules-driven</strong> \u2014 conditional logic branches based on data values. Good for compliance or tiered routing. Most real AI workflows combine two or three of these."},{id:"q1-5",num:"1.5",question:"What is the AI project cycle?",answer:"The AI Project Cycle runs: <strong>problem definition \u2192 data collection \u2192 model selection \u2192 evaluation \u2192 deployment \u2192 monitoring</strong>. Deployment and monitoring take longer than most teams budget \u2014 usually 60% of total project time. Skipping proper problem definition is where 85% of AI projects fail before they start."}]},{id:"s2",num:"02",title:"Stages of",titleItalic:"Development.",deck:"How AI systems mature from prototype to production \u2014 and the adoption curve most organizations get stuck on.",cards:[{id:"q2-1",num:"2.1",question:"What are the 5 stages of AI adoption?",answer:"Five stages: <strong>Aware</strong> (experimenting with prompts), <strong>Active</strong> (running pilots), <strong>Operational</strong> (AI is a production dependency), <strong>Systemic</strong> (AI shapes how teams are structured), and <strong>Transformational</strong> (the business model itself changes). Most teams stall at Operational \u2014 they have working AI but it hasn't changed how decisions get made."},{id:"q2-2",num:"2.2",question:"What are the 5 layers of AI development?",answer:"Infrastructure \u2192 Data \u2192 Model development and operations \u2192 Application \u2192 Cross-layer governance. The cross-layer governance piece is the one that gets ignored until there's an incident. When you have AI making decisions in production, you need an audit trail, rollback capability, and ownership assignments at every layer \u2014 before something goes wrong, not after."},{id:"q2-3",num:"2.3",question:"What are the 8 stages of a workflow?",answer:"Creation \u2192 Initiation \u2192 Execution \u2192 Review \u2192 Approval \u2192 Documentation \u2192 Archival \u2192 Iteration. The Iteration stage is where AI adds disproportionate value \u2014 a workflow that can learn from its own execution history will outperform a static one within weeks."},{id:"q2-4",num:"2.4",question:"What are the 4 types of AI?",answer:'<strong>Reactive</strong> (no memory), <strong>Limited memory</strong> (uses recent context \u2014 most modern LLMs), <strong>Theory of mind</strong> (not yet achieved), <strong>Self-aware</strong> (theoretical). Every AI in production today is limited memory. The "agentic AI" hype is largely about making limited-memory systems behave more like theory-of-mind ones through tool use and persistent context.'},{id:"q2-5",num:"2.5",question:"What are the 4 pillars of AI?",answer:"Across frameworks, the consistent pillars are: <strong>Data</strong> (quality and volume), <strong>Compute</strong> (infrastructure and cost), <strong>Algorithms</strong> (model architecture), and <strong>People</strong> (domain expertise to supervise and improve the system). Of these, People is the longest-lead bottleneck. You can rent compute and buy data. You can't rapidly acquire practitioners who know both the domain and the models."}]},{id:"s3",num:"03",title:"Tools &",titleItalic:"Platforms.",deck:"The honest rundown on what's actually useful versus what's just well-funded.",cards:[{id:"q3-1",num:"3.1",question:"What is the best AI workflow tool?",answer:"It depends on where you sit on the complexity curve. <strong>No-code</strong>: Zapier AI, Make, monday.com. <strong>Low-code</strong>: n8n, Pipedream. <strong>Code-first</strong>: LangChain, LlamaIndex, custom Claude/GPT API integrations. No-code gets you to 80% quickly and hits a wall. Code-first has no ceiling but requires engineering time. Most production teams end up hybrid."},{id:"q3-2",num:"3.2",question:"Can ChatGPT create workflows?",answer:"Yes \u2014 ChatGPT can design, describe, and write the code for workflows. It can also be a step inside a workflow via the API. The distinction matters: using ChatGPT to <em>build</em> a workflow is a productivity tool. Using the API as a <em>node</em> in a running workflow is an architectural decision. The latter is where teams underestimate latency and cost at scale."},{id:"q3-3",num:"3.3",question:"Which AI tool is most popular?",answer:"ChatGPT remains the most widely recognized AI tool. But popularity in a consumer context doesn't translate to best-in-class for workflows. <strong>Claude 3.5 Sonnet</strong> is widely considered superior for nuanced writing, coding, and reasoning tasks as of mid-2026. Gemini leads on multimodal and Google Workspace integration. Match the model to the task, not the brand recognition."},{id:"q3-4",num:"3.4",question:"What are the 7 components of AI?",answer:"For an AI agent architecture: <strong>goal definition, perception/input, memory, reasoning/planning, tool use, action execution, and output/feedback</strong>. This maps directly to workflow design. Goal = trigger condition. Perception = data ingestion. Memory = context + retrieval. Reasoning = the model call. Tool use = API integrations. Action = write to DB, send message. Feedback = log outcome for evaluation."}]},{id:"s4",num:"04",title:"Building",titleItalic:"Workflows.",deck:"The practical how \u2014 from blank canvas to something running in production.",callout:{eyebrow:"Before you build",heading:"Map the failure modes first.",body:"Draw the workflow. Then ask: what happens when the model returns garbage? What happens when the API is down? <strong>Every branch that leads to silent failure needs a fallback before you ship.</strong>"},cards:[{id:"q4-1",num:"4.1",question:"How do you generate a workflow using AI?",answer:"Connect a data source (email, form, webhook) \u2192 define what the AI model does with that data (classify, extract, generate) \u2192 wire the output to an action (update a record, send a message, trigger another step). The hard part is writing the prompt that's robust to edge cases \u2014 that takes iteration and logging, not just a clever initial draft."},{id:"q4-2",num:"4.2",question:"How do you develop workflows?",answer:"Define a clear, measurable goal. Map chronological tasks. Assign ownership at each step (human or AI). Select tools. Build the happy path first, then stress-test with edge cases. The most common mistake is building the automation before establishing the baseline metric \u2014 if you don't know your current error rate, you can't prove the AI improved it."},{id:"q4-3",num:"4.3",question:"What are the 4 C's of AI compliance?",answer:"In the context of responsible AI workflow design: <strong>Compliance</strong> (meets regulatory requirements), <strong>Confidence</strong> (can you quantify the model's certainty), <strong>Consistency</strong> (same behavior on similar inputs), and <strong>Clarity</strong> (can you explain the output). These aren't theoretical \u2014 they're the questions an auditor asks when a workflow makes a wrong decision at scale."},{id:"q4-4",num:"4.4",question:"What is L1 L2 L3 in AI workflows?",answer:"<strong>L1</strong> handles routine, rule-based tasks. <strong>L2</strong> handles exceptions with AI-assisted decision-making, escalating to humans when confidence is low. <strong>L3</strong> handles complex, judgment-intensive tasks where AI augments human expertise. Deploy L1 broadly (high ROI, low risk), L2 selectively, and L3 sparingly \u2014 not because L3 isn't valuable but because it requires the most oversight."}]},{id:"s5",num:"05",title:"Challenges &",titleItalic:"Best Practices.",deck:"Why 85% of AI projects fail \u2014 and what the 15% do differently.",cards:[{id:"q5-1",num:"5.1",question:"What is the biggest problem with AI?",answer:"In production workflows, the biggest problem is <strong>lack of transparency</strong> in how models make decisions. When a workflow produces a wrong output, you need to know which step failed and why. Without logging and explainability tooling built in from day one, debugging becomes archaeology. Second biggest: data quality. Models are amplifiers \u2014 they amplify good data into great outputs and bad data into confidently wrong ones."},{id:"q5-2",num:"5.2",question:"Why do 85% of AI projects fail?",answer:'Top reasons: vague problem definition (no measurable success condition), poor data quality, underestimating deployment and monitoring cost, building for the demo rather than the edge case, and lack of domain expertise on the team. Most projects "fail" by not reaching production, not by producing wrong results. Getting to production is an organizational problem more than a technical one.'},{id:"q5-3",num:"5.3",question:"What skills are needed to work in AI workflows?",answer:"Core: prompt engineering, API integration, basic Python or JavaScript, data cleaning fundamentals. Differentiating: systems thinking (understanding how components fail), domain expertise, evaluation methodology (how do you score outputs?), and cost modeling (how do you prevent runaway API spend?). The actual bottleneck in most organizations is people who can scope, build, and evaluate a workflow end to end."},{id:"q5-4",num:"5.4",question:"What do humans have that AI can never have?",answer:"Accountability. A model can produce an output; only a human can own the consequence of acting on it. This is the non-technical moat for human workers in AI-augmented workflows. Design your workflows with explicit human ownership of outcomes, not just human review of outputs."}]}]},{slug:"who-hallucinates-more-chatgpt-or-claude",title:"Who Hallucinates More: ChatGPT or Claude?",titleDisplay:"Who Hallucinates More:",titleDisplayItalic:"ChatGPT or Claude?",description:"Five benchmarks, two competing verdicts, and the one variable every comparison article gets wrong. Know which model to trust before the answer matters.",publishedAt:"2026-05-23",author:"MCP Scraper",authorInitials:"M",tags:["AI hallucination","ChatGPT","Claude","LLM accuracy","AI benchmarks"],category:"AI Accuracy",badge:"32 questions answered",fieldGuideLabel:"Field guide",deck:'Five benchmarks give five different winners \u2014 and every "Claude wins" article was benchmarked on a model that is no longer the default. Here is what the current data actually says, and what to do with it.',readTimeMinutes:14,stats:[{value:"5",label:"sections"},{value:"32",label:"questions answered"},{value:"5",label:"benchmarks compared"},{value:"0",label:"fluff"}],ctaHeading:"Verify before you ship with",ctaHeadingItalic:"live data.",ctaBody:"MCP Scraper gives you real-time SERP intelligence, PAA harvests, and page extraction so your AI workflows are grounded in current sources \u2014 not cached claims from articles written about models that no longer exist.",sections:[{id:"what-is-hallucination",num:"01",title:"What You're Actually Asking",titleItalic:"About.",deck:'Before comparing rates, you need to know what the word "hallucination" means \u2014 and it turns out no benchmark, no article, and no AI company uses the same definition. A 3% rate and a 15% rate can describe the same model on the same day.',callout:{eyebrow:"Terminology",heading:"Hallucination and confabulation are not the same thing \u2014 and the distinction explains why Claude and ChatGPT get different labels.",body:"Confabulation is the specific pattern of plausibly gap-filling missing knowledge with invented detail \u2014 the brain (or model) connecting dots that were never there. Hallucination is the broader term covering any confident false output. <strong>Claude's uncertainty-admission training was designed to interrupt confabulation specifically.</strong> ChatGPT's RLHF was tuned on human preference, which tends to reward confident, complete-sounding answers even when the model is uncertain. The same root behavior gets opposite training signals in each system."},cards:[{id:"q1-1",num:"1.1",question:"what is AI hallucination",answer:`<strong>AI hallucination is when a language model produces confident, fluent output that is factually wrong \u2014 a citation that doesn't exist, a date that never happened, a quote no one said.</strong> The term is borrowed loosely from psychiatry, where hallucination means perceiving something that isn't there. In practice, LLM hallucinations look less like delusions and more like plausible-sounding autocomplete: the model generates the statistically likely continuation of a sentence, not a grounded fact lookup. The critical word is "confident" \u2014 hallucinations are dangerous not because models are wrong, but because they are wrong without signaling any uncertainty. A practitioner's real concern is not hallucination frequency but hallucination detectability: a model that hallucinates rarely but never hedges is far more dangerous in production than one that hallucinates often and flags it.`},{id:"q1-2",num:"1.2",question:"why do AI chatbots hallucinate",answer:`<strong>AI chatbots hallucinate because they are trained to predict the most plausible next token, not to retrieve verified facts from a ground-truth database.</strong> The architecture is fundamentally generative \u2014 the model produces text that fits the statistical patterns in its training corpus, and sometimes those patterns lead it to fill gaps with invented specifics. Three compounding factors make hallucination worse: sparse coverage of a topic in training data (the model extrapolates), conflicting information in the corpus (the model blends), and RLHF reward signals that favor fluent, complete-sounding outputs over hedged ones (the model stops saying "I'm not sure"). The reason ChatGPT and Claude hallucinate at different rates on different tasks is not architecture alone \u2014 it is which of these three failure modes each system's training most aggressively corrects for. If your task exposes sparse training coverage (niche domain knowledge, recent events), neither model can save you without grounded retrieval.`},{id:"q1-3",num:"1.3",question:"what is confabulation in AI",answer:`<strong>Confabulation in AI is the specific pattern where a model fills a knowledge gap with invented-but-plausible detail rather than refusing or hedging</strong> \u2014 the model "connects the dots" that were never actually there. The clinical term comes from neurology, where patients with certain memory disorders produce false memories that feel entirely real to them. In LLMs, confabulation is the mechanism behind the most dangerous class of hallucinations: not random nonsense but well-constructed fabrications \u2014 a fake paper with a real author's name, a plausible-sounding legal citation, a drug dosage derived by averaging nearby real figures. The distinction matters for tooling: hallucination detectors that look for low confidence scores will often miss confabulation, because the model's internal confidence on a confabulated output can be high. Grounding against primary sources \u2014 not just asking the model to self-check \u2014 is the only reliable counter.`},{id:"q1-4",num:"1.4",question:"what is the difference between AI hallucination and confabulation",answer:"<strong>Hallucination is the broad category; confabulation is the specific failure mode where the model invents plausible gap-fills rather than flagging its own uncertainty.</strong> All confabulation is hallucination, but not all hallucination is confabulation \u2014 a model that confidently states a wrong date is hallucinating, but it isn't necessarily confabulating if the error traces to a corrupted training example rather than a gap-bridging inference. The distinction changes what interventions work: suppressing confabulation requires training models to recognize the edges of their own knowledge and refuse at those boundaries (which is what Constitutional AI's self-critique loop does for Claude). Suppressing hallucination more broadly requires grounding \u2014 retrieval-augmented generation, citation enforcement, source verification. Practitioners who use the words interchangeably will apply the wrong fix."},{id:"q1-5",num:"1.5",question:"are AI hallucinations the same as lying",answer:`<strong>No \u2014 hallucination is a failure of knowledge, not a failure of intent, which means the usual remedies for dishonesty (adversarial red-teaming, filtering, policy enforcement) don't reduce it.</strong> A lying agent knows the truth and conceals it; a hallucinating model has no ground-truth representation to conceal \u2014 it generates the output that fits the learned distribution, whether that output is accurate or not. This distinction is not just philosophical. Treating hallucination as lying leads organizations to apply trust-and-safety interventions (content moderation, output filtering) rather than epistemic interventions (grounding, uncertainty calibration, retrieval). The more practically damaging confusion is the reverse: treating hallucination as a fixable "bad behavior" that fine-tuning will eventually eliminate, rather than as a structural property of generative models that requires architectural solutions.`},{id:"q1-6",num:"1.6",question:"why does ChatGPT make things up",answer:`<strong>ChatGPT makes things up because its RLHF training consistently rewarded fluent, complete-sounding answers \u2014 and human raters often cannot tell in the moment whether a specific claim is true.</strong> When the model encounters a query at the edge of its training knowledge, it faces two options: produce a hedged, incomplete answer (which RLHF raters historically penalized as unhelpful) or produce a fluent, confident-sounding answer that fills the gap (which raters often rewarded as useful). Over millions of training examples, that signal compounds: the model learns that confident gap-filling is the preferred behavior. OpenAI's release notes for GPT-5.5 Instant specifically cite "reduces hallucination in sensitive areas such as law, medicine, and finance" as a named improvement \u2014 which is an implicit acknowledgment that prior versions were not calibrated to refuse when uncertain. The fix is not better knowledge; it is better uncertainty signaling.`,source:"https://techcrunch.com/2026/05/05/openai-releases-gpt-5-5-instant-a-new-default-model-for-chatgpt/"}]},{id:"the-verdict-depends",num:"02",title:"The Verdict",titleItalic:"Depends.",deck:"One proprietary test shows Claude hallucinating more than ChatGPT (15% vs. 12%). A different benchmark run on the same models the same year shows Claude with the lowest contradiction rate of five providers. Both studies are real. Neither is lying. The winner changes when the measurement changes \u2014 and no competitor article tells you which measurement matches your actual task.",callout:{eyebrow:"Deposition",heading:`Every "Claude wins" verdict was written against a different product than the one you're using today.`,body:'GPT-5.5 Instant became the default ChatGPT in May 2026. Claude Opus 4.7 is the current frontier Claude. <strong>The top SERP articles comparing hallucination rates were benchmarked primarily on GPT-4 Turbo and Claude 3 variants.</strong> The benchmark scores you are reading describe models that are no longer the default. This is not a minor caveat \u2014 task-type inversion, refusal-rate confounds, and methodology differences all compound when the model version gap is also wrong. The deposition question is not "which model wins?" It is: "Which benchmark, on which task type, on which model version, measured how?"'},cards:[{id:"q2-1",num:"2.1",question:"who hallucinates more ChatGPT or Claude",answer:`<strong>Neither model consistently hallucinates more \u2014 the winner changes based on the task type, benchmark methodology, and which model version is being measured.</strong> On BullshitBench v2, Claude Sonnet 4.6 hits a 3% hallucination rate with a 91% detection rate, while OpenAI GPT models are "stuck in the 55\u201365% range" for detection. On Vectara's harder enterprise dataset (February 2026), GPT-4.1 scores 5.6% versus Claude Sonnet 4.6 at 10.6% \u2014 a reversal. On AA-Omniscience, Claude Opus 4.1 achieves 0% hallucination (via refusal), while GPT-5.5 reaches 86% error on the same benchmark. The honest answer for practitioners: Claude tends to outperform on tasks requiring uncertainty calibration and open-recall; ChatGPT tends to outperform on grounded tasks with source material present. Your use case determines the verdict.`,source:"https://medium.com/@anyapi.ai/llm-hallucination-index-2026-why-claude-4-6-7b2d13ed9f0c"},{id:"q2-2",num:"2.2",question:"does Claude hallucinate less than ChatGPT",answer:`<strong>Claude hallucinates less than ChatGPT on open-recall and uncertainty-calibration benchmarks, but GPT models can outperform Claude on grounded generation tasks where source material is provided.</strong> On the Vectara HHEM original dataset (April 2025), GPT-5 scores 1.4% versus Claude-3.7-Sonnet at 4.4% \u2014 ChatGPT wins. On BullshitBench v2, Claude Sonnet 4.6 scores 3% with a 91% detection rate \u2014 Claude wins. On AA-Omniscience, Claude Opus 4.1 achieves 0% hallucination via confident refusal \u2014 Claude wins decisively. The most useful reframe: Claude tends to hallucinate less on tasks where "I don't know" is an acceptable output; ChatGPT can score lower on structured summarization tasks where the source material bounds the answer space.`,source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q2-3",num:"2.3",question:"which AI is more accurate ChatGPT or Claude",answer:"<strong>GPT models score higher on grounded factual accuracy when source material is present; Claude scores higher on calibration \u2014 knowing when not to answer.</strong> On FACTS Overall Scores (grounded generation), GPT-5 scores 61.8 versus Claude Opus 4.5 at 51.3. On AA-Omniscience, Claude Opus 4.1 achieves 0% hallucination while GPT-5.5 reaches 86% error. These are not contradictions \u2014 they measure different things. FACTS rewards producing correct answers given a source; AA-Omniscience rewards refusing answers when knowledge is uncertain. Accuracy in a production system means both: getting the answer right when you have the source, and refusing when you don't. No single model currently dominates both dimensions simultaneously.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q2-4",num:"2.4",question:"what is the hallucination rate of ChatGPT in 2026",answer:`<strong>ChatGPT's hallucination rate in 2026 ranges from 1.4% on Vectara's original RAG benchmark to 86% on AA-Omniscience's domain-knowledge open-recall test \u2014 the same model, different methodologies.</strong> On BullshitBench v2, OpenAI GPT models are "stuck in the 55\u201365% range" for hallucination detection. GPT-5 with thinking mode achieves 1.6% on HealthBench (medical domain). O3 hits 51% hallucination on SimpleQA; o4-mini reaches 79% on PersonQA. The number you see in any article reflects the benchmark used, not a universal accuracy property. The most applicable figure depends on your task: if you are doing RAG summarization, Vectara's 1.4% is relevant; if you are asking ChatGPT to recall domain-specific facts without source material, the AA-Omniscience figure is the honest baseline.`,source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q2-5",num:"2.5",question:"what is the hallucination rate of Claude in 2026",answer:`<strong>Claude's hallucination rate in 2026 spans from 0% (Claude Opus 4.1 on AA-Omniscience, via refusal) to 58% (Claude Opus 4.5 on the same benchmark when not configured to refuse) \u2014 a range that makes any single number misleading.</strong> On BullshitBench v2, Claude Sonnet 4.6 hits 3% with a 91% detection rate, making it the strongest performer in that benchmark class. On Vectara's enterprise dataset (February 2026), Claude Sonnet 4.6 scores 10.6% and Claude Opus 4.6 scores 12.2%. The spread is explained by task type: Claude's Constitutional AI training produces strong refusal behavior on uncertain factual questions, which collapses the hallucination rate on benchmarks that reward "I don't know" responses and inflates it on benchmarks that penalize non-answers.`,source:"https://medium.com/@anyapi.ai/llm-hallucination-index-2026-why-claude-4-6-7b2d13ed9f0c"},{id:"q2-6",num:"2.6",question:"which AI has the lowest hallucination rate in 2026",answer:`<strong>Gemini-2.0-Flash-001 holds the lowest published Vectara HHEM score at 0.7% on the original dataset \u2014 but that benchmark measures factual consistency in RAG summarization, not open-ended recall.</strong> On open-recall benchmarks, Claude Opus 4.1 achieves 0% on AA-Omniscience by refusing uncertain questions, while o3-mini-high scores 0.8% on Vectara. The "lowest hallucination rate" title changes with every benchmark and model release cycle; the more useful question is which model has the lowest hallucination rate on your specific task class. For enterprise RAG pipelines with provided source material, GPT-4.1 at 5.6% on the harder Vectara dataset is currently competitive. For open-domain factual recall with uncertainty, Claude's refusal behavior produces the lowest confirmed error rate.`,source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q2-7",num:"2.7",question:"how is AI hallucination measured",answer:"<strong>AI hallucination is measured by comparing model outputs against a verified ground-truth set and scoring the proportion of confident claims that are factually wrong \u2014 but the ground-truth set, task type, and scoring rules vary so widely across benchmarks that the resulting numbers are rarely comparable.</strong> Three methodology families dominate: RAG consistency tests (Vectara HHEM measures whether a summary stays faithful to the source document), factual recall tests (SimpleQA, PersonQA ask the model open questions with known correct answers), and calibration tests (AA-Omniscience scores how often a model produces wrong answers on questions it should refuse). The same model can score in the top tier on one family and bottom tier on another. Before citing a hallucination rate, the practitioner question is: what task type does this benchmark represent, and does that match what I'm actually asking the model to do?",source:"https://chatgptguide.ai/ai-hallucination-rates-report-gpt-claude-gemini/"}]},{id:"why-claude-behaves-differently",num:"03",title:"Why Claude Behaves Differently",titleItalic:"(And Why That's Complicated.)",deck:`Constitutional AI was built to interrupt confabulation at the output layer \u2014 not to make Claude more knowledgeable, but to make it refuse when it isn't. That design makes Claude's hallucination rate look better on open-recall benchmarks and worse on grounded tasks where refusing an answer is the wrong move. The "safer model" label hides a trade-off every competitor article misses.`,callout:{eyebrow:"Architecture",heading:"Claude's 0% hallucination score on AA-Omniscience is achieved by refusing to answer \u2014 GPT-5.5 attempts the same questions and scores 86% error.",body:`These are not equivalent failure modes. <strong>Claude's refusal behavior is a deliberate uncertainty-admission signal trained by Constitutional AI's self-critique loop.</strong> GPT-5.5's 86% error rate on that benchmark reflects RLHF training that rewards confident, complete-sounding output even under epistemic uncertainty. A practitioner choosing between them for a task where refusal is unacceptable \u2014 a legal brief, a diagnostic intake form, a real-time research summary \u2014 needs to know that "lower hallucination rate" may mean "higher refusal rate," not "more accurate answers."`},cards:[{id:"q3-1",num:"3.1",question:"what is Constitutional AI and does it reduce hallucinations",answer:`<strong>Constitutional AI is Anthropic's training methodology where Claude critiques and revises its own outputs against a set of principles \u2014 and yes, it reduces a specific class of hallucination: confabulation driven by overconfidence.</strong> The core mechanism is a self-critique loop: at training time, Claude is prompted to evaluate its own responses against a constitution of principles (including honesty norms) and revise outputs that violate them. Over millions of examples, this trains the model to flag uncertainty rather than elaborate plausibly over it. What it does not do is give Claude better knowledge \u2014 it makes the model more likely to output "I'm not sure" or refuse at the boundary of its knowledge. The result is measurably lower hallucination rates on open-recall benchmarks and, as a side effect, higher refusal rates on tasks where humans expect confident answers. The "safer model" framing is accurate but incomplete: Constitutional AI reduces dangerous confabulation, not all incorrect outputs.`},{id:"q3-2",num:"3.2",question:"why does Claude hallucinate less than ChatGPT",answer:"<strong>Claude hallucinates less than ChatGPT on uncertainty-sensitive tasks because Constitutional AI's self-critique training penalized overconfident outputs at the architectural level \u2014 not because Claude has better underlying knowledge.</strong> ChatGPT's RLHF training was tuned on human preference ratings, and human raters consistently prefer confident, complete-sounding answers to hedged, partial ones \u2014 even when the hedged answer is more accurate. That preference signal, applied at scale, teaches the model to fill gaps confidently. Constitutional AI's revision loop applies a different signal: outputs that violate honesty norms (overconfident claims under uncertainty) are scored negatively by the model itself and revised. On benchmarks like AA-Omniscience, this difference is dramatic: Claude Opus 4.1 achieves 0% hallucination by refusing uncertain questions; GPT-5.5 attempts the same questions and produces 86% error. The practical implication is that Claude's advantage narrows or reverses when the task context provides source material that bounds the answer \u2014 because grounding reduces the need for uncertainty calibration.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q3-3",num:"3.3",question:"why does Claude say I don't know more than ChatGPT",answer:`<strong>Claude says "I don't know" more than ChatGPT because uncertainty admission was a first-class design goal in Constitutional AI, not an afterthought in RLHF fine-tuning.</strong> Anthropic explicitly trained Claude to identify the edges of its own knowledge and signal them rather than bridge them. The constitutional principle "prefer accurate uncertainty estimates over confident wrong answers" was applied via the self-critique loop \u2014 meaning at training time, Claude learned to score its own overconfident outputs negatively. ChatGPT's training did the opposite: human raters penalized partial or hedged answers as unhelpful, pushing the model toward confident completeness. For practitioners, the implication is that Claude's "I don't know" is a calibration signal worth respecting \u2014 it correlates with genuine knowledge boundaries. ChatGPT's confident answers do not carry the same calibration signal and require independent verification more often.`},{id:"q3-4",num:"3.4",question:"why does Claude admit uncertainty more than ChatGPT",answer:"<strong>Claude admits uncertainty more than ChatGPT because its training reward function directly penalized overconfidence, while ChatGPT's reward function indirectly penalized uncertainty by preferring fluent, complete-sounding outputs.</strong> These are mirror-image training problems with mirror-image results. At the architectural level, both models have the same epistemic limitation: they cannot know what they do not know. What differs is how each handles that edge. Claude's Constitutional AI self-critique loop was designed to surface that edge and output it. ChatGPT's RLHF fine-tuning learned to smooth over it. The consequence for production use is that when Claude expresses uncertainty, it is more likely to be a genuine signal. When ChatGPT expresses certainty, it is less likely to be a reliable signal than the confident phrasing suggests. This asymmetry is the single most important behavioral difference between the two systems for high-stakes use cases."},{id:"q3-5",num:"3.5",question:"does Claude refuse to answer questions it doesn't know",answer:"<strong>Yes \u2014 Claude is trained to refuse or heavily hedge questions at the boundary of its knowledge, and this behavior is measurable: on AA-Omniscience, Claude Opus 4.1 achieves 0% hallucination by refusing uncertain domain-knowledge questions rather than attempting them.</strong> This refusal behavior is not a safety filter applied after generation \u2014 it is a trained output preference baked into the model via Constitutional AI's self-critique loop. The practical consequence is two-sided: Claude produces fewer confident wrong answers than ChatGPT, but it also produces more non-answers on questions where an attempt \u2014 even an imperfect one \u2014 would be useful. For tasks where a partial answer is better than no answer (brainstorming, hypothesis generation, exploratory research), ChatGPT's higher attempt rate is a feature. For tasks where a wrong answer causes real harm (legal, medical, compliance), Claude's refusal behavior is the more defensible default.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"}]},{id:"real-world-consequences",num:"04",title:"When Getting It Wrong",titleItalic:"Has Consequences.",deck:"ChatGPT has already fabricated legal citations in federal court, hallucinated drug dosages, and invented academic papers that passed first-pass review. The more unsettling risk is newer: when ChatGPT, Claude, and Gemini all hallucinate the same false claim, cross-checking them doesn't give you three independent sources. It gives you the same error three times.",callout:{eyebrow:"False Consensus Risk",heading:"If three major LLMs hallucinate the same lie about your business, it can become the new truth \u2014 and no benchmark measures this.",body:"The correlated hallucination problem emerges from shared training data, overlapping RLHF pipelines, and convergent fine-tuning on the same web corpus. <strong>When Claude, ChatGPT, and Gemini all reproduce the same unsupported claim, a practitioner who cross-checks across models gets false triangulation rather than independent verification.</strong> This risk is entirely absent from every current competitor article on hallucination rates \u2014 and it is most acute for entities (companies, people, products) that appear in training data in ways the entity cannot audit or correct."},cards:[{id:"q4-1",num:"4.1",question:"what are examples of ChatGPT hallucinations",answer:"<strong>The most documented ChatGPT hallucinations include fabricated legal case citations presented to federal courts, invented academic papers with real author names, and confident wrong drug dosages in medical queries.</strong> The legal citation failures are the best-documented: in the Mata v. Avianca case, a New York attorney submitted a brief citing multiple cases that ChatGPT invented wholesale \u2014 cases that did not exist anywhere in the legal record. Academic hallucinations are structurally similar: ChatGPT generates plausible-sounding paper titles, journal names, and DOIs that pass casual verification because all the component elements (author names, journal names, topic keywords) are real \u2014 only the assembled paper is fabricated. The pattern in all major cases is the same: ChatGPT produces hallucinations that are specifically designed, by the training distribution, to pass the first verification step a non-expert would apply."},{id:"q4-2",num:"4.2",question:"has ChatGPT hallucinated in court",answer:"<strong>Yes \u2014 the most consequential documented case is Mata v. Avianca, where an attorney used ChatGPT to research case law and submitted a brief citing multiple cases that did not exist, resulting in federal court sanctions.</strong> The cases ChatGPT generated were plausible: they had realistic docket numbers, party names consistent with real aviation litigation, and summaries that read as coherent legal precedent. None of them could be located by opposing counsel or the court because they were entirely fabricated. The attorney was sanctioned for failing to verify the citations. What makes this case significant beyond its notoriety is the mechanism: ChatGPT did not produce random nonsense \u2014 it confabulated, generating outputs that fit the expected form of real legal citations so closely that a trained attorney did not catch them on first read. This is the confabulation failure mode operating at the level of maximum real-world cost."},{id:"q4-3",num:"4.3",question:"did ChatGPT make up fake legal cases",answer:`<strong>Yes \u2014 in the Mata v. Avianca case, ChatGPT generated at least six fake legal cases that were submitted to a federal court as real precedent, making this the first major documented instance of LLM hallucination causing legal sanctions against a practicing attorney.</strong> The fabricated cases had realistic-looking citations: Varghese v. China Southern Airlines, Shaboon v. Egyptair, Zicherman v. Korean Air Lines, and others \u2014 all plausible-sounding aviation negligence precedents. When the court asked for copies of the actual decisions, the attorney could not produce them because they did not exist. The episode became a landmark not just for AI liability but for the broader question of what "verification" means when a model's hallucinations are structurally indistinguishable from real citations to a non-expert reader. Neither Claude nor any other model has a documented comparable case \u2014 but the mechanism exists in all models that generate legal text without grounding.`},{id:"q4-4",num:"4.4",question:"what happened in the Mata v. Avianca ChatGPT hallucination case",answer:"<strong>In Mata v. Avianca, a personal injury lawsuit filed in the Southern District of New York, attorney Steven Schwartz used ChatGPT to research aviation negligence precedents and submitted a court brief citing six cases that ChatGPT had fabricated.</strong> When opposing counsel could not locate the cited cases, the court ordered Schwartz to produce the actual decisions. He could not \u2014 the cases existed only in ChatGPT's output. The court sanctioned Schwartz and his firm. Schwartz's defense was that he was unfamiliar with ChatGPT's tendency to generate false information; the court found that reliance on an AI tool without verification constituted professional negligence. The case is now the canonical example cited in AI liability discussions because it makes concrete the abstract warning that LLM hallucinations have real-world costs \u2014 and because the mechanism was confabulation, not noise: the fake cases were structurally indistinguishable from real ones."},{id:"q4-5",num:"4.5",question:"can ChatGPT hallucinate medical information",answer:"<strong>Yes \u2014 ChatGPT can hallucinate medical information, and the risk is highest in exactly the scenarios where clinicians are most likely to use it: rare conditions, drug-drug interactions, and off-label dosages that are underrepresented in training data.</strong> GPT-4o scores 15.8% hallucination on HealthBench, a medical domain benchmark; GPT-5 with thinking mode reduces this to 1.6%, but the improvement is conditional on the thinking mode being enabled and the query being within the benchmark's scope. The practical risk for medical use is not just the headline hallucination rate \u2014 it is the confabulation pattern: ChatGPT generating specific-sounding dosages or protocol details that are plausible but wrong, in the confident register that clinical notes require. The consensus across medical AI research is that no current LLM should be used as a primary information source for treatment decisions without RAG grounding against validated clinical databases.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q4-6",num:"4.6",question:"does ChatGPT hallucinate more on academic citations",answer:"<strong>ChatGPT hallucinates on academic citations at a notably higher rate than on general-text tasks because academic citations combine several conditions that maximize confabulation risk: sparse training coverage of specific papers, high structural regularity (author, title, journal, year, DOI), and user verification behavior that rarely extends beyond checking the format.</strong> The model has learned that citations follow predictable patterns. When asked for a citation it does not have in training, it generates a citation that fits those patterns \u2014 assembling a plausible author name, a realistic journal, and a plausible year around a fabricated paper. Dedicated academic search integrations (like ChatGPT's search tool when enabled) significantly reduce citation hallucination by grounding against live databases. Without grounding, treating any LLM-generated academic citation as provisional and verifying it against Google Scholar, CrossRef, or a DOI resolver is not optional \u2014 it is the baseline standard of care."}]},{id:"choose-and-verify",num:"05",title:"Choose a Model.",titleItalic:"Verify the Answer.",deck:"The right model for your task depends on whether wrong answers or missing answers cost you more. The right verification method depends on whether you need real-time source grounding, prompt-level controls, or live SERP intelligence to know whether the benchmark you're relying on has already been superseded. Static articles can't give you that. Here's what can.",callout:{eyebrow:"MCP Scraper",heading:"Every hallucination benchmark is already measuring a different model than the one you're running today.",body:"GPT-5.5 Instant and Claude Opus 4.7 are the current defaults as of May 2026. <strong>The top SERP articles were benchmarked on GPT-4 Turbo and Claude 3 variants.</strong> Live PAA intelligence from MCP Scraper shows which benchmark claims are currently circulating in the SERP, which task-specific questions are going unanswered (legal, medical, enterprise, scientific writing), and whether the competitive landscape shifted while the static comparison articles were being written. A practitioner who needs the current answer \u2014 not a cached verdict \u2014 needs a live source, not another article that will be wrong in six months."},cards:[{id:"q5-1",num:"5.1",question:"how do you stop ChatGPT from hallucinating",answer:'<strong>You cannot stop ChatGPT from hallucinating entirely, but the four techniques that reliably reduce it are: retrieval-augmented generation with verified sources, chain-of-thought prompting with explicit uncertainty flagging, citation enforcement in the system prompt, and output verification against primary sources before use.</strong> RAG is the highest-leverage intervention: grounding responses against a controlled, verified document set eliminates the knowledge-gap confabulation that produces most dangerous hallucinations. Chain-of-thought prompting ("explain your reasoning step by step, and flag any step where you are less than certain") forces the model to surface uncertainty it would otherwise paper over. Adding "if you are unsure, say so explicitly rather than guessing" to system prompts has measurable effect on calibration. None of these eliminate the problem \u2014 they reduce it. For high-stakes outputs (legal, medical, financial), human verification against primary sources remains the standard. The tools help; they do not replace verification.'},{id:"q5-2",num:"5.2",question:"how to reduce AI hallucinations with prompt engineering",answer:`<strong>The prompt engineering techniques with the strongest documented effect on hallucination reduction are: explicit uncertainty instructions, step-by-step reasoning requirements, role-scoping, and source-citation enforcement \u2014 applied together, not independently.</strong> Explicit uncertainty instructions ("say 'I don't know' rather than guessing") improve calibration on both Claude and ChatGPT because both models are capable of signaling uncertainty; they need permission to do it. Step-by-step reasoning forces the model to commit to intermediate claims that can be checked, catching confabulation earlier in the chain. Role-scoping ("you are a fact-checker; do not include any claim you cannot source") narrows the output distribution toward the verified. Source citation enforcement ("provide a source URL for every factual claim") creates a verification trail. The compounding insight most engineers miss: these techniques are not additive linearly \u2014 models trained with uncertainty signals (like Claude) show larger improvements from uncertainty prompts than models trained against them (like older ChatGPT versions).`},{id:"q5-3",num:"5.3",question:"does asking ChatGPT to cite sources reduce hallucinations",answer:"<strong>Asking ChatGPT to cite sources reduces the frequency of unverifiable claims in the output, but it does not eliminate fabricated citations \u2014 and a fabricated citation with a plausible URL is harder to catch than a claim with no citation at all.</strong> The mechanism is real: source-citation prompting shifts the model toward outputs where citation is possible, which correlates with better-grounded claims. But ChatGPT can and does generate plausible-looking DOIs, arXiv IDs, and URLs that resolve to nothing. The citation requirement creates a false confidence layer \u2014 the output looks verified when it isn't. The correct workflow is source-citation prompting plus independent verification of every cited source before use. For enterprise pipelines, automated link-checking (do all cited URLs actually resolve?) is the minimum; content verification (does the cited source actually say what the LLM claims it says?) is the standard that eliminates the fabricated-citation failure mode."},{id:"q5-4",num:"5.4",question:"what is retrieval-augmented generation and does it stop hallucinations",answer:"<strong>Retrieval-augmented generation (RAG) is an architecture that grounds LLM outputs by retrieving relevant documents from a verified source set and providing them as context \u2014 and it is currently the most effective single intervention for reducing hallucination in production systems.</strong> Instead of asking a model to recall facts from training, a RAG system retrieves the relevant passage from a controlled document store and asks the model to summarize or synthesize it. On Vectara's HHEM benchmark, which specifically tests this summarization-from-source behavior, even older model versions achieve hallucination rates below 5% \u2014 because the knowledge gap confabulation mechanism is eliminated when the answer exists in the provided context. What RAG does not stop: hallucinations that occur when the retrieved document does not contain the answer and the model interpolates anyway, and hallucinations in the retrieval step itself (if a semantic search retrieves the wrong document). RAG reduces hallucination dramatically in bounded domains; it is not a universal cure for open-domain queries."},{id:"q5-5",num:"5.5",question:"which AI is more trustworthy ChatGPT or Claude",answer:`<strong>Claude is more trustworthy for tasks where calibrated uncertainty is the primary requirement; ChatGPT is more trustworthy for tasks where producing an answer \u2014 even an imperfect one \u2014 is the primary requirement.</strong> Trust is not a single dimension. On calibration trust (does the model's expressed confidence correlate with its actual accuracy?), Claude leads: Constitutional AI's uncertainty training produces hedges that track genuine knowledge gaps. On coverage trust (will the model attempt the question rather than refuse?), ChatGPT leads: RLHF training produces higher attempt rates on hard questions. The user perception data reflects calibration trust: 62% of verified ChatGPT (GPT-4/4o) users report "occasional confident inaccuracies" versus 24% of Claude 3.5 users. For practitioners, the framework is: trust Claude more when a wrong answer is worse than no answer; trust ChatGPT more when a partial answer is better than a refusal.`,source:"https://chatgptguide.ai/ai-hallucination-rates-report-gpt-claude-gemini/"},{id:"q5-6",num:"5.6",question:"which AI is safer to use for high-stakes tasks",answer:`<strong>For high-stakes tasks where a wrong answer has irreversible consequences, Claude's refusal-calibrated behavior makes it the safer default \u2014 but safe use of either model requires human verification against primary sources, not model selection alone.</strong> Claude's design advantage in high-stakes contexts is the refusal signal: when Claude says it is uncertain, that signal has been trained to track real knowledge boundaries. ChatGPT's confident outputs in the same situations are less reliably calibrated. However, "safer model" does not mean "reliable without verification" \u2014 Claude Opus 4.5 scores 58% hallucination on AA-Omniscience when not configured to refuse, and Claude Opus 4.7 scores 36%. The practical standard for high-stakes work is: use Claude for the calibration signal, ground with RAG against verified sources, and treat any AI-generated factual claim in a legal, medical, or financial context as provisional until independently confirmed.`,source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q5-7",num:"5.7",question:"which AI hallucinates less for enterprise use",answer:"<strong>For enterprise RAG pipelines where source documents are provided, GPT-4.1 currently outperforms Claude on Vectara's harder enterprise dataset (5.6% vs. Claude Sonnet 4.6's 10.6%) \u2014 but for enterprise use cases requiring open-domain knowledge retrieval or strict uncertainty signaling, Claude's refusal calibration produces fewer dangerous confident errors.</strong> The enterprise use case is split along the same task-type boundary that governs all hallucination comparisons: grounded generation with provided source material favors GPT models; open-domain recall with uncertainty requirements favors Claude. For enterprise deployments at the highest risk level (legal, medical, compliance), the architecture recommendation from available benchmark data is: pair Claude with a RAG pipeline, use Claude's uncertainty signal as a flag for human review, and verify any output where the model expresses high confidence without a cited source. The combination outperforms either model used alone.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q5-8",num:"5.8",question:"can you trust ChatGPT for medical or legal advice",answer:'<strong>No \u2014 neither ChatGPT nor any current LLM should be trusted as a primary source for medical or legal advice without independent verification against authoritative primary sources, and the documented failure cases make the risk concrete.</strong> In the Mata v. Avianca case, ChatGPT fabricated legal citations that passed initial attorney review, resulting in federal court sanctions \u2014 the most expensive hallucination failure mode documented in legal practice. On HealthBench, GPT-4o hallucinates medical information at a 15.8% rate; GPT-5 with thinking mode reduces this to 1.6%, but that reduction depends on task-specific configurations unavailable in standard ChatGPT use. For legal research, both ChatGPT and Claude should be used as research accelerators \u2014 identifying potentially relevant cases and concepts \u2014 with every specific citation independently verified against Westlaw, LexisNexis, or primary court documents before any professional use. The standard of care is not "use the safer model"; it is "verify every claim."',source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"}]}]},{slug:"claude-code",title:"Claude Code: The Evaluation Guide Every Tutorial Skips",titleDisplay:"Claude Code",titleDisplayItalic:"Evaluated.",description:"The cost math, head-to-head comparisons, and trust answers that every Claude Code tutorial skips \u2014 so you can decide before you adopt.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["Claude Code","AI coding","developer tools","Anthropic","AI agents"],category:"Developer Tools",badge:"45 questions answered",fieldGuideLabel:"Field guide",deck:"Every article about Claude Code tells you it's an AI coding agent you just need to learn to use. This one tells you how to evaluate whether it's worth adopting \u2014 with the cost math, the head-to-head comparisons, and the trust answers that every tutorial skips.",readTimeMinutes:18,stats:[{value:"6",label:"sections"},{value:"45",label:"questions"},{value:"22",label:"gap questions covered"},{value:"0",label:"fluff"}],ctaHeading:"Scrape smarter with",ctaHeadingItalic:"real web data.",ctaBody:"MCP Scraper gives your Claude Code agents the live web intelligence they need \u2014 SERP data, People Also Ask harvests, competitor page extraction, and structured data feeds \u2014 without rate limits or browser fingerprinting.",sections:[{id:"what-is-claude-code",num:"01",title:"What Claude Code",titleItalic:"Actually Is",deck:'Most people who call Claude Code an "AI coding assistant" are describing a category it has already outgrown \u2014 it is closer to a junior engineer that runs in your terminal than a smarter autocomplete.',callout:{eyebrow:"Built-in definition",heading:"Claude Code operates on your full codebase, not just the file you have open.",body:"Unlike Copilot or Cursor's inline suggestions, Claude Code reads your entire project tree, executes shell commands, runs tests, and commits \u2014 making it an <strong>agentic system</strong>, not an autocomplete extension."},cards:[{id:"q1-1",num:"1.1",question:"What exactly is Claude Code?",answer:"<strong>Claude Code is an agentic coding tool that reads your codebase, edits files, runs commands, and integrates with your development tools</strong> \u2014 available in your terminal, IDE, desktop app, and browser. It is not a chat interface you paste code into; it operates on your local filesystem, executes shell commands with your authorization, and can spawn parallel sub-agents on separate subtasks. Unlike autocomplete tools that react to the file you have open, Claude Code takes a goal as input and works through the steps to achieve it across however many files that requires. The distinction matters for evaluation: you are not buying a smarter tab-completion, you are buying an agent that can misunderstand goals, degrade in long sessions, and occasionally do exactly what you said instead of what you meant."},{id:"q1-2",num:"1.2",question:"How does Claude Code work?",answer:"<strong>You describe a goal in natural language; Claude Code reads relevant files, writes or edits code, runs shell commands, and iterates until the task is complete \u2014 operating on your local filesystem throughout.</strong> The model (Claude Sonnet 4.6 by default, or Opus 4.7 for harder problems) holds up to 1M tokens of context, allowing it to reason across an entire medium-sized codebase in a single session. It runs a loop: read context, plan, execute, observe output, adjust \u2014 stopping when the goal is met or when it needs clarification. The practical implication is that prompt quality matters more than most tutorials admit: vague goals produce vague results, and Claude Code will complete a vague goal confidently."},{id:"q1-3",num:"1.3",question:"What can you do using Claude Code?",answer:"<strong>Write and refactor code, run and fix tests, review pull requests, create scripts, set up CI/CD pipelines, manage files, and spawn parallel agents on subtasks</strong> \u2014 across any language or framework. Beyond code editing, Claude Code integrates with GitHub Actions and GitLab CI/CD natively, can trigger pull requests from a Slack mention, connects to external tools via Model Context Protocol (MCP), and can be scheduled as a routine that runs on Anthropic-managed infrastructure while your computer is off. The surface area is broad enough that the more useful question is what it cannot do reliably \u2014 and that list appears in Section 06."},{id:"q1-4",num:"1.4",question:"What is Claude Code best for?",answer:"<strong>Multi-file refactors, greenfield scaffolding, complex debugging cycles, and any task that requires reading a large codebase to understand context before making changes</strong> are where Claude Code's 1M-token context window creates a genuine capability gap over file-local autocomplete tools. It is also well-suited for tasks that require coordination across multiple steps \u2014 setting up a test suite, migrating an API client, or auditing a codebase for a specific pattern \u2014 where the work is too spread out for a single prompt in a chat interface. Where it underperforms: novel algorithmic design, highly domain-specific regulatory code, and real-time systems \u2014 detailed in Section 06."},{id:"q1-5",num:"1.5",question:"Why is Claude Code so good at coding?",answer:"<strong>It scores 80.8% on SWE-bench Verified</strong> \u2014 the leading benchmark for autonomous software engineering \u2014 and uses models with a 1M-token context window, allowing it to hold an entire medium-sized codebase in memory at once. The SWE-bench score means it solves roughly 4 in 5 real-world GitHub issues autonomously; competing tools score lower on the same benchmark. Context window size is the second structural advantage: autocomplete tools reason over hundreds of tokens; Claude Code reasons over millions, which is the difference between fixing a function and fixing a system. The 20% failure rate on SWE-bench is the equally important number \u2014 Section 06 covers what failure looks like in practice."},{id:"q1-6",num:"1.6",question:"Why is everyone obsessed with Claude Code?",answer:"<strong>Claude Code is the first widely-adopted agentic coding tool that works at the project level rather than the file level</strong>, and it frequently completes tasks developers expected to take hours in under ten minutes \u2014 without requiring IDE changes or workflow restructuring. It runs in a terminal, which means it drops into any existing development environment without replacing the editor you already use. The combination of project-level context, shell execution, and a 1M-token window crossed a threshold where developers found themselves assigning real work to it rather than using it as a drafting aid \u2014 and that shift in how people use it, not any single feature, is what generated the adoption curve."},{id:"q1-7",num:"1.7",question:"How can Claude Code be so good?",answer:"<strong>The underlying models \u2014 Claude Sonnet 4.6 and Opus 4.7 \u2014 were trained with a 1M-token context window and ranked first on SWE-bench</strong>; paired with full filesystem access and shell execution, the capability gap over autocomplete tools is structural, not marginal. Most autocomplete tools use models optimized for next-token prediction in a small window; Claude Code uses models optimized for multi-step reasoning across large contexts \u2014 a different training objective that produces qualitatively different behavior. The important caveat for practitioners: benchmark performance reflects average case; your specific codebase, stack, and task distribution may diverge significantly from the benchmark distribution."}]},{id:"setup-and-surfaces",num:"02",title:"Setup, Surfaces, and",titleItalic:"Who Can Use It",deck:'You do not need to be a developer to start a session with Claude Code \u2014 but the gap between "starting a session" and "getting reliable results" is wider than any installation guide will tell you.',cards:[{id:"q2-1",num:"2.1",question:"How do you start using Claude Code?",answer:"<strong>Install via one curl command (`curl -fsSL https://claude.ai/install.sh | bash`), or use Homebrew (`brew install --cask claude-code`) on Mac or WinGet (`winget install Anthropic.ClaudeCode`) on Windows</strong>; authenticate with a Claude Pro subscription or an Anthropic API key, then run `claude` in your project directory. The tool runs on macOS (Intel and Apple Silicon), Windows (x64 and ARM64), Linux, and WSL. A desktop application for macOS and Windows was also released on April 14, 2026, which adds a GUI launcher and Git worktree support for isolated parallel sessions. The fastest path to a first working session is the API key route \u2014 no subscription required, first session costs cents."},{id:"q2-2",num:"2.2",question:"Can Claude Code start from scratch?",answer:"<strong>Yes \u2014 give it an empty directory and a description and it will scaffold the project structure, create files, initialize git, and write initial code without any existing codebase to read.</strong> This is one of the use cases where Claude Code's agentic loop is most visible: it plans the file structure, writes each file, runs an initial build to check for errors, and iterates \u2014 the same way a developer would approach a greenfield project. The caveat is that greenfield outputs still require review: Claude Code will make architectural decisions based on its training data, and those decisions may not match your team's standards, your target infrastructure, or your preferred dependencies."},{id:"q2-3",num:"2.3",question:"Do you need to be a programmer to use Claude Code?",answer:'<strong>Not for well-defined, bounded tasks</strong> \u2014 product managers and researchers have used it to run competitive analyses, clean data, and build simple automation \u2014 but verifying outputs and debugging failures still benefits from technical fluency. The gap that non-programmers run into is not starting Claude Code; it is recognizing when its output is wrong. Claude Code will generate syntactically valid code that does the wrong thing, use a deprecated API without flagging it, or misunderstand a requirement in a way that only becomes visible when the code runs. Knowing what "correct" looks like is a prerequisite for using any agentic coding tool reliably, and that knowledge does not come from the tool itself.'},{id:"q2-4",num:"2.4",question:"Can beginners use Claude Code?",answer:"<strong>Yes, and beginners can complete real tasks in a first session</strong> \u2014 the friction of setup is low, the natural language interface is accessible, and for tasks with clear success criteria (write a Python script that does X, convert this CSV to JSON), the output is often directly usable. The challenge for beginners is the review step: knowing whether a diff is correct requires enough understanding of the code to evaluate it. The practitioner pattern that works for non-expert users is to use Claude Code for tasks where the output is directly testable \u2014 write a test, run it, see if it passes \u2014 rather than for tasks where correctness requires reading and understanding the generated logic."},{id:"q2-5",num:"2.5",question:"Can I use Claude Code without coding knowledge?",answer:`<strong>For bounded, testable tasks \u2014 yes; for open-ended engineering work \u2014 no.</strong> The limiting factor is not the tool's interface but your ability to recognize when Claude has misunderstood the goal, which requires knowing what "correct" looks like for the specific task. Users without coding knowledge have successfully used Claude Code for data processing, file organization, simple script automation, and research tasks where the output is text or structured data they can evaluate directly. The failure mode is adopting it for work where correctness requires code comprehension \u2014 and then merging Claude's output without understanding it.`},{id:"q2-6",num:"2.6",question:"Can you use Claude Code for personal use?",answer:"<strong>Yes \u2014 there is no restriction to commercial or professional use</strong>; the Pro plan at $20/month or an Anthropic API key with pay-as-you-go billing both cover personal projects without restriction. Personal use cases that work well include automating repetitive file tasks, building personal utilities, learning a new language or framework by having Claude scaffold examples and explain them, and running competitive analysis scripts. The API key path is often more economical for personal use: if you spend under a few dollars per day on average, pay-as-you-go will cost less than the $20/month Pro subscription."},{id:"q2-7",num:"2.7",question:"Can you use Claude Code just to chat?",answer:"<strong>Technically yes, but it is an expensive and poorly-optimized path for conversation only</strong> \u2014 Claude Code is built for filesystem-aware, command-executing sessions, and for general conversation, Claude.ai is a better fit and may be cheaper depending on your plan. Running a chat-only session inside Claude Code consumes the same token budget as a coding session, which means you are burning rate-limit capacity on messages that would cost less (or nothing, on a free Claude.ai tier) in the standard interface. The one exception: if you are in the middle of a coding session and need to think through an architectural question without switching contexts, the terminal is a reasonable place to have that conversation."},{id:"q2-8",num:"2.8",question:"Can you prompt Claude Code?",answer:"<strong>Yes \u2014 you interact with Claude Code entirely through natural language prompts in the terminal</strong>, and you can store persistent instructions in a CLAUDE.md file in your project root so preferences, coding standards, and architectural decisions carry across sessions without re-stating them. The CLAUDE.md file is the most underused feature for teams: it allows you to encode your stack's conventions, preferred libraries, test patterns, and style rules once, so every Claude Code session starts with that context loaded. Practitioners who skip CLAUDE.md setup spend significantly more tokens re-explaining context that should be a project-level constant."}]},{id:"cost-math-and-evaluation",num:"03",title:"The Real Cost Math",titleItalic:"Before You Commit",deck:"The subscription price is not the most important number \u2014 the ratio between what you pay at Max 5x versus what the same usage costs on raw API tokens is 18-to-1, and almost no review article publishes it.",callout:{eyebrow:"Evaluation layer \u2014 what every tutorial skips",heading:"You can run Claude Code today without a paid subscription.",body:"An Anthropic API key unlocks full Claude Code functionality on a <strong>pay-as-you-go basis</strong> \u2014 no $20/month Pro plan required. The average developer spends about <strong>$6 per day</strong> at API rates; for light users, this path is cheaper than a monthly plan and removes the cost-before-commit barrier entirely."},cards:[{id:"q3-1",num:"3.1",question:"How to use Claude Code for free?",answer:'<strong>There is no free Claude Code plan</strong>, but you can use it without a subscription by providing an Anthropic API key and paying per token \u2014 and for light users, this often costs less per month than the $20/month Pro subscription. The Free plan on Claude.ai does not include Claude Code access. The API key path requires a funded Anthropic account but no minimum spend: a few evaluation sessions will cost a few dollars, not $20. The distinction matters: "no free plan" and "no way to try it without committing $20" are different conditions, and almost every article conflates them.'},{id:"q3-2",num:"3.2",question:"Can I run Claude Code locally for free?",answer:"<strong>Not for free with Claude models</strong>, but Claude Code can be configured to run against local Ollama-compatible models, which eliminates API costs entirely \u2014 at the cost of model quality relative to Claude Sonnet 4.6. Running Claude Code against a local Ollama model means all inference stays on your machine: no API call, no token spend, no data leaving your network. The trade-off is that local open-source models perform significantly below Claude Sonnet 4.6 on SWE-bench and similar coding benchmarks, so the output quality for complex tasks is not comparable. For privacy-sensitive experimentation or cost-zero prototyping, local Ollama is a legitimate path; for production coding work, the quality gap is real."},{id:"q3-3",num:"3.3",question:"Can I try Claude Code without paying?",answer:"<strong>The API key path requires a funded Anthropic account, but there is no minimum spend</strong> \u2014 you can run several evaluation sessions for a few dollars without committing to any monthly plan. Create an Anthropic account, add a small credit balance ($5\u2013$10 is enough for meaningful evaluation), generate an API key, and authenticate Claude Code with it. At Sonnet 4.6 rates ($3/MTok input, $15/MTok output), a few hours of coding sessions will cost well under $10. This is the evaluation path that no ranking article describes explicitly, which is why most searchers believe the choice is binary: $20/month Pro or nothing."},{id:"q3-4",num:"3.4",question:"Is Claude Code free now?",answer:'<strong>No \u2014 the Free plan does not include Claude Code access</strong>; the tool requires Pro ($20/month), Max ($100\u2013$200/month), Team Premium ($100\u2013$125/seat/month), or an Anthropic API key with pay-as-you-go billing. As of 2026-05-24, Anthropic has not announced a free tier for Claude Code. The "free" path that does exist is the API key route for low-volume users who spend less monthly than the Pro subscription would cost \u2014 which is free of subscription commitment but not free of per-token cost. If you are searching this question because you saw "free" mentioned somewhere, it likely refers to the absence of a required subscription for the API key path, not zero-cost access.'},{id:"q3-5",num:"3.5",question:"How much is Claude Code per month?",answer:"<strong>Pro: $20/month (or $17/month billed annually); Max 5x: $100/month; Max 20x: $200/month; Team Premium: $125/seat/month (or $100/seat/month annually), minimum 5 seats, Claude Code included; API key: pay-as-you-go, average approximately $6/developer/day.</strong> Team Standard ($25/seat/month) does not include Claude Code. The Max plans exist because Pro has a usage ceiling \u2014 approximately 44,000 tokens per 5-hour window \u2014 that active power users hit daily; Max 5x roughly doubles that to 88,000 tokens, and Max 20x reaches approximately 220,000 tokens per window. For developers who would otherwise pay API rates at those volumes, the Max plan is dramatically cheaper than the alternative."},{id:"q3-6",num:"3.6",question:"Is it worth it to pay for Claude for coding?",answer:"<strong>At Pro ($20/month), the break-even is roughly one hour of professional developer time saved per month</strong> \u2014 for developers using it daily on real tasks, the ratio is heavily favorable. The harder question is whether you need Max-tier throughput: if you hit the Pro usage ceiling regularly, you are spending time waiting for windows to reset instead of working, and the $80/month step-up to Max 5x pays for itself quickly. For occasional users \u2014 a few sessions per week on bounded tasks \u2014 the API key path at average $6/developer/day will cost less than $20/month and provides the same capability without the subscription commitment."},{id:"q3-7",num:"3.7",question:"Is Claude Code actually worth it?",answer:"<strong>For developers running multi-file tasks daily, yes \u2014 the Max plan is approximately 18x cheaper than equivalent API usage at full capacity</strong>; for occasional users, the API key path is more economical and the subscription adds no value. The 18x figure comes from the projected cost of purchasing the same token volume directly at Sonnet 4.6 rates ($3/MTok input): at Max 20x throughput sustained, the API equivalent would run approximately $3,650/month versus $200/month for the Max 20x plan. The honest framing for an evaluation decision: start with the API key path, measure your actual daily spend for two weeks, then decide whether the Pro or Max subscription saves money relative to your real usage pattern."},{id:"q3-8",num:"3.8",question:"How expensive is it to use Claude Code?",answer:"<strong>Light use on an API key: under $2/day; moderate use on Pro: $20/month flat; heavy agentic use on Max 5x: $100/month for approximately 88,000 tokens per 5-hour window</strong>, versus approximately $3,650/month if you paid API rates for the same volume. Ninety percent of API-path users spend under $12/day. Prompt caching reduces costs further for long sessions with repeated context: Sonnet 4.6 cache reads cost $0.30/MTok versus $3/MTok for fresh input \u2014 a 90% discount on context that is already in the cache. The Batch API adds a 50% discount across all token prices for non-real-time workloads. Heavy users who ignore caching and batching pay 2\u20133x more than necessary."},{id:"q3-9",num:"3.9",question:"Is Claude Code no longer pro?",answer:"<strong>Claude Code remains available on the Pro plan</strong> \u2014 the Max plans (5x and 20x) are higher-throughput tiers added for power users, not replacements for Pro. Pro was not removed or downgraded; it retains the same model access (Sonnet 4.6 and Opus 4.7) as Max, with tighter usage limits per 5-hour window (approximately 44,000 tokens). The confusion likely stems from Anthropic's introduction of the Max tier, which is marketed heavily to power users \u2014 but Pro is still the primary entry point for individual developers who do not consistently hit usage ceilings."}]},{id:"comparison-and-switching",num:"04",title:"Claude Code vs. the Tools",titleItalic:"You Already Use",deck:"The three tools most developers compare against Claude Code \u2014 Cursor, GitHub Copilot, and ChatGPT \u2014 answer different questions than Claude Code does, and picking the wrong framing makes the comparison meaningless.",callout:{eyebrow:"Decision-stage question the SERP ignores",heading:"Most professional teams use Claude Code alongside Cursor or Copilot, not instead of them.",body:"The most common production stack is <strong>Cursor for inline editing</strong> (72% autocomplete acceptance rate with Supermaven) <strong>+ Claude Code for complex multi-file tasks</strong> in the terminal \u2014 or Copilot in the IDE + Claude Code for architectural work. Picking one and dropping the other is a false choice."},cards:[{id:"q4-1",num:"4.1",question:"Why are people leaving ChatGPT and going to Claude?",answer:"<strong>Claude's models score higher on coding benchmarks (80.8% SWE-bench Verified) and have a substantially longer context window (1M tokens versus 128k for GPT-4o)</strong>, and Claude Code offers deeper filesystem integration than ChatGPT Codex or the GPT-4 API. For developers specifically, the context window difference is the most consequential: 1M tokens allows Claude Code to reason over an entire codebase; 128k limits competing tools to a subset of files. The SWE-bench gap is real but less dramatic than marketing implies \u2014 both tools fail a meaningful percentage of tasks, and the right comparison is not benchmark scores but how each tool behaves on your specific workload."},{id:"q4-2",num:"4.2",question:"Is ChatGPT or Claude better?",answer:'<strong>For agentic coding tasks, Claude Code leads on SWE-bench Verified at 80.8%</strong>; for general conversation, document analysis, and multimodal tasks, the gap between the two is smaller and depends on the specific benchmark. Neither is universally better: GPT-4o has advantages in certain multimodal contexts; Claude Sonnet 4.6 leads on long-context coding tasks. The evaluation question for a developer is not "which is better overall" but "which handles my workload better" \u2014 the two tools have different context window sizes, different pricing structures, and different agentic execution models, and those differences matter more than aggregate benchmark rankings.'},{id:"q4-3",num:"4.3",question:"Why are people switching to Claude?",answer:`<strong>The three primary reasons developers cite: longer context window (1M versus 128k), stronger agentic task performance on SWE-bench, and Claude Code's full-filesystem terminal approach versus chat-based alternatives.</strong> A secondary factor is Constitutional AI training, which produces a model that more often says "I don't know" or flags uncertainty rather than generating confidently wrong output \u2014 a meaningful difference for code review workflows where false confidence is costly. Developers who switched from ChatGPT-based workflows most commonly cite hitting GPT-4o's context limit on large codebase tasks as the triggering event.`},{id:"q4-4",num:"4.4",question:"Is Copilot cheaper than ChatGPT?",answer:"<strong>GitHub Copilot Pro at $10/month is the lowest-priced individual plan in this comparison</strong>: ChatGPT Plus is $20/month, Claude Pro is $20/month, and Cursor Pro is $20/month. Copilot also offers a team plan at $19/seat/month and enterprise at $39/seat/month. The price comparison is misleading without capability context: Copilot at $10/month provides IDE-integrated autocomplete and code chat; it does not include a standalone agentic coding tool equivalent to Claude Code. Developers who need both inline autocomplete (Copilot's strength) and multi-file agentic task completion (Claude Code's strength) are looking at $10 + $20 = $30/month minimum, not a choice between them."},{id:"q4-5",num:"4.5",question:"Who hallucinates more \u2014 ChatGPT or Claude?",answer:"<strong>Both hallucinate; the more useful comparison for coding tasks is SWE-bench score</strong>, which measures how often a model actually solves a real-world issue correctly rather than generating plausible-looking wrong code \u2014 and Claude Code leads at 80.8%. Claude's Constitutional AI training is designed to produce more calibrated uncertainty: the model is more likely to say it does not know something than to confabulate a confident but wrong answer. In practice, both tools will generate syntactically valid code that fails tests, fabricate library method names, and misread logic \u2014 the difference is in frequency and in how they signal uncertainty. For any coding output from either tool, running tests and reviewing diffs is non-optional."},{id:"q4-6",num:"4.6",question:"What is Claude Code and Ollama?",answer:"<strong>Ollama runs open-source language models locally on your machine; Claude Code can be configured to use Ollama-compatible models as its backend, eliminating API costs and keeping all data entirely local.</strong> This configuration is relevant for two use cases: cost-zero experimentation without API spend, and air-gapped or high-security environments where sending code context to an external API is not permissible. The trade-off is model quality: local open-source models perform significantly below Claude Sonnet 4.6 on SWE-bench and complex coding tasks. The Ollama path is worth knowing because most articles presenting Claude Code as requiring a paid subscription omit it entirely \u2014 for teams with strong privacy requirements or zero budget, it is the only viable evaluation path."}]},{id:"safety-trust-and-privacy",num:"05",title:"Safety, Data Handling, and",titleItalic:"What Claude Code Can Actually See",deck:"The most common trust concerns about Claude Code \u2014 screenshot access, code ownership, data leakage \u2014 have clear factual answers, and none of the ranking articles provide them.",callout:{eyebrow:"Enterprise and privacy-sensitive teams",heading:"Claude Code does not store your code on Anthropic's servers between sessions.",body:"Your files run locally on your machine; only the <strong>conversation context</strong> (your prompts and Claude's responses) is sent to Anthropic's API. The Enterprise plan adds <strong>HIPAA-ready data handling</strong>, audit logs, custom data retention controls, and SCIM provisioning \u2014 features that unblock adoption for regulated industries."},cards:[{id:"q5-1",num:"5.1",question:"Is it safe to use Claude Code?",answer:"<strong>For most professional use, yes \u2014 code executes locally, file access is local, and only prompt context traverses the API</strong>; sensitive or regulated data (HIPAA, PII) requires the Enterprise plan for compliant data handling. The practical risk surface for most developers is not data leakage but local command execution: Claude Code executes shell commands you authorize, and approving a destructive command without reviewing it produces real damage. For teams in regulated industries, the Enterprise plan adds HIPAA-ready data handling, audit logs, and custom data retention controls \u2014 the specific features required for compliant adoption in healthcare, finance, and similar sectors."},{id:"q5-2",num:"5.2",question:"Can you get banned from Claude Code?",answer:"<strong>Yes \u2014 Anthropic's usage policies apply to Claude Code, and automated or agentic misuse can trigger account suspension.</strong> The prohibited uses are the same as those that apply to Claude.ai: generating malware, automating scraping in violation of a site's terms, producing content that violates Anthropic's usage policy, and misrepresenting Claude-generated output as human-authored in contexts where that matters. Agentic tools that execute at scale create higher-velocity policy surface than chat interfaces \u2014 a Claude Code routine running unattended can produce policy violations faster than a human would catch them. Review Anthropic's usage policy before deploying Claude Code in automated, unmonitored pipelines."},{id:"q5-3",num:"5.3",question:"Does Claude Code leak?",answer:`<strong>No evidence of systemic data leakage exists</strong>; prompt context is transmitted to Anthropic's API as part of normal operation but is not persisted beyond the session by default under standard plans. What "leak" means technically: your code appears in the prompt context sent to the API for inference; it is not stored, indexed, or accessible to other users. Enterprise plans add explicit custom retention controls and data-use opt-outs for teams who need contractual data handling guarantees rather than policy-level assurances. The meaningful risk for most teams is not leakage to other users but the transmission of proprietary code to an external API at all \u2014 a policy question, not a technical vulnerability.`},{id:"q5-4",num:"5.4",question:"Is Claude Code unsafe?",answer:"<strong>The primary risk surface is local command execution, not network security</strong> \u2014 Claude Code executes shell commands you authorize on your machine, and approving a destructive command without reviewing it produces damage that is real and often irreversible. Claude Code's design requires your explicit approval before executing commands in most configurations, but users who approve commands quickly without reading the proposed action bypass the primary safety mechanism. The secondary risk is prompt injection in multi-agent configurations \u2014 a subtask agent receiving malicious instructions from an external source. Neither risk is exotic; both are manageable with standard review practices."},{id:"q5-5",num:"5.5",question:"Can I trust Claude Code?",answer:`<strong>Trust is context-dependent: for local development tasks, yes; for regulated data, only under the Enterprise plan; for security-sensitive environments, verify the data handling documentation before adopting.</strong> The relevant trust question for most practitioners is not "is Anthropic malicious" but "does sending my codebase context to an external API comply with my organization's data handling policy" \u2014 and that is a legal and compliance question, not a technical one. Enterprise teams evaluating Claude Code should request Anthropic's data processing agreement and review the HIPAA-ready configuration before making adoption decisions; the documentation exists and is specific.`},{id:"q5-6",num:"5.6",question:"Does Claude Code take screenshots?",answer:"<strong>No \u2014 Claude Code does not have screen capture capability.</strong> It reads and writes files on your filesystem and executes terminal commands, but it cannot capture your screen, access your clipboard, read data from applications outside the project directory you opened it in, or observe your browser activity. The question comes up because agentic AI tools are often conflated with general system-access tools; Claude Code's access is specifically scoped to your project directory and the shell commands you authorize it to run. If you are evaluating Claude Code for an environment where screen capture would be a security concern, that concern does not apply to this tool's architecture."},{id:"q5-7",num:"5.7",question:"Does ChatGPT own my code?",answer:"<strong>OpenAI's standard terms do not claim ownership of output code</strong>, and Anthropic's terms similarly do not claim ownership of code that Claude Code generates \u2014 but both companies' default terms may use your inputs to improve their models unless you opt out or upgrade to an enterprise plan. For most developers, code ownership is not the risk: the risk is whether code you submit as input context can be used in model training. Anthropic's Enterprise plan includes explicit data-use opt-outs; OpenAI's Enterprise plan does the same. On individual plans for either tool, review the current terms of service for training data opt-out provisions, which have changed multiple times across both platforms."},{id:"q5-8",num:"5.8",question:"Why is Claude being blacklisted?",answer:"<strong>Some organizations block Claude via firewall because it is an external API service \u2014 not because of a known security vulnerability in Claude specifically.</strong> IT departments that treat all AI API services as unauthorized external data connections will block Claude Code, ChatGPT, Copilot, and similar tools under the same policy, regardless of their individual security properties. The blocking is a data governance decision, not a technical finding against Claude. For enterprise teams trying to get Claude Code approved through IT, the relevant artifacts are Anthropic's data processing agreement, the HIPAA-ready Enterprise configuration documentation, and Anthropic's SOC 2 compliance status \u2014 not the tool's general reputation."}]},{id:"limits-and-skepticism",num:"06",title:"What Claude Code Gets Wrong",titleItalic:"(And What It Cannot Do)",deck:"The performance ceiling that marketing materials never publish: Claude Code hallucinates, degrades in long sessions, and will confidently generate plausible-looking wrong code \u2014 knowing the failure modes before you adopt matters more than knowing the benchmark score.",cards:[{id:"q6-1",num:"6.1",question:"What is Claude not good at?",answer:"<strong>Novel algorithmic design requiring mathematical proof, highly domain-specific regulatory code with no training signal, real-time systems where every millisecond matters, and tasks requiring external context it cannot access</strong> \u2014 production databases, proprietary internal documentation, undocumented internal APIs \u2014 are the consistent failure categories. Claude Code reasons over what is in its context window; anything that must be inferred from systems it cannot read produces hallucinated or superficially correct but functionally wrong output. The most expensive failure mode in practice is not obvious errors but plausible-looking code that passes a surface review and fails in production \u2014 which is why running tests is not optional."},{id:"q6-2",num:"6.2",question:"Is there a limit to how much you can use Claude Code?",answer:"<strong>Yes \u2014 rate limits vary by plan: Pro gets approximately 44,000 tokens per 5-hour window; Max 5x approximately 88,000; Max 20x approximately 220,000</strong>; hitting the ceiling pauses access until the window resets. These are approximate figures derived from usage reports; Anthropic does not publish the exact token limits by plan. The practical effect: heavy Pro users who run multiple long agentic sessions in a day will hit the ceiling and wait; Max 5x handles most power-user workflows without interruption. On the API key path, there are no usage windows \u2014 you pay per token with no ceiling, which is one reason the API path can be preferable for users who need uninterrupted long sessions."},{id:"q6-3",num:"6.3",question:"Is Claude Code getting dumber?",answer:'<strong>Claude Code is not being degraded \u2014 Anthropic has not reduced model capability</strong>; perceived quality drops in long sessions typically trace to context window saturation, not model downgrade. When a session accumulates enough turns that older context is compressed or dropped to fit within the context window, Claude Code loses access to earlier decisions, file states, and constraints \u2014 and the output quality degrades visibly. The fix is to start a new session for major new tasks rather than extending a single session indefinitely. The "getting dumber" perception is real; the cause is session management, not model regression.'},{id:"q6-4",num:"6.4",question:"Does Claude Code hallucinate?",answer:"<strong>Yes \u2014 it will generate incorrect code, fabricate library method names, and misread file logic</strong>, and the SWE-bench score of 80.8% means it fails roughly 1 in 5 real-world tasks by the benchmark's definition. Hallucination in coding contexts looks different from hallucination in conversation: the output is syntactically valid, the method names look plausible, and the logic structure appears correct \u2014 it fails when you run it. Always run tests and review diffs before merging Claude Code output; treating it as authoritative without validation is the most common source of expensive errors. The 80.8% score is the performance ceiling under benchmark conditions; your specific codebase, stack, and task distribution will produce a different empirical failure rate."},{id:"q6-5",num:"6.5",question:"Can you run Claude Code without internet?",answer:"<strong>Not with Claude models \u2014 all inference goes through Anthropic's API, which requires an active internet connection.</strong> The exception is running Claude Code configured against local Ollama-compatible models, which work fully offline with no API call required. For teams in air-gapped environments or with strict egress policies, the Ollama configuration is the only viable path to offline Claude Code use \u2014 at the cost of model quality. Anthropic does not currently offer an on-premises deployment option for Claude models equivalent to some enterprise AI vendors; the Enterprise plan provides stronger data handling guarantees but still routes inference through Anthropic's API."},{id:"q6-6",num:"6.6",question:"Is Claude Code actually useful?",answer:"<strong>For developers running multi-file refactors, complex debugging cycles, or greenfield scaffolding, yes \u2014 the SWE-bench benchmark performance and consistent practitioner time-savings reports are aligned</strong>; for users expecting zero-verification autonomous output, no. The honest evaluation frame: Claude Code is a force multiplier for developers who can review its output, not an autonomous agent that eliminates the need for developer judgment. Teams that have adopted it most successfully use it for the high-context, high-effort tasks that benefit most from 1M-token reasoning \u2014 not as a replacement for understanding the codebase."},{id:"q6-7",num:"6.7",question:"Is Claude Code still the best coding agent?",answer:'<strong>As of mid-2026, Claude Code leads on SWE-bench Verified at 80.8% and on context window size at 1M tokens</strong>; Cursor leads on inline autocomplete acceptance rate at 72% with Supermaven; the "best" answer depends entirely on whether you optimize for agentic task completion or IDE-native editing speed. The benchmark lead is real but not permanent: SWE-bench scores across competing tools have risen steadily, and the gap between leaders narrows with each model generation. The more durable evaluation criterion than benchmark ranking is which tool handles the specific failure modes that matter most in your codebase \u2014 which requires empirical testing, not reading rankings.'}]}]},{slug:"vibe-coding",title:"Vibe Coding: The Complete Honest Guide",titleDisplay:"Vibe",titleDisplayItalic:"Coding.",description:"What vibe coding actually is, which free tools work, whether you can get hired, and the data layer no tutorial mentions \u2014 30 questions answered without the hype.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["vibe coding","AI","development","tools","careers"],category:"AI Development",badge:"30 questions answered",fieldGuideLabel:"Field guide",deck:`Andrej Karpathy coined "vibe coding" on February 2, 2025. By November it was Collins Word of the Year. Most of what you'll read about it sells a tool. This is the version that tells you what works, what breaks, what it pays, and what you still have to build yourself after the prototype runs.`,readTimeMinutes:14,stats:[{value:"6",label:"sections"},{value:"30",label:"questions"},{value:"7",label:"tools compared"},{value:"0",label:"fluff"}],ctaHeading:"Your vibe-coded app needs",ctaHeadingItalic:"real data.",ctaBody:"MCP Scraper gives your AI-generated tools the live web data they need \u2014 SERP results, People Also Ask harvests, page extraction, and structured scraping \u2014 without writing a single line of custom scraping logic.",sections:[{id:"s1",num:"01",title:"What Is Vibe",titleItalic:"Coding?",deck:"The definition everyone links to, the origin story most posts get half-wrong, and the uncomfortable truth no one else in the top results will say.",callout:{eyebrow:"Origin story",heading:'"The hottest new programming language is English."',body:"Andrej Karpathy said that in 2023. On February 2, 2025, he named the practice: vibe coding. <strong>The tweet reached 4.5 million views in days.</strong> Merriam-Webster added the term on March 8, 2025. Collins named it Word of the Year on November 6, 2025. The definition matters because everyone is now selling their own version of it."},cards:[{id:"q1-1",num:"1.1",question:"Is vibe coding just having AI code for you?",answer:`<strong>Not exactly.</strong> In Karpathy's original framing, vibe coding is a specific mode: you describe what you want, you accept the code without reading it, and you treat bugs as a vibe to ride rather than a problem to debug. Merriam-Webster's listing makes the same point \u2014 coders don't need to understand how the code works and must accept that bugs will be present. That is narrower than "AI-assisted development," where you still review, accept, or reject every suggestion the model makes. Most products marketed as "vibe coding" today \u2014 Cursor, Copilot, Windsurf \u2014 are actually closer to AI-assisted development. <em>The distinction matters because the risk profile is completely different the moment you start reading the output.</em>`},{id:"q1-2",num:"1.2",question:"Is vibe coding difficult?",answer:`<strong>For prototypes, no. For anything you have to maintain, yes.</strong> The first working version of a tool \u2014 a price tracker, a Slack bot, a small dashboard \u2014 comes out of a vibe coding session in under an hour. The difficulty curve spikes the moment you try to extend it, debug a regression, or scale it past the original prompt. According to recent surveys, 66% of developers report spending more time fixing "almost-right" AI code than they save generating it. That gap is not in the tutorials. The skill that becomes hard is not writing code \u2014 it is reading code you didn't write, figuring out which AI suggestion to trust, and knowing when to throw the prototype away and rebuild it properly.`},{id:"q1-3",num:"1.3",question:"Is vibe coding a skill?",answer:"<strong>Yes \u2014 but not the skill most people expect.</strong> The transferable skill is prompt precision: describing system behavior clearly enough that the model produces the right thing on the first try. Add to that systems thinking (understanding how parts of an app talk to each other), and the judgment to spot when AI output is wrong before it ships. None of those require knowing syntax. All of them improve with practice. Vibe coders who succeed treat prompting as engineering \u2014 they iterate on prompts the way developers iterate on code, with versioned specs and explicit edge cases. Vibe coders who fail treat prompting as wishful thinking and re-prompt with the same vague description hoping for a better outcome."},{id:"q1-4",num:"1.4",question:"Can anyone be a vibe coder?",answer:`<strong>For personal tools and throwaway prototypes \u2014 yes.</strong> For anything with real users, the practical floor is higher than the marketing suggests. The clearest data point: 25% of Y Combinator's W25 cohort is running codebases that are almost entirely AI-generated, but every one of those teams has a technical founder who understands what the AI is producing. <em>The pattern in successful vibe-coded products is not "no engineering knowledge" \u2014 it is "engineering judgment without the syntax overhead."</em> If you have never thought about how data flows through a system, you can still ship a working prototype. You will struggle the first time it breaks in front of a real user, and that moment arrives faster than most tutorials let on.`},{id:"q1-5",num:"1.5",question:"What is the uncomfortable truth about vibe coding?",answer:`<strong>AI generates code that looks correct and isn't.</strong> The output reads cleanly, runs locally, and passes a casual review. It also reportedly contains SQL injection paths, leaked API keys in client-side bundles, overly permissive CORS configurations, and authentication logic that fails silently on edge cases. According to recent surveys, trust in AI code accuracy fell from 40% to 29% over the last year \u2014 not because the models got worse, but because more developers spent enough time with the output to see what it actually does. The "lower technical barrier" framing is honest for Day 0. It is actively misleading for Day 1 and beyond, when the cost of not understanding your own code starts showing up in production.`}]},{id:"s2",num:"02",title:"Vibe Coding",titleItalic:"Tools.",deck:"Seven tools ranked honestly \u2014 including which ones are free, which are worth paying for, and the one question the tool comparison tables never answer.",callout:{eyebrow:"Tool warning",heading:"No single tool handles Day 0 and Day 1+.",body:"Every tool is Day 0 optimized. None owns Day 1+. The emerging pattern: tools that let you build fast have weak maintenance stories. <strong>The moment you need to debug, audit, or extend AI-generated code, you're on your own.</strong> Pick your tool knowing this gap exists \u2014 and budget separately for the data, deployment, and review layers that none of them include."},cards:[{id:"q2-1",num:"2.1",question:"What is the best vibe coding platform for beginners?",answer:'<strong>Bolt.new and Replit are the two clearest entry points for non-developers.</strong> Both run entirely in the browser, both deploy a working app in one session, both require zero local setup. Bolt.new leans further toward "type what you want, see it run." Replit has a stronger long-term workspace because the project persists with files you can edit later. For developers who already have a code editor and want AI inside it, Cursor (2M+ users, $2B ARR) and Windsurf (1M+ active users) are the strongest options. The honest split: pick browser-based if you have never installed VS Code; pick Cursor or Windsurf if you have.'},{id:"q2-2",num:"2.2",question:"Are there any free vibe coding tools?",answer:"<strong>Yes. A definitive free-tier map.</strong> Bolt.new offers a free plan with daily token limits \u2014 enough to ship a small project, not enough to iterate heavily. Replit's free tier supports small public projects and runs in the browser. Aider is fully open-source and free, runs locally against any LLM you connect. Continue.dev is a free VS Code extension. Claude.ai and ChatGPT both offer free-tier chat that can generate complete code you paste into a runner. GitHub Copilot has a free tier for individuals. Cursor has a free tier with limited completions. <em>The only thing that is not free across all of these: heavy daily usage. Every product gates volume, not access.</em>"},{id:"q2-3",num:"2.3",question:"Which AI for vibe coding?",answer:'<strong>Claude and the GPT family are the two strongest code-generation models.</strong> For IDE-integrated use, two products dominate: GitHub Copilot leads on raw scale (20M total users, 4.7M paid subscribers, reportedly 42% of the AI coding assistant market) and Claude Code leads on satisfaction (91% CSAT in recent surveys, the highest of any AI coding tool measured). Developers using Copilot reportedly complete tasks 55% faster, with PR time dropping from 9.6 days to 2.4 days. The pragmatic answer: use Copilot if you live in VS Code and want the largest ecosystem; use Claude Code if you want the model most developers say produces the fewest "almost-right" answers.'},{id:"q2-4",num:"2.4",question:"Which AI agent is best for vibe coding?",answer:"<strong>For autonomous multi-step tasks, the leaders are Cursor's Composer, Windsurf's Cascade, and Claude Code.</strong> Cascade handles the most aggressive agentic workflows out of the box \u2014 multi-file refactors, end-to-end feature builds \u2014 and Windsurf was acquired by Cognition in December 2025 for reportedly ~$250M, which consolidated agentic capability under one roof. Claude Code is the highest-satisfaction agentic tool in developer surveys. <em>The honest tradeoff: more autonomy means less review, which means more time fixing surprises later.</em> Pick the level of autonomy you actually want to audit."},{id:"q2-5",num:"2.5",question:"Can you vibe code with ChatGPT?",answer:"<strong>Yes, but ChatGPT alone is a copy-paste workflow.</strong> You describe what you want, ChatGPT generates the code, you paste it into a runner \u2014 Replit, your local terminal, a CodeSandbox tab. That works, and it is genuinely free at the entry tier. The friction is everything between paste and run: you don't get inline edits, you don't get file-level context, and you re-paste the whole project every time you want a change. A better free path is ChatGPT plus Replit (ChatGPT writes, Replit runs and persists) or ChatGPT inside an IDE through a plugin. ChatGPT is a viable free-tier route into vibe coding. It is not the best long-term home for a real project."},{id:"q2-6",num:"2.6",question:"What's the best app for vibe coding?",answer:`<strong>"App" depends on the device.</strong> On desktop: Cursor is the most powerful for developers, Bolt.new is the fastest for non-coders, and Replit is the most balanced if you want both browser convenience and a real workspace. On mobile: Replit's iOS app is the only environment that supports full project builds from a phone. <em>Claude.ai and ChatGPT also run in mobile browsers and can generate code you deploy elsewhere, but neither is a complete vibe coding environment on its own.</em> The next section covers iPhone-specific workflows in detail \u2014 most posts skip this entirely, even though the question shows up clearly in search.`}]},{id:"s3",num:"03",title:"How to Start",titleItalic:"Vibe Coding.",deck:"A step-by-step path for non-developers, the one thing every tutorial skips (getting real data into your tool), and the iPhone question every other guide ignores.",callout:{eyebrow:"Critical gap",heading:"The prototype always works. The live version needs a data layer.",body:"Prompting AI to build a tool is 30 minutes. Getting it live data is where most projects die. Every useful vibe-coded tool eventually needs to read from the real web. Price trackers need prices. Research tools need pages. Lead generators need business data. <strong>That's where MCP Scraper enters the workflow.</strong>"},cards:[{id:"q3-1",num:"3.1",question:"How should I start vibe coding?",answer:'<strong>Three steps. In this order.</strong> First, pick a browser-based tool \u2014 Bolt.new or Replit \u2014 so you skip every "install this, configure that" trap that kills momentum on day one. Second, describe one specific thing you want to build, not a general app idea. "A tool that emails me when the price of these three Amazon products drops" beats "an e-commerce price tracker" because the model can build the first one in one pass. Third, test the output immediately and iterate with precise corrections \u2014 name the file, the function, and the exact behavior you want changed. <em>The most common beginner failure is prompting vaguely and then re-prompting with more vagueness, hoping the model figures it out.</em>'},{id:"q3-2",num:"3.2",question:"Can you learn coding from vibe coding?",answer:"<strong>Yes for systems thinking. No for syntax.</strong> Vibe coding teaches you to think in terms of inputs, outputs, state, and failure modes \u2014 the conceptual layer that makes a developer effective. It teaches debugging logic, because you spend real time reading errors and asking the model what they mean. It teaches prompt precision, which transfers to spec-writing in any technical role. What it does not teach is syntactic fluency, language-specific idioms, or low-level architecture decisions. If your goal is to be hired as a traditional software engineer at a company that interviews on syntax, vibe coding is an accelerator for concepts and a poor substitute for fundamentals. If your goal is to ship products, the concepts matter more."},{id:"q3-3",num:"3.3",question:"Can I vibe code on my iPhone?",answer:"<strong>Yes, with real limitations.</strong> Replit has a functional iOS app that supports full project builds, file editing, and deployment from your phone. Claude.ai and ChatGPT both run in Safari and can generate complete code you paste into Replit or another runner. Bolt.new is browser-accessible on mobile, though the desktop layout is the supported experience. No current iOS tool matches the desktop IDE experience for serious projects. For prototyping a small tool, drafting a Slack bot, or sketching the first version of an app idea, your iPhone is a viable vibe coding environment \u2014 and it is the only environment most travelers have on day one of an idea."},{id:"q3-4",num:"3.4",question:"Does Apple ban vibe coded apps?",answer:"<strong>No.</strong> Apple's App Store Review Guidelines do not categorically ban AI-generated code. What they ban \u2014 and have always banned \u2014 are thin wrappers around web content, spam apps with no original functionality, and apps that violate content or privacy policies. None of those rules are AI-specific. The accountability standard is unchanged: the developer is responsible for the app's behavior, regardless of how the code was produced. <em>A vibe-coded native app that genuinely does something useful for the user, handles data responsibly, and meets the same review bar as any other app will pass review.</em> A vibe-coded app that wraps a website in a webview and adds nothing will not \u2014 and would not have passed review in 2018 either."},{id:"q3-5",num:"3.5",question:"Is vibe coding good?",answer:`<strong>For prototyping and personal tools \u2014 unambiguously yes. For production at scale \u2014 only with engineering oversight.</strong> The data supports both halves. 84% of developers use or plan to use AI coding tools, with average savings of 3.6 hours per week. 25% of Y Combinator's W25 cohort runs nearly all-AI-generated codebases. At the same time, 66% of developers report spending more time fixing "almost-right" AI code than they save generating it, according to recent surveys. The honest read: vibe coding is excellent for the 80% of ideas that never needed production-grade code in the first place, and it is genuinely risky for the 20% that do. The skill is knowing which one you're building.`}]},{id:"s4",num:"04",title:"The Real",titleItalic:"Limits.",deck:"Security vulnerabilities, the Day 0 vs. Day 1+ cliff, what happens when AI-generated code hits production \u2014 and how to protect yourself.",callout:{eyebrow:"Industry stat",heading:"66% of developers reportedly spend more time fixing AI code than generating it.",body:"This is not an argument against AI coding tools. It is an argument for using them with your eyes open. <strong>The productivity gains are real \u2014 if you know when to trust the output and when to audit it.</strong>"},cards:[{id:"q4-1",num:"4.1",question:"Do vibe coders understand their code?",answer:"<strong>Most do not, by design.</strong> Karpathy's original framing is explicit: you give in to the vibes, you accept the code without reading it, you let the model handle bugs by reprompting rather than debugging. Merriam-Webster's definition repeats the same idea \u2014 practitioners don't need to understand how the code works and must accept bugs will be present. That is fine for personal tools you can throw away. It is a serious liability for anything deployed to real users: you cannot debug a system you do not understand, you cannot catch security issues you never look for, and you cannot tell a user with confidence what your app actually does with their data. The Day 0 advantage becomes the Day 1+ liability."},{id:"q4-2",num:"4.2",question:"What are the security risks of vibe coding?",answer:'<strong>AI-generated code commonly introduces SQL injection vulnerabilities, insecure API key handling, overly permissive CORS configurations, and authentication logic that fails silently on edge cases.</strong> None of these are visible to a non-developer reviewing the output, because the code looks correct. The mitigation is not "review your code more carefully" \u2014 that asks vibe coders to do exactly what they came here not to do. The mitigation is an automated security scanner. Snyk, GitHub Advanced Security, and Semgrep each run continuously and flag the common AI-generated mistakes before deploy. <em>Pair every vibe coding session with a scanner that runs on commit, and you eliminate the most common production-breaking class of mistake without learning to read every line.</em>'},{id:"q4-3",num:"4.3",question:"When should you not vibe code?",answer:"<strong>Three hard cases.</strong> First, systems handling personal data under GDPR, HIPAA, PCI-DSS, or similar \u2014 AI-generated code needs auditing you cannot DIY, and the regulatory penalty for getting it wrong is higher than the time you saved. Second, financial transaction logic \u2014 silent rounding errors, race conditions, and authorization gaps compound quickly and quietly. Third, anything you need to maintain for more than 12 months without a developer \u2014 AI-generated codebases become unmaintainable faster than hand-written code because the structure was never designed for change. <em>Vibe code freely when the cost of being wrong is small. Bring in engineering when the cost of being wrong is asymmetric.</em>"},{id:"q4-4",num:"4.4",question:"What is the Day 1+ problem in vibe coding?",answer:"<strong>Every vibe coding tool is optimized for the first working build. None of them is optimized for what comes after.</strong> Day 1+ problems are predictable: adding a second feature without breaking the first, debugging a regression you cannot trace, scaling to enough users that the original architecture cracks, and integrating third-party APIs whose contracts change. The tools do not advertise this gap because Day 0 demos sell better than maintenance demos. One developer-practitioner survey concluded that no single tool today can build and maintain an entire application end-to-end. <em>Plan for the Day 1+ moment before it arrives. Pick a tool with file-level access, version control, and an escape hatch into real code review.</em>"},{id:"q4-5",num:"4.5",question:"Can a vibe-coded app go viral?",answer:"<strong>Yes \u2014 with documented examples.</strong> One builder shipped more than ten vibe-coded apps that were used reportedly close to a million times in total before scaling and maintenance pressure forced a pause. Kevin Roose's LunchBox Buddy \u2014 a fridge-photo-to-meal-suggestion tool \u2014 was cited in the New York Times as an early vibe coding demo. Refetch, an open-source Hacker News alternative, was reportedly built in 15 hours of vibe coding on Appwrite Cloud. The pattern across these cases is consistent: vibe-coded apps scale to early traction with no problem, and the crisis arrives when traffic, data volume, or feature scope exceeds what the original AI-generated architecture was built for. That crisis is solvable. It just doesn't solve itself."}]},{id:"s5",num:"05",title:"Career &",titleItalic:"Hiring.",deck:"Six questions no competitor will answer \u2014 whether vibe coding is a real job, what it pays, and how to position it when you're applying.",callout:{eyebrow:"Hiring signal",heading:"The fastest-growing startups have already decided vibe coding is production-ready.",body:"25% of Y Combinator's current cohort runs codebases that are almost entirely AI-generated. IBM cites the stat. Google ignores it. Medium skips it. None asks the obvious next question: what does that mean for the person reading this article? <strong>The job market is catching up.</strong>"},cards:[{id:"q5-1",num:"5.1",question:"Is vibe coding a real job now?",answer:`<strong>It is becoming one \u2014 fastest at startups, slowest at large enterprises.</strong> Early-stage companies increasingly list "AI-assisted developer," "AI product builder," "prompt engineer," and "no-code/AI builder" roles. Freelance platforms like Upwork and Fiverr have active vibe coding service categories with steady project volume. At enterprise scale, formal "vibe coder" titles are still rare \u2014 but the practice is embedded in 18% of developers' day-to-day work according to JetBrains' January 2026 survey, and at the YC startups where 25% of codebases are nearly all AI-generated, it is the default daily practice. <em>The title is lagging the work by roughly 18 months. The work is already mainstream.</em>`},{id:"q5-2",num:"5.2",question:"Do companies hire vibe coders?",answer:'<strong>Early-stage startups and solo-founder companies actively do.</strong> Common titles include "growth engineer," "founding engineer," "AI product builder," and "technical founder in residence." The hiring signal is clearest in YC-backed companies and Series A startups where shipping speed matters more than code purity. Traditional enterprise software companies have been slower \u2014 their hiring processes are designed to test syntax and system-design fundamentals, which vibe coders often have not formally studied. The market direction is consistent with broader adoption: the AI coding tools market reportedly reached $7.37 billion in 2025, and 84% of developers use or plan to use AI tools. <em>The companies hiring fastest are the ones building fastest.</em>'},{id:"q5-3",num:"5.3",question:"Do vibe coders get hired?",answer:"<strong>If you can show working products \u2014 yes. The portfolio matters more than the title.</strong> Demonstrating that you shipped a functional tool used by real people is more compelling to early-stage hiring managers than a CS degree or a coding bootcamp certificate. A live URL with real users beats a GitHub repo with no traction. The friction point is larger companies whose engineering interviews are designed around whiteboard syntax problems vibe coders have not drilled on. The pragmatic path: build five shippable tools, get real users for at least one, document what you built and what you learned, and apply to companies whose hiring is portfolio-driven rather than interview-driven. The first job is the hardest. After the first job, the portfolio compounds."},{id:"q5-4",num:"5.4",question:"How much do vibe coders make?",answer:`<strong>Ranges vary by context. Approximate market signal as of 2026:</strong> Freelance project rates on Upwork and Fiverr for AI-built tools reportedly range from $500 to $5,000 per project, depending on scope and the client's budget. Full-time "AI-assisted developer," "AI product builder," and "founding engineer" roles at startups reportedly range from $80K to $140K base depending on seniority and location, with equity on top. Indie hackers shipping revenue-generating vibe-coded apps have publicly reported product revenue from $1K to $20K per month, with outliers higher. <em>The ceiling scales with what you build, not what you know. Salaried roles cap your upside; products do not.</em> Treat these as ballparks \u2014 no single survey aggregates them yet, and the market is moving monthly.`},{id:"q5-5",num:"5.5",question:"Is vibe coding a real job?",answer:`<strong>It depends on which job market you are targeting.</strong> In the startup and indie developer ecosystem \u2014 yes, building with AI is a marketable practice with paying roles and growing demand. In regulated industries (healthcare, finance, defense) and large enterprise environments \u2014 not yet as a standalone role, because compliance and code-audit requirements still demand engineers who can read every line. The trajectory is clearly toward normalization: AI reportedly writes about 41% of all new code today, 84% of developers are using or planning to use AI tools, and 25% of YC's current cohort runs nearly-all-AI-generated codebases. <em>The current state is early but real. The forward curve points toward "yes" being the default answer within two to three years.</em>`}]},{id:"s6",num:"06",title:"The Future of",titleItalic:"Coding.",deck:"Whether AI will replace coders, what the realistic 2026\u20132040 arc looks like, and the one skill that becomes more valuable as AI writes more code.",callout:{eyebrow:"Key take",heading:"AI replaces code-writing. It doesn't replace problem-solving.",body:"The developers who will struggle are those whose value is in typing code fast. <strong>The developers who will thrive are those whose value is in knowing what to build, how systems should behave, and when AI output is wrong.</strong> Vibe coding is the fastest way to find out which one you are."},cards:[{id:"q6-1",num:"6.1",question:"Will coders be replaced by AI?",answer:'<strong>Rote code-writing is already being replaced.</strong> AI reportedly writes about 41% of all code today, and 84% of developers use or plan to use AI coding tools. The role-level effect is more specific than "coders are obsolete": developers who specialize in syntax and straightforward implementation are most exposed; developers who specialize in architecture, system design, and problem framing are least exposed. The practice of writing code is being automated. The judgment of what to build, why to build it, and how to know whether it works is not \u2014 and there is no current evidence that it will be soon. <em>"Coder" is becoming a smaller part of "software developer." The other parts are growing.</em>'},{id:"q6-2",num:"6.2",question:"Will AI replace coders by 2040?",answer:"<strong>By 2040, AI will likely generate the majority of code by volume.</strong> What survives \u2014 and what the labor market will pay for \u2014 is the meta-skill: defining problems precisely, evaluating AI output critically, and directing systems toward intended outcomes. The vibe coder of 2026 who develops those meta-skills is better positioned than the traditional developer who ignores them. The career risk over the next 15 years is not AI itself. It is the choice not to adapt to AI. The 25% YC cohort statistic is the leading indicator: the fastest-growing companies are already operating at the model that the rest of the market will reach. Plan accordingly."},{id:"q6-3",num:"6.3",question:"What skill becomes most valuable as AI writes more code?",answer:"<strong>Prompt precision.</strong> The ability to describe complex system behavior clearly enough that AI produces the right output on the first try is the emerging premium skill. Closely paired with it: the ability to audit AI output for correctness, security, and architectural soundness \u2014 not by reading every line, but by knowing which questions to ask and which automated checks to run. Both are learnable by non-developers. Neither requires fluency in any programming language. <em>The developers who treat prompting as engineering \u2014 versioned, specified, tested \u2014 are already pulling ahead of the developers who treat it as a chat.</em> That gap will widen."},{id:"q6-4",num:"6.4",question:"What does a vibe-coded app need to work in the real world?",answer:"<strong>Three things \u2014 and the middle one is where most vibe coders stall.</strong> First, a deployment target: Vercel, Replit, Railway, Fly.io. Second, real data access \u2014 every useful live tool eventually needs to read from the web, and writing custom scrapers, handling JavaScript-rendered pages, and maintaining selectors as sites change is the wall most prototypes hit. Third, a feedback loop with actual users. The data access layer is the most underestimated step in the entire vibe coding workflow. MCP Scraper is the data layer vibe coders reach for when the prototype needs to start consuming real-world inputs \u2014 SERP results, People Also Ask trees, page extraction, YouTube transcripts \u2014 without writing and maintaining the scraping code themselves."}]}]},{slug:"people-also-ask-seo",title:"People Also Ask SEO: The Complete 2026 Guide",titleDisplay:"People Also Ask",titleDisplayItalic:"SEO.",description:"What PAA boxes are, how Google generates them, which tools harvest them at scale, and why the manual two-step loop breaks the moment you need programmatic data.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["SEO","people also ask","content strategy","PAA","keyword research"],category:"SEO",badge:"31 questions answered",fieldGuideLabel:"Field guide",deck:"PAA strategy ends where programmatic begins. Every competitor guide shows you how to expand boxes \u2014 none shows you how to harvest them at scale via API and pipe the data directly into your content system.",readTimeMinutes:12,stats:[{value:"85%",label:"of Google searches show PAA boxes"},{value:"13.6%",label:"CTR for purchase-intent PAA clicks"},{value:"31",label:"questions answered in this guide"},{value:"90%",label:"of AI Overviews include PAA boxes"}],ctaHeading:"Stop copying questions from a SERP accordion.",ctaHeadingItalic:"MCP Scraper delivers structured PAA trees via API \u2014 one call, any keyword, directly into your pipeline.",ctaBody:"MCP Scraper is the PAA harvesting API for developers and SEOs who need more than 4 visible questions. Extract hundreds of People Also Ask questions from any query via REST or MCP.",sections:[{id:"s1",num:"01",title:"What Is People",titleItalic:"Also Ask?",deck:"PAA mechanics, Google behavior, and what the AI-generation shift means for the feature \u2014 the foundational questions most guides leave unanswered.",callout:{eyebrow:"Key take",heading:"PAA is not a content placement. It is a real-time map of query intent.",body:"Google surfaces PAA on roughly 85% of searches. Winning a placement is useful. Understanding why Google shows the questions it does \u2014 and how 12.6% of answers are now AI-generated \u2014 is what separates a tactic from a strategy. <strong>Optimize the placement. Understand the system.</strong>"},cards:[{id:"q1-1",num:"1.1",question:"What do people also ask in SEO?",answer:`<strong>People Also Ask (PAA) is a dynamic SERP accordion Google introduced in 2016</strong> that surfaces related questions predicted to follow a searcher's original query. It now appears on roughly <strong>85% of searches</strong> \u2014 not a niche placement opportunity but a near-universal feature that functions as Google's real-time map of query intent for every topic. Early testing began in April 2015 as "Related questions" before the official July 2016 naming. For SEOs, PAA is two things simultaneously: a visibility placement to win, and a research signal showing which follow-up questions Google believes matter most to your searchers.`},{id:"q1-2",num:"1.2",question:"How does Google generate people also ask?",answer:'Google generates PAA algorithmically from <strong>co-occurrence patterns in search sessions</strong> \u2014 which queries follow which, and what content satisfies them. One significant recent shift: <strong>12.6% of PAA answers are now AI-generated by Google itself</strong>, not pulled from any web page. This means winning a PAA placement no longer guarantees your content is the displayed source. The answer Google shows may be synthesized entirely from its own models, with your page as a citation at best. Understanding this shift changes the strategic goal from "win the placement" to "be the authoritative source Google trusts when it generates the answer."'},{id:"q1-3",num:"1.3",question:"What does it mean when Google says people also searched for?",answer:'<strong>"People also searched for" is a distinct SERP feature</strong> \u2014 it appears after a user clicks a result and returns to the search page, surfacing refinement queries based on what searchers do next. <strong>PAA appears before the click</strong>, predicting follow-up intent from the query itself. They signal different moments in the search session and target different optimization strategies. "People also searched for" is a post-click refinement signal; PAA is a pre-click intent prediction. Conflating the two leads to misaligned content strategy \u2014 targeting refinement queries when your page needs to answer the predicted follow-ups, or vice versa.'},{id:"q1-4",num:"1.4",question:"How do Google people also ask work?",answer:"<strong>PAA boxes are infinite-scroll</strong>: expanding one question loads 2\u20134 additional questions, allowing Google to map an entire topic graph from a single seed query. The block appears after the first organic result in more than <strong>58% of SERPs</strong>, making it a top-three SERP element in roughly two-thirds of all searches. Featured answers average <strong>40\u201350 words</strong>. The infinite-scroll mechanism is the most operationally important detail for SEOs: a single seed keyword can branch into hundreds of related questions, which is why harvesting PAA programmatically returns a fundamentally different volume of data than manually expanding a visible accordion."},{id:"q1-5",num:"1.5",question:"How to find people also ask?",answer:"Three methods, in order of scale: <strong>manual SERP expansion</strong> \u2014 open Google, type your keyword, click the accordion arrows (free, works at 1\u20135 keywords); <strong>UI tools like AlsoAsked</strong>, which automate question collection and return visual question trees ($12\u2013$47/mo, works at 5\u20131,000 keywords); and <strong>programmatic API access via MCP Scraper</strong>, which returns structured PAA data for any keyword without a browser (works at any scale). The right method depends entirely on how many keywords you need to cover. The manual path is not inferior \u2014 it is simply volume-limited, and that limit arrives faster than most SEOs expect."},{id:"q1-6",num:"1.6",question:"What type of questions can Google not answer?",answer:"PAA avoids four structural categories: <strong>highly time-sensitive breaking news</strong> (where no authoritative answer has stabilized), <strong>deeply personal queries</strong> (medical, legal, financial specifics tied to individual circumstance), <strong>normative controversies</strong> where no consensus source exists, and <strong>queries where the intent is too ambiguous</strong> for any single question formulation to make sense. Understanding these boundaries helps SEOs identify where PAA placements are structurally off the table \u2014 not due to competition, but due to feature design. If your topic falls into one of these categories, optimize for other SERP features rather than PAA."}],interactiveHtml:`<div class="si si-calculator"><span class="si-heading">How many PAA questions does one keyword unlock?</span><div class="si-calc-inputs"><label class="si-label">Seed keywords<div class="si-slider-row"><input type="range" class="si-range" id="sip1-s" min="1" max="50" value="3"><span class="si-range-val" id="sip1-sv">3</span></div></label><label class="si-label">Expansion depth (click levels)<div class="si-slider-row"><input type="range" class="si-range" id="sip1-d" min="1" max="3" value="2"><span class="si-range-val" id="sip1-dv">2</span></div></label></div><div class="si-calc-output"><div class="si-stat"><span class="si-stat-num" id="sip1-m">12</span><span class="si-stat-label">questions visible manually</span></div><div class="si-stat si-stat-accent"><span class="si-stat-num" id="sip1-f">252</span><span class="si-stat-label">questions in the full PAA tree</span></div></div><script>(function(){var s=document.getElementById('sip1-s'),d=document.getElementById('sip1-d'),sv=document.getElementById('sip1-sv'),dv=document.getElementById('sip1-dv'),mo=document.getElementById('sip1-m'),fo=document.getElementById('sip1-f');function upd(){var ss=+s.value,dd=+d.value;sv.textContent=ss;dv.textContent=dd;mo.textContent=ss*4;fo.textContent=ss*(dd===1?28:dd===2?84:252);}s.addEventListener('input',upd);d.addEventListener('input',upd);})()</script></div>`},{id:"s2",num:"02",title:"PAA as an",titleItalic:"SEO Strategy",deck:"Intent mechanics, keyword discovery, and why PAA is a top-5 SEO strategy because of AI Overviews \u2014 not despite them.",callout:{eyebrow:"The number that changes the strategy",heading:"Purchase-intent PAA queries drive 13.6% interaction rates. Overall PAA: 3%.",body:"The 4\xD7 gap is not a curiosity \u2014 it is the entire prioritization framework. <strong>Which questions are worth targeting first is a data problem, and the data problem requires harvesting tools, not intuition.</strong>"},cards:[{id:"q2-1",num:"2.1",question:"What are the 4 types of intent in SEO?",answer:"The four intent types \u2014 <strong>informational, navigational, commercial, transactional</strong> \u2014 are not equally valuable in PAA strategy. Purchase-intent queries drive a <strong>13.6% PAA interaction rate</strong>, versus 3% for searches overall. That 4\xD7 gap means intent classification is not just a content exercise \u2014 it is the highest-leverage variable in deciding which PAA questions are worth targeting first. Most SEO teams apply intent classification to keyword strategy but never extend it to PAA prioritization. Applying it there is the arbitrage: harvest PAA for purchase-intent and commercial queries first, and the ROI per question answered separates immediately from the pack."},{id:"q2-2",num:"2.2",question:"What are the 3 C's of search intent?",answer:"Content type, content format, and content angle \u2014 the <strong>3 C's</strong> \u2014 determine structural eligibility for a PAA placement before Google evaluates topical relevance. <strong>Format matters most</strong>: PAA answers average 40\u201350 words, which means a 2,000-word section cannot win a placement regardless of its quality. PAA requires an extractable, self-contained answer at the right word count. The content angle determines whether your framing matches the specific question Google is showing \u2014 a page that answers the general topic but not the exact question in the accordion is not eligible. The 3 C's applied to PAA become a pre-qualification checklist, not an afterthought."},{id:"q2-3",num:"2.3",question:"What is the 80/20 rule in SEO?",answer:"Applied to PAA, the 80/20 rule holds in the data: the <strong>13.6% interaction rate for purchase-intent queries versus 3% overall</strong> suggests that a small subset of PAA questions drives disproportionate engagement and commercial value. Identifying that subset \u2014 by intent type and query category \u2014 is precisely what PAA harvesting tools solve, because the questions that matter most are rarely obvious from the seed keyword alone. A human reviewing a SERP accordion sees 4 questions. A programmatic harvest of the full PAA tree for that seed can surface 200. The 80/20 rule applies to that full dataset, not to the visible 4."},{id:"q2-4",num:"2.4",question:"What are the best SEO tools?",answer:"For PAA strategy specifically, a complete stack looks like: <strong>Google Search Console</strong> (free, shows existing PAA appearances for your domain), <strong>AlsoAsked</strong> (UI-based question trees, $12\u2013$47/mo, best for manual research up to 1,000 seeds), <strong>Semrush</strong> (broad SERP feature tracking), and <strong>MCP Scraper</strong> (programmatic PAA API for pipeline integration). The right combination depends on whether your workflow is UI-based or data-pipeline-based. Teams doing content at scale need both tiers \u2014 the UI tool for ad-hoc research and the API for systematic collection. The two are complementary, not competing."},{id:"q2-5",num:"2.5",question:"Which is the best free SEO tool for beginners?",answer:"For PAA research specifically, <strong>Google Search Console</strong> is the strongest free starting point \u2014 it shows which PAA features your site already appears in, at no cost. <strong>AlsoAsked offers free monthly credits</strong> without requiring a registered account. MCP Scraper is not the beginner recommendation; it requires API literacy and suits developers moving into structured data workflows, not first-time SEOs. The honest path for a beginner: start with Search Console to see what you already have, use AlsoAsked free credits to map the questions you're missing, and add the API layer when manual research becomes the rate-limiting step in your workflow."},{id:"q2-6",num:"2.6",question:"What are the top 5 SEO strategies?",answer:"PAA optimization earns a top-5 slot not because its direct interaction rate is high (3% overall) but because of its relationship with AI Overviews: <strong>PAA co-appears with AI Overviews in 90% of cases</strong>. Winning a PAA placement now doubles as qualifying content to be sourced by AI Overviews \u2014 a compound visibility return that traditional link-building cannot replicate. Optimizing for PAA is, in practice, optimizing for AI Overview sourcing eligibility. The 90% co-appearance figure means these two features share the same content signal. Teams that ignore PAA are leaving the most reliable AI Overview proxy on the table."}],interactiveHtml:`<div class="si si-decision"><span class="si-heading">Is your PAA strategy leaving data on the table?</span><div class="si-questions"><div class="si-q"><p class="si-q-text">Do you prioritize PAA questions by intent type \u2014 targeting purchase-intent questions before informational ones?</p><div class="si-q-opts"><label class="si-opt"><input type="radio" name="sip2-q0" value="1"> Yes \u2014 intent drives my question selection</label><label class="si-opt"><input type="radio" name="sip2-q0" value="0"> No \u2014 I target all PAA questions equally</label></div></div><div class="si-q"><p class="si-q-text">Do you refresh PAA-targeted content at least once per quarter?</p><div class="si-q-opts"><label class="si-opt"><input type="radio" name="sip2-q1" value="1"> Yes \u2014 freshness is part of my process</label><label class="si-opt"><input type="radio" name="sip2-q1" value="0"> No \u2014 I optimize once and move on</label></div></div><div class="si-q"><p class="si-q-text">Is your target keyword list larger than 50 keywords?</p><div class="si-q-opts"><label class="si-opt"><input type="radio" name="sip2-q2" value="1"> Yes \u2014 50 or more keywords</label><label class="si-opt"><input type="radio" name="sip2-q2" value="0"> No \u2014 fewer than 50</label></div></div></div><div class="si-result" id="sip2-result" hidden><p class="si-result-label" id="sip2-rl"></p><p class="si-result-body" id="sip2-rb"></p></div><script>(function(){var results=[{label:'Manual research fits your current scale.',body:'Your keyword list is manageable and your process is structured. No tool changes needed yet \u2014 the manual path handles your volume.'},{label:'You are at the edge of what manual research supports.',body:'One gap in intent prioritization or freshness cadence is limiting your results. Focusing on purchase-intent PAA first is the highest-ROI next step.'},{label:'Your PAA strategy needs a data layer.',body:'At 50+ keywords with systematic intent filtering and quarterly freshness requirements, programmatic access is no longer optional \u2014 manual research is your rate-limiting step.'}];function check(){var score=0,answered=0;for(var i=0;i<3;i++){var r=document.querySelector('input[name="sip2-q'+i+'"]:checked');if(r){score+=+r.value;answered++;}}if(answered<3)return;var res=score>=2?results[2]:score===1?results[1]:results[0];var el=document.getElementById('sip2-result');document.getElementById('sip2-rl').textContent=res.label;document.getElementById('sip2-rb').textContent=res.body;el.hidden=false;}['sip2-q0','sip2-q1','sip2-q2'].forEach(function(n){document.querySelectorAll('input[name="'+n+'"]').forEach(function(r){r.addEventListener('change',check);});});})()</script></div>`},{id:"s3",num:"03",title:"AI and the",titleItalic:"Future of PAA",deck:"Is SEO dead? Will AI replace it? Committed answers backed by data \u2014 including why AI Overviews make PAA more important, not less.",callout:{eyebrow:"Counter-intuitive finding",heading:"AI Overviews make PAA more important, not less.",body:"PAA co-appears with AI Overviews in 90% of searches. Winning a PAA placement is currently the most reliable proxy for AI Overview sourcing eligibility. <strong>The teams abandoning PAA because of AI are removing themselves from the exact feature that signals AI citation readiness.</strong>"},cards:[{id:"q3-1",num:"3.1",question:"Is SEO dead or evolving in 2026?",answer:"<strong>Structurally shifting, not dying</strong> \u2014 but the shift is significant. 58.5% of US Google searches already end without a click. AI Overviews reduce position-1 CTR by up to 58%, and searches triggering AI Overviews show an 83% zero-click rate. <strong>PAA co-appears with AI Overviews in 90% of cases.</strong> The game has moved from generating clicks to being cited as an answer source \u2014 visibility without click-through is the new baseline. The SEOs who treat this as a crisis are measuring the wrong thing. The SEOs who treat it as a repositioning opportunity are already optimizing for sourcing frequency, not position rank."},{id:"q3-2",num:"3.2",question:"Will SEO be replaced by AI?",answer:"<strong>AI is not replacing SEO</strong> \u2014 it is automating the parts that were manual and amplifying the parts that require data access. The practitioners who win are those using AI for content generation while feeding it with live, structured SEO data: PAA trees, SERP features, intent signals. The ones who lose are optimizing one page at a time while their competitors run data pipelines that update daily. The replacement narrative confuses the tool with the discipline. SEO as structured analysis of how content earns visibility in search systems is not going anywhere. The execution layer is being automated. The strategic layer is becoming more valuable, not less."},{id:"q3-3",num:"3.3",question:"What is SEO being replaced by?",answer:"The emerging term is <strong>answer-engine optimization (AEO)</strong> \u2014 structuring content so AI systems cite it as a source, not just rank it in blue links. PAA boxes are currently the most reliable signal for what questions AI Overviews will answer from a given domain, because the two features co-appear in <strong>90% of cases</strong>. Winning PAA now is training data sourcing practice for the AI-first SERP. AEO is not a replacement for SEO \u2014 it is an extension of it toward the citation layer. The content signals that earn PAA placements and the signals that earn AI Overview citations overlap significantly. PAA optimization is AEO in its most accessible form."},{id:"q3-4",num:"3.4",question:"Can ChatGPT write SEO articles?",answer:"<strong>ChatGPT can draft content but cannot perform PAA research</strong> \u2014 it has no live SERP access, returns no real-time question clusters, and cannot tell you which questions Google is surfacing for your keyword today. PAA data is a live signal that requires querying the SERP, not a language model. Any AI-assisted SEO workflow that skips a live data layer is building content strategy on stale assumptions. The correct architecture is: live PAA data (from a harvesting API) feeding structured inputs to the LLM, with the LLM handling drafting and the data layer handling research. Skipping the data layer produces well-written answers to questions no one is actually asking."},{id:"q3-5",num:"3.5",question:"Can ChatGPT do an SEO audit?",answer:"<strong>ChatGPT can review content structure</strong>, flag missing headers, and suggest improvements to on-page copy \u2014 but it cannot audit live SERP features, check which PAA questions your competitors currently hold, or identify PAA placement gaps across your keyword set. Those tasks require real-time structured data pulled from the SERP itself, not pattern-matching against training data. The distinction is not about writing quality \u2014 it is about data access as a category. LLM-based audits are useful for qualitative content review. They are structurally incapable of competitive PAA analysis, which requires live harvesting at query time."},{id:"q3-6",num:"3.6",question:"Which AI is best for SEO?",answer:'<strong>No single AI model is "best for SEO"</strong> \u2014 the question frames it wrong. The winning configuration is: a live data API (PAA harvesting, SERP feature extraction) feeding structured inputs to an LLM for content generation and optimization. The AI model handles language; the data layer handles reality. MCP Scraper occupies the data layer slot \u2014 it supplies the question data that makes content decisions defensible rather than intuitive. Conflating the two is why AI-assisted SEO often produces content that reads well but targets the wrong questions. The model choice matters far less than whether your workflow has a live data layer at all.'}],interactiveHtml:`<div class="si si-quiz"><span class="si-heading">Quick check</span><div class="si-quiz-q" data-correct="1"><p class="si-quiz-q-text">PAA co-appears with AI Overviews in what percentage of searches?</p><div class="si-quiz-opts"><button class="si-quiz-opt" data-idx="0">23% of searches</button><button class="si-quiz-opt" data-idx="1">90% of searches</button><button class="si-quiz-opt" data-idx="2">58% of searches</button></div><p class="si-quiz-explanation" hidden>The 90% co-appearance rate is the number that changes everything \u2014 winning a PAA placement is currently the most reliable proxy for AI Overview sourcing eligibility. Teams abandoning PAA because of AI are removing themselves from the exact signal that governs AI citations.</p></div><script>(function(){document.querySelectorAll('.si-quiz .si-quiz-q').forEach(function(qBlock){var correct=+qBlock.dataset.correct;var explanation=qBlock.querySelector('.si-quiz-explanation');qBlock.querySelectorAll('.si-quiz-opt').forEach(function(btn){btn.addEventListener('click',function(){if(qBlock.dataset.answered)return;qBlock.dataset.answered='1';var idx=+btn.dataset.idx;qBlock.querySelectorAll('.si-quiz-opt').forEach(function(b,i){b.disabled=true;if(i===correct)b.classList.add('si-quiz-correct');else if(i===idx&&idx!==correct)b.classList.add('si-quiz-wrong');});explanation.hidden=false;});});});})()</script></div>`},{id:"s4",num:"04",title:"PAA Tools",titleItalic:"Compared",deck:"AlsoAsked vs. Semrush vs. MCP Scraper \u2014 honest tradeoffs, including where MCP Scraper is and is not the right choice.",cards:[{id:"q4-1",num:"4.1",question:"How does AlsoAsked work?",answer:"<strong>AlsoAsked crawls Google PAA boxes for a given keyword</strong> and builds a branching tree of related questions, which it presents as a visual map, PNG export, or CSV. Bulk upload processes up to 1,000 seed terms in a single job, returning a large question set from recursive PAA expansion. It includes multi-region and multi-language support, API access, and webhook integration across all paid tiers. AlsoAsked is best positioned for UI-based workflows where a researcher is manually reviewing and selecting questions. The limit of the approach is pipeline integration: CSV exports require a human in the loop, and the API, while available on all paid tiers, is designed around the same single-query model as the UI."},{id:"q4-2",num:"4.2",question:"Is AlsoAsked free?",answer:"<strong>AlsoAsked offers free monthly credits</strong> for non-registered users, with no credit card required. Paid plans start at $12/mo (Basic, 100 credits) and go to $47/mo (Pro, 1,000 credits); the $23/mo Lite tier (300 credits) is listed as most popular. Annual billing saves 20%. <strong>All paid tiers \u2014 including Basic \u2014 include API access</strong>, so the API is not gated behind a premium plan. The free tier is genuinely useful for occasional PAA research. The paid tiers are priced for practitioners doing regular question harvesting. The $12 Basic plan is a legitimate entry point for SEOs who need more than the free credits allow but are not yet at pipeline scale."},{id:"q4-3",num:"4.3",question:"What are the 4 pillars of SEO?",answer:'Technical, on-page, off-page, and content \u2014 the <strong>traditional four pillars</strong> \u2014 are all affected by PAA strategy. But the content pillar increasingly depends on the technical pillar for data access: a content team that cannot harvest PAA programmatically is relying on manual research that caps out at dozens of keywords. The teams closing content at scale have connected their data layer directly to their publishing pipeline. PAA strategy now bridges two pillars simultaneously. The content question ("which questions should we answer?") is answered by the technical infrastructure ("what does the PAA API return for this seed keyword?"). That bridge is the competitive gap most content teams have not crossed.'},{id:"q4-4",num:"4.4",question:"Can a beginner do SEO?",answer:"A beginner can start PAA research immediately \u2014 <strong>free tier on AlsoAsked</strong> (no account required), Google Search Console for existing rankings, and manual SERP expansion for small keyword sets. The step-change to programmatic PAA requires basic API literacy, not advanced SEO expertise. <strong>Developers entering content teams are often better positioned</strong> for the API path than experienced SEOs who have never worked with structured data outputs. The entry barrier is not SEO knowledge \u2014 it is familiarity with REST APIs and JSON. A developer who has never done SEO can integrate the MCP Scraper API faster than an experienced SEO who has never touched an API endpoint."},{id:"q4-5",num:"4.5",question:"Can I do SEO by myself?",answer:"Solo SEOs can run effective PAA strategy \u2014 the <strong>manual workflow handles 1\u20135 target keywords well</strong>. The friction hits at roughly 50 keywords: at that volume, manually expanding PAA trees and logging questions becomes the rate-limiting step, not the content writing. The programmatic API path is the unlock at scale, not a requirement for getting started. The honest threshold: if your keyword list fits on one spreadsheet page and you update it quarterly, manual PAA research is sufficient. If your keyword list is dynamic, multi-locale, or feeds an automated content system, the API path pays for itself in the first week of research time saved."}],interactiveHtml:`<div class="si si-comparison"><span class="si-heading">Pick a tool to compare</span><div class="si-tabs" role="tablist"><button class="si-tab si-tab-active" role="tab" aria-selected="true" data-panel="sip4-p0">AlsoAsked</button><button class="si-tab" role="tab" aria-selected="false" data-panel="sip4-p1">Semrush</button><button class="si-tab" role="tab" aria-selected="false" data-panel="sip4-p2">MCP Scraper</button></div><div class="si-panels"><div class="si-panel" id="sip4-p0" role="tabpanel"><p class="si-panel-title">AlsoAsked \u2014 UI-based PAA question trees</p><ul class="si-panel-pros"><li>Visual question tree with PNG and CSV export</li><li>Free tier with no account required</li><li>Bulk upload up to 1,000 seeds, multi-region support</li></ul><ul class="si-panel-cons"><li>CSV export requires human review in the loop</li><li>API mirrors the single-query UI model \u2014 not designed for pipeline volume</li></ul><p class="si-panel-verdict">Best for UI-based research at up to 1,000 seeds per month. The right tool when a researcher is manually reviewing and selecting questions.</p></div><div class="si-panel" id="sip4-p1" role="tabpanel" hidden><p class="si-panel-title">Semrush \u2014 Full SEO suite with PAA tracking</p><ul class="si-panel-pros"><li>PAA alongside rankings, backlinks, and site audit in one platform</li><li>Historical SERP feature data for trend analysis</li></ul><ul class="si-panel-cons"><li>PAA is not its primary strength \u2014 depth limited vs. dedicated tools</li><li>Expensive if PAA is your only use case</li><li>No programmatic extraction of full PAA trees</li></ul><p class="si-panel-verdict">Right if you already use Semrush for keyword research and want PAA visibility added. Overkill for PAA-only workflows.</p></div><div class="si-panel" id="sip4-p2" role="tabpanel" hidden><p class="si-panel-title">MCP Scraper \u2014 Web dashboard + API + MCP server</p><ul class="si-panel-pros"><li>Dashboard at mcpscraper.dev: run PAA, SERP, Maps, YouTube, and Facebook Ads from one UI</li><li>Same data available via REST API and as MCP tools for Claude, Cursor, and Copilot</li><li>Full PAA tree per seed, results as cards or structured JSON/Markdown export</li></ul><ul class="si-panel-cons"><li>Covers seven surfaces \u2014 more than you need if PAA is your only use case</li><li>Credit-based billing: each surface costs credits, not a flat subscription per tool</li></ul><p class="si-panel-verdict">Right when you need PAA alongside SERP data, Maps intelligence, or competitive ad research \u2014 and when you want the same data accessible to both your team and your AI agents.</p></div></div><script>(function(){document.querySelectorAll('.si-comparison').forEach(function(comp){comp.querySelectorAll('.si-tab').forEach(function(tab){tab.addEventListener('click',function(){comp.querySelectorAll('.si-tab').forEach(function(t){t.classList.remove('si-tab-active');t.setAttribute('aria-selected','false');});comp.querySelectorAll('.si-panel').forEach(function(p){p.hidden=true;});tab.classList.add('si-tab-active');tab.setAttribute('aria-selected','true');document.getElementById(tab.dataset.panel).hidden=false;});});});})()</script></div>`},{id:"s5",num:"05",title:"Scaling PAA",titleItalic:"Extraction",deck:"The programmatic case \u2014 why the manual UI loop breaks at scale, and what a PAA data pipeline actually looks like.",callout:{eyebrow:"The scale threshold",heading:"PAA questions shift by location, device, language, and time. Scraping them once is not a strategy.",body:"Google mines search sessions continuously. PAA is a live signal, not a static dataset. <strong>Recurring programmatic harvesting on a schedule is the correct workflow for any live content operation \u2014 not a one-time manual pull followed by a spreadsheet filed away.</strong>"},cards:[{id:"q5-1",num:"5.1",question:"What are the 5 important concepts of SEO?",answer:"Applied to PAA at scale, five concepts that drive results: <strong>query intent mapping</strong> (which questions signal purchase-readiness), <strong>zero-volume question discovery</strong> (PAA surfaces questions that keyword tools miss entirely), <strong>PAA tree traversal</strong> (one seed keyword branches into hundreds of related questions), <strong>freshness signaling</strong> (content updated within 90 days appears 4.3\xD7 more frequently in PAA features), and <strong>structured data delivery</strong>. The last four require programmatic access. The freshness multiplier is the most underused lever in PAA strategy \u2014 most teams optimize the answer once and move on, missing the ongoing recency advantage that recurring updates deliver."},{id:"q5-2",num:"5.2",question:"What are the 3 pillars of SEO?",answer:"At programmatic scale, the three operational pillars become <strong>data acquisition, content production, and distribution</strong> \u2014 not the traditional crawlability, content, and authority. PAA harvesting via API sits at the data-acquisition layer, upstream of every content and publishing decision. MCP Scraper operates at that layer: it does not write content or build links, it supplies the question data that makes content decisions defensible rather than intuitive. The scope boundary is a trust signal: a tool that claims to do everything does nothing well. The data acquisition layer is the one most content teams have not built yet \u2014 and it is the layer that compounds."},{id:"q5-3",num:"5.3",question:"What are the 3 C's of SEO?",answer:'Content, code, and credibility \u2014 but in a programmatic PAA workflow, <strong>"content" starts upstream with machine-readable question data</strong>, not a brainstorming session. The gap between "which questions should we answer" and "draft created" collapses when PAA API output feeds directly into a content brief template or LLM prompt. The manual research phase that typically takes days becomes a <strong>sub-second API call</strong>. The "code" pillar is what enables this \u2014 a REST endpoint that accepts a seed keyword and returns a structured PAA tree is not a luxury for large teams; it is the unlock that makes content operations at any scale less dependent on individual research time.'},{id:"q5-4",num:"5.4",question:"What is the Google 20% rule?",answer:"Google's 20% rule \u2014 the practice of giving engineers discretionary time for side projects \u2014 produced features including Gmail and Google Maps. <strong>PAA itself emerged from the same underlying logic</strong>: Google continuously mines search session data to predict follow-up intent. That mining is live and ongoing, which is why PAA questions shift by location, device, language, and time \u2014 and why scraping them once and filing the results is not a strategy. <em>Recurring harvesting on a schedule is.</em> The operational implication: build a workflow that pulls PAA data for your priority keyword set on a monthly or weekly cadence, and treat the outputs as a live editorial signal, not a one-time research deliverable."}],interactiveHtml:`<div class="si si-checklist"><span class="si-heading">PAA pipeline setup</span><div class="si-check-progress-row"><div class="si-check-bar-wrap"><div class="si-check-bar" id="sip5-bar" style="width:0%"></div></div><span class="si-check-count" id="sip5-count">0 / 5 done</span></div><ul class="si-check-list"><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Define your seed keyword list \u2014 start with your top 50 pages by traffic</label></li><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Choose your extraction method: manual for ≤10 keywords, API for 50+</label></li><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Classify PAA output by intent type \u2014 prioritize purchase-intent questions first</label></li><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Set a freshness schedule \u2014 monthly for competitive topics, quarterly for stable ones</label></li><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Connect PAA output to your content brief template or LLM prompt</label></li></ul><script>(function(){var checks=document.querySelectorAll('.sip5-chk'),bar=document.getElementById('sip5-bar'),count=document.getElementById('sip5-count'),total=checks.length;function upd(){var done=document.querySelectorAll('.sip5-chk:checked').length;bar.style.width=(done/total*100)+'%';count.textContent=done+' / '+total+' done';}checks.forEach(function(c){c.addEventListener('change',upd);});})()</script></div>`},{id:"s6",num:"06",title:"Practical PAA",titleItalic:"Tactics",deck:"Quick wins, freshness mechanics, anti-bot realities, and why the programmatic path is where SEO income separates.",cards:[{id:"q6-1",num:"6.1",question:"What are basic SEO skills?",answer:"The baseline PAA skill set is: <strong>reading query intent accurately</strong>, <strong>writing concise Q&A answers</strong> (40\u201350 words is the PAA sweet spot), <strong>implementing FAQ and HowTo schema markup</strong>, and <strong>maintaining content freshness</strong> \u2014 pages updated within 90 days appear 4.3\xD7 more frequently in PAA features than stale content. The advanced skill is API integration for teams transitioning to programmatic workflows. The basics above are achievable without any tooling beyond Google Search Console, and they compound: intent-matched 40-word answers with correct schema and a quarterly update cadence will outperform longer, less-structured content on PAA placements across nearly every category."},{id:"q6-2",num:"6.2",question:"Is SEO a high income skill?",answer:'PAA-driven SEO is splitting into two earning brackets. <strong>Practitioners who optimize content manually</strong> \u2014 identifying questions, formatting answers, checking rankings \u2014 are doing work that AI tools increasingly replicate, which compresses rates. <strong>Practitioners who build PAA data pipelines</strong> and integrate them into content systems are doing technical-strategic work that remains rare. The programmatic path is where income separates, not because it is harder to learn, but because few people have crossed the data-engineering threshold yet. The "data-engineering threshold" is lower than it sounds: REST API literacy, basic JSON handling, and familiarity with content pipeline tooling. The gap is not a skills gap \u2014 it is an awareness gap.'},{id:"q6-3",num:"6.3",question:"Why is Google checking if I'm human?",answer:"CAPTCHA and anti-bot verification appear when collecting PAA data without proper infrastructure. <strong>Headless browsers sending high-frequency requests</strong> without realistic session behavior, residential proxy coverage, or rate limiting trigger Google's bot-detection systems. MCP Scraper's API handles the anti-bot layer on the infrastructure side \u2014 the caller passes a keyword and receives structured PAA output; the detection friction never reaches the application layer. This is one of the most underestimated friction points in programmatic PAA collection: teams that build their own scrapers spend disproportionate engineering time on detection evasion rather than on the content strategy that the data enables."},{id:"q6-4",num:"6.4",question:"Am I being monitored by Google?",answer:"Google monitors search behavior continuously to update PAA questions \u2014 <strong>session patterns, query sequences, click data, and location signals</strong> all feed the PAA algorithm in real time. This is why the same keyword returns different PAA questions depending on location, language, device, and time of day. It also explains why a static PAA dataset goes stale: the questions Google surfaces this week may differ meaningfully from last month's harvest. <em>Recurring collection is the correct workflow for any live content operation.</em> The monitoring is a feature, not a surveillance concern \u2014 it means PAA is a continuously refreshed intent signal, and the teams harvesting it regularly have a permanently updated editorial dataset that teams relying on one-time research do not."}],interactiveHtml:`<div class="si si-codegen"><span class="si-heading">Generate your PAA API request</span><div class="si-codegen-inputs"><label class="si-label">Keyword<input class="si-text-input" id="sip6-kw" value="people also ask SEO" placeholder="e.g. best CRM software"></label><label class="si-label">Max questions<input class="si-text-input" id="sip6-mq" value="50" placeholder="50" type="number" min="1" max="200"></label></div><div class="si-codegen-output-wrap"><pre class="si-codegen-pre"><code id="sip6-out"></code></pre><button class="si-copy-btn" id="sip6-copy">Copy curl</button></div><script>(function(){function getKw(){return document.getElementById('sip6-kw').value||'your keyword';}function getMq(){return+(document.getElementById('sip6-mq').value||50);}function generate(){return JSON.stringify({query:getKw(),maxQuestions:getMq()},null,2);}function upd(){document.getElementById('sip6-out').textContent=generate();}document.querySelectorAll('#sip6-kw,#sip6-mq').forEach(function(i){i.addEventListener('input',upd);});document.getElementById('sip6-copy').addEventListener('click',function(){var body=JSON.stringify({query:getKw(),maxQuestions:getMq()});var cmd="curl -X POST https://mcpscraper.dev/harvest/sync -H 'Content-Type: application/json' -H 'x-api-key: YOUR_API_KEY' -d '"+body+"'";navigator.clipboard.writeText(cmd).then(function(){var btn=document.getElementById('sip6-copy');btn.textContent='Copied!';setTimeout(function(){btn.textContent='Copy curl';},1500);});});upd();})()</script></div>`}]},{slug:"what-is-an-mcp",title:"What Is an MCP? The Honest Developer's Guide",titleDisplay:"What Is",titleDisplayItalic:"an MCP?",description:"MCP explained for developers who need to decide \u2014 not just understand. Which platforms adopted it, when to skip it, and how it compares to REST, Zapier, and Copilot.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["MCP","model context protocol","AI tools","developer guide","Claude"],category:"AI Development",badge:"30 questions answered",fieldGuideLabel:"Field guide",deck:"Every article about what is an MCP tells you it's a universal plug-and-play connector for AI. This one tells you which platforms actually adopted it, when to skip it entirely, and why the USB-C analogy hides the decision your architecture actually requires.",readTimeMinutes:11,stats:[{value:"6",label:"sections"},{value:"30",label:"questions answered"},{value:"97M",label:"monthly MCP SDK downloads"},{value:"0",label:"fluff"}],ctaHeading:"Build for the agent era.",ctaHeadingItalic:"MCP Scraper is an MCP-native tool \u2014 extract PAA data the way AI agents actually work.",ctaBody:"MCP Scraper gives AI agents the web data they need \u2014 PAA questions, SERP results, and page content via REST or MCP. Built for the agent-native stack.",sections:[{id:"s1",num:"01",title:"MCP",titleItalic:"Defined",deck:"What MCP actually is, what it does, and the architectural distinction every explainer skips.",callout:{eyebrow:"Quick take",heading:"MCP is not USB-C.",body:"USB-C is a hardware connector. MCP is an architectural decision \u2014 a protocol that determines whether your AI agent can discover and call tools at runtime or must be hardcoded by a developer. <strong>The analogy hides the trade-off. This section names it.</strong>"},cards:[{id:"q1-1",num:"1.1",question:"What is an MCP in AI?",answer:"MCP \u2014 <strong>Model Context Protocol</strong> \u2014 is an open standard launched by Anthropic on November 25, 2024, that gives AI agents a single, standardized interface to connect with external tools, data sources, and workflows. Before MCP, every AI model and every external tool required a custom integration written by a developer. MCP eliminates that custom work by defining a shared communication contract that any agent and any tool can implement once and then use interchangeably. <em>An agent calling GitHub, Postgres, and Slack</em> no longer needs three bespoke connectors \u2014 it needs one MCP client. The practical implication most explainers miss: MCP is not primarily a developer convenience \u2014 it's what makes AI agents autonomous enough to run without a human in the loop on every tool call."},{id:"q1-2",num:"1.2",question:"What does the MCP do?",answer:"MCP enables AI agents to <strong>discover, invoke, and receive results from external tools at runtime</strong> \u2014 without a developer pre-configuring every possible tool call. The mechanism is a standard called <code>tools/list</code>: an agent sends this request to any MCP server and gets back a live manifest of every capability that server exposes, including names, descriptions, and input schemas. The agent then selects the right tool, calls it, and receives structured output \u2014 all without leaving its session. <em>Connecting an agent to a CRM, a code repo, and a web scraper</em> used to mean three separate integrations; MCP makes them all callable through the same protocol. The deeper point: MCP doesn't just save integration work \u2014 it makes tool use something the agent decides at runtime rather than something a developer hardcodes at build time."},{id:"q1-3",num:"1.3",question:"What is MCP in simple terms?",answer:`MCP is a <strong>shared language that any AI agent and any tool can both speak</strong>. Once a tool publishes an MCP server, every MCP-compatible agent \u2014 Claude, ChatGPT, Cursor, GitHub Copilot \u2014 can call it without a custom connector. The analogy that actually holds: imagine if every REST API automatically understood every other API's auth, parameter format, and response schema without any glue code. That's what MCP does for AI agents. <em>A Postgres database, a Slack workspace, and a web data service</em> all become callable through the same interface once they expose an MCP server. The distinction that matters in practice: "simple" doesn't mean effortless \u2014 the tool still has to implement the MCP server; MCP just ensures that implementation only has to happen once.`},{id:"q1-4",num:"1.4",question:"Is MCP a tool or framework?",answer:`MCP is neither \u2014 it's a <strong>protocol</strong>, which means a specification for how two systems communicate, not software you install or a framework you build on top of. Tools implement MCP (a Postgres connector that speaks MCP is a tool). Frameworks integrate MCP clients (LangChain, AutoGen). MCP itself is the rulebook those implementations follow \u2014 specifically, a JSON-RPC 2.0 message format plus a defined session lifecycle. IBM's documentation makes this distinction explicitly: MCP is not an agent framework. The practical consequence: asking "which MCP should I choose?" is like asking "which HTTP should I use?" \u2014 the protocol is the same; your choice is which server or client library implements it.`},{id:"q1-5",num:"1.5",question:"What problems does MCP solve?",answer:"MCP solves the <strong>N\xD7M integration problem</strong>: without it, connecting N AI models to M external tools produces N\xD7M custom integrations, each built and maintained separately. MCP collapses this to M+N \u2014 implement the protocol once on each side and every combination works. The secondary problems it solves flow from the same root: inconsistent auth patterns across integrations, brittle hardcoded tool calls that break when APIs change, and the inability for AI agents to discover new tools at runtime. <em>An enterprise with 10 AI models and 50 internal tools</em> faces 500 custom integrations without MCP and 60 protocol implementations with it. What no explainer says plainly: MCP doesn't reduce the number of tools you build \u2014 it reduces the number of connectors that break when either side changes."},{id:"q1-6",num:"1.6",question:"What is Anthropic?",answer:"Anthropic is the AI safety company that created Claude and launched MCP on November 25, 2024. <strong>In December 2025, Anthropic donated MCP governance to the Agentic AI Foundation under the Linux Foundation</strong> \u2014 a move that made MCP formally vendor-neutral and accelerated adoption by removing the perception that the protocol was an Anthropic-controlled standard. Anthropic was founded in 2021 by former OpenAI researchers, including Dario Amodei and Daniela Amodei. Its two primary contributions to AI infrastructure are Claude (a family of large language models) and MCP (the protocol this post covers). The governance transfer is the detail most explainers skip: Anthropic no longer controls MCP \u2014 the Linux Foundation body does, which is why OpenAI, Microsoft, and Google were willing to adopt it."}]},{id:"s2",num:"02",title:"MCP vs",titleItalic:"Everything",deck:"MCP vs REST, HTTP, Zapier, Copilot, and LLMs \u2014 the comparisons that actually determine whether MCP belongs in your stack.",callout:{eyebrow:"Key distinction",heading:"REST serves developers. MCP serves AI agents.",body:"That one sentence determines whether MCP belongs in your architecture. If the caller is a developer writing code, use REST. If the caller is an agent making runtime decisions, MCP earns its overhead. <strong>The mistake is treating them as competing choices rather than different interface layers.</strong>"},cards:[{id:"q2-1",num:"2.1",question:"What is an MCP vs API?",answer:"REST APIs serve developers; <strong>MCP serves AI agents</strong> \u2014 and that distinction determines which belongs in your system. A REST API is a stateless HTTP endpoint with static documentation a human reads and then hardcodes calls against. MCP is JSON-RPC 2.0 over a persistent session, with a live tool manifest the agent reads at runtime and uses to decide what to call. The critical difference: with REST, a developer writes the integration once and it breaks when the API changes; with MCP, the agent re-discovers the tool manifest on every session and adapts. MCP doesn't replace REST \u2014 MCP servers use REST internally. What changes is the interface layer above: REST is for humans integrating systems; MCP is for agents choosing tools."},{id:"q2-2",num:"2.2",question:"Why MCP instead of REST API?",answer:"Use MCP instead of REST when the consumer of the API is an AI agent making runtime decisions \u2014 not a developer writing hardcoded calls. <strong>MCP's <code>tools/list</code> endpoint lets an agent discover what a server can do without any human-written glue code</strong>. With REST, someone has to read the OpenAPI spec and write the integration; with MCP, the agent reads the live capability manifest and writes the call itself. The answer changes if you control both ends: if you own the API and the code calling it, REST is simpler and faster. MCP earns its overhead when the caller is an autonomous agent that needs to self-direct across a changing tool landscape \u2014 the moment you want the agent to decide which tool to use, not just execute the one you told it to."},{id:"q2-3",num:"2.3",question:"Why use MCP instead of HTTP?",answer:"Raw HTTP has no standard for <strong>tool discovery, session state, or AI-native authentication</strong> \u2014 MCP adds all three on top of HTTP. An agent calling raw HTTP endpoints has to know the URL, method, parameters, and auth scheme in advance, hardcoded. An MCP server exposes a <code>tools/list</code> manifest so the agent discovers capabilities dynamically, maintains state across a session, and uses a standardized OAuth 2.1 flow for auth rather than each API's bespoke scheme. The practical failure mode of raw HTTP at scale: when the tool landscape changes \u2014 a new endpoint, a deprecated parameter \u2014 every hardcoded HTTP call breaks silently; MCP's session-level capability manifest surfaces changes the agent can adapt to. Building on raw HTTP for agents is not wrong at toy scale \u2014 it fails at production scale when the number of tools grows past what any developer can maintain manually."},{id:"q2-4",num:"2.4",question:"Is MCP like Zapier?",answer:"MCP and Zapier solve adjacent problems with different audiences. <strong>Zapier automates workflows between apps for non-technical users; MCP is a protocol for AI agents to call tools programmatically</strong>. Zapier's model is: a human configures a trigger-and-action workflow once; Zapier runs it. MCP's model is: an AI agent discovers available tools at runtime and decides which ones to call. The two are not mutually exclusive \u2014 Zapier built MCP support on top of its library of 9,000+ apps and 30,000+ actions, meaning an AI agent with an MCP client can now reach every Zapier-connected app without the human workflow-configuration step. The practitioner nuance: Zapier MCP is useful when you want AI agent access to Zapier's app coverage without building individual MCP servers for each app."},{id:"q2-5",num:"2.5",question:"What is the difference between MCP and Copilot?",answer:`GitHub Copilot is an AI coding assistant that now operates as an <strong>MCP client</strong> \u2014 MCP is the protocol Copilot uses to reach external tools, not a competitor to Copilot. The relationship: Copilot is software running in VS Code or a browser; when Copilot needs to call an external tool (a Jira ticket, a GitHub repo, a web search), it does so through MCP. Before MCP, Copilot's tool integrations were custom connectors maintained by Microsoft. With MCP, any tool that publishes an MCP server becomes callable by Copilot without Microsoft writing a dedicated integration. The distinction developers miss: evaluating "MCP vs. Copilot" is a category error \u2014 Copilot is an adopter of MCP, not an alternative to it.`},{id:"q2-6",num:"2.6",question:"What is the difference between MCP and LLM?",answer:"An LLM is the <strong>reasoning engine</strong> \u2014 the model that reads input, generates text, and makes decisions. MCP is the protocol that gives that engine hands. Without MCP (or a comparable integration layer), an LLM can only work with what's in its context window \u2014 text and pre-loaded data. With MCP, the LLM can call tools at runtime: retrieve a live database record, execute a search query, write a file, trigger a workflow. <em>Claude 3.5 Sonnet reasoning about a customer support ticket</em> is an LLM at work; Claude calling a CRM to retrieve the customer's history mid-conversation is MCP at work. The frame that matters: LLMs decide; MCP acts. A sophisticated AI agent needs both."},{id:"q2-7",num:"2.7",question:"Is MCP just a JSON?",answer:`MCP uses <strong>JSON-RPC 2.0</strong> as its message format, but calling it "just JSON" misses the protocol. JSON-RPC 2.0 defines the envelope \u2014 request IDs, method names, parameters, error codes. MCP adds on top of that a defined session lifecycle: initialization handshake \u2192 capability negotiation \u2192 tool discovery \u2192 tool calls \u2192 termination. JSON alone specifies none of that structure. The distinction matters when debugging: a malformed MCP session isn't a JSON syntax error \u2014 it's a lifecycle state error, which requires understanding the protocol's state machine, not just validating JSON. The practitioner test: if your MCP server returns valid JSON but ignores the initialization handshake, every MCP client will reject it \u2014 not because the JSON is wrong, but because the protocol contract is broken.`}]},{id:"s3",num:"03",title:"MCP Adoption \u2014 Who",titleItalic:"Actually Uses It",deck:"Which platforms adopted MCP, which haven't, and the one misconception that sends most developers to the wrong conclusion.",callout:{eyebrow:"Highest-value misconception",heading:"MCP is not only for Claude.",body:"MCP has 300+ clients as of March 2026 \u2014 including ChatGPT, GitHub Copilot, VS Code, Cursor, and Google Gemini. Anthropic created it. The Linux Foundation now governs it. <strong>Write one MCP server. Every major AI platform can call it.</strong>"},cards:[{id:"q3-1",num:"3.1",question:"Is MCP only for Claude?",answer:"No \u2014 and this is the highest-value misconception in the MCP ecosystem. <strong>MCP has 300+ clients as of March 2026</strong>, including ChatGPT, GitHub Copilot, VS Code, Cursor, Windsurf, AWS Bedrock, Google Gemini, and JetBrains IDEs. Anthropic created MCP but donated its governance to the Agentic AI Foundation under the Linux Foundation in December 2025, making it a vendor-neutral standard no single company controls. OpenAI formally adopted MCP in March 2025 \u2014 four months after Anthropic launched it. The practical implication for developers: an MCP server you build today is reachable by Claude, ChatGPT, and GitHub Copilot without any modification \u2014 write the server once, serve every major agent platform."},{id:"q3-2",num:"3.2",question:"Does ChatGPT use MCP?",answer:"Yes \u2014 <strong>OpenAI formally adopted MCP in March 2025</strong> and added MCP support to ChatGPT apps in September 2025. OpenAI's adoption removed the last credible argument that MCP was a Claude-exclusive or Anthropic-controlled standard. When OpenAI adopted MCP, the protocol had an estimated 22 million monthly SDK downloads; by March 2026 that figure reached 97 million. ChatGPT is now one of more than 300 MCP clients \u2014 meaning any MCP server you build is natively callable from ChatGPT without any OpenAI-specific integration work. The sequence developers should know: Anthropic launched \u2192 OpenAI adopted \u2192 Microsoft integrated \u2192 Google followed \u2014 MCP's cross-vendor adoption happened in under 18 months."},{id:"q3-3",num:"3.3",question:"Does Microsoft use MCP?",answer:"Yes \u2014 Microsoft integrated MCP into <strong>GitHub Copilot, Visual Studio Code, and Copilot Studio</strong>. The GitHub Copilot extension in VS Code is among the most widely used MCP clients in the developer tooling ecosystem. Microsoft's adoption matters structurally: it brought MCP into enterprise environments at scale, since VS Code and GitHub Copilot are standard tooling in most engineering organizations. The consequence for MCP server builders: publishing an MCP server means your tool is callable from VS Code's AI features \u2014 the IDE that runs on more developer machines than any other. Microsoft's integration also means MCP is no longer a decision individual developers make; it's a platform decision that enterprise engineering teams inherit from their tooling."},{id:"q3-4",num:"3.4",question:"Is Apple using MCP?",answer:"Apple has not made a public MCP announcement as of May 2026. <strong>The honest answer is: unknown, but structurally likely.</strong> MCP clients that run on Apple platforms \u2014 Claude Desktop, VS Code, Cursor \u2014 are in wide use on macOS. If Apple builds agentic AI features into iOS or macOS, adopting MCP would give it instant interoperability with every existing MCP server ecosystem rather than requiring Apple to build a proprietary tool integration standard from scratch. The precedent: every other major AI platform that initially appeared absent from MCP (OpenAI, Microsoft, Google) has since formally adopted it. The practitioner read: Apple's silence is not a rejection \u2014 it's the gap between enterprise announcement cycles and protocol adoption reality."},{id:"q3-5",num:"3.5",question:"Is Zapier MCP free?",answer:`Yes \u2014 <strong>Zapier MCP is included on all Zapier plans, including the Free tier</strong>, at no additional cost. There is no separate product SKU. The only cost is task consumption: each MCP tool call uses 2 tasks from your existing Zapier task quota. A developer on Zapier's free plan can connect an AI agent to Zapier's 9,000+ app library and 30,000+ actions today, using their existing task allocation. The nuance that changes the math: "free" means no incremental charge, not zero cost \u2014 if an agent makes 500 MCP tool calls in a month, that consumes 1,000 tasks from your quota. High-volume agent workflows will exhaust free-tier quotas quickly and require a paid plan.`}]},{id:"s4",num:"04",title:"When MCP",titleItalic:"Breaks Down",deck:"When not to use MCP, why production projects stall, and the honest comparison to adjacent tools.",callout:{eyebrow:"Practitioner test",heading:"MCP has no native auth.",body:"The spec recommends OAuth 2.1 with PKCE. Your MCP server only has it if you built it. <strong>Every production MCP project that stalled did so at auth, not at the protocol itself.</strong>"},cards:[{id:"q4-1",num:"4.1",question:"When not to use MCP?",answer:"Skip MCP when your system is <strong>developer-to-API rather than agent-to-tool</strong>. If a human developer is writing the integration code, REST is simpler and adds no session-management overhead. Skip MCP when you control both ends of the integration and don't need runtime discovery \u2014 if you own the calling code and the tool, you already know what the tool does; the <code>tools/list</code> handshake is unnecessary overhead. Skip MCP when latency is the primary constraint \u2014 persistent sessions add round-trip initialization costs that stateless REST calls don't. Skip MCP when your AI use case is inference-only: if the model generates text without calling any external system, MCP adds complexity with zero benefit. The practitioner test: if a human could write the integration code once and it would never need to change, use REST."},{id:"q4-2",num:"4.2",question:"Why are people moving away from MCP?",answer:`The most common friction points are auth complexity, debugging difficulty, and server quality variance. <strong>MCP has no native authentication</strong> \u2014 developers must implement OAuth 2.1 with PKCE themselves, and the gap between "runs locally" and "ships securely to production" is where most MCP projects stall. Debugging is harder than REST because persistent sessions have lifecycle state: a broken MCP connection isn't a failed HTTP request \u2014 it's a state machine that failed at initialization, capability negotiation, or mid-session, and the error may not surface clearly. The 10,000+ public MCP servers have wildly inconsistent quality \u2014 some are maintained production services; many are experimental projects with no uptime guarantees. "Moving away" overstates it: developers who understand MCP's limits ship successfully; the ones who expected plug-and-play get surprised by the operational requirements.`},{id:"q4-3",num:"4.3",question:"Is A2A dead?",answer:"A2A \u2014 Google's Agent-to-Agent protocol \u2014 is not dead, but it has not achieved the ecosystem density MCP has. <strong>MCP and A2A address different layers</strong>: MCP handles tool access (an agent calling an external capability); A2A handles agent coordination (one agent delegating a task to another agent). They are complementary, not competing. The adoption gap is real: MCP has 97 million monthly SDK downloads and 300+ clients as of March 2026; A2A's ecosystem is smaller by every public measure. The honest framing: A2A is the correct protocol for multi-agent orchestration problems; MCP is the correct protocol for tool-calling problems. A developer who needs both can use both \u2014 and the most sophisticated agent architectures will."},{id:"q4-4",num:"4.4",question:"Will MCP replace API?",answer:`No \u2014 <strong>MCP wraps APIs rather than replacing them</strong>. MCP servers use REST APIs internally; they expose MCP above and call REST below. The correct model: REST remains the implementation layer; MCP becomes the interface layer that AI agents interact with. This isn't a philosophical position \u2014 it's how production MCP servers are built. <em>An MCP server for Stripe</em> doesn't replace Stripe's REST API; it wraps it, adding the <code>tools/list</code> manifest and session management that AI agents expect. The question developers should ask instead: not "will MCP replace REST?" but "at which layer does MCP belong in my stack?" \u2014 the answer is always above your existing APIs, never instead of them.`},{id:"q4-5",num:"4.5",question:"Why are people against Copilot?",answer:"Copilot critics object to Microsoft's <strong>pricing model, data training practices, and the perception that Copilot is a productivity layer rather than a reasoning upgrade</strong>. Common complaints include the cost per seat relative to perceived productivity gains, concerns about code written in Copilot being used to train future models, and the view that Copilot autocompletes rather than reasons. These criticisms are entirely separate from MCP \u2014 MCP is the protocol Copilot uses to reach external tools; it doesn't change Copilot's pricing, training data policies, or reasoning depth. The relevant clarification for developers evaluating both: being against Copilot doesn't mean avoiding MCP \u2014 every other major AI platform (Claude, ChatGPT, Cursor) also uses MCP and has none of Copilot's specific controversies."}]},{id:"s5",num:"05",title:"MCP Architecture",titleItalic:"& Security",deck:"Transport layers, encryption, auth, and the gap between what MCP promises and what your server actually ships.",callout:{eyebrow:"Security gap",heading:"MCP won't refuse an HTTP connection.",body:"The spec recommends HTTPS. Enforcement is the developer's job. Most MCP tutorials run over HTTP for simplicity \u2014 developers copy that configuration to production and ship an insecure server without any warning from the protocol. <strong>TLS is your responsibility, not MCP's.</strong>"},cards:[{id:"q5-1",num:"5.1",question:"Is MCP built on HTTP?",answer:"MCP supports two transport layers: <strong>stdio for local in-process communication and HTTP with Server-Sent Events (SSE) for remote connections</strong>. Local MCP servers \u2014 the kind that run on a developer's machine alongside Claude Desktop \u2014 use stdio, which is faster and simpler because the agent and server are in the same process space. Remote production MCP servers use HTTP/SSE, which enables cross-network communication but requires TLS since MCP has no native encryption layer. Most production MCP servers use the HTTP transport \u2014 it's what makes an MCP server callable from any agent on any machine. The choice isn't either-or: a single MCP server implementation can support both transports, but most developers building for production start with HTTP/SSE."},{id:"q5-2",num:"5.2",question:"Does MCP use HTTP or HTTPS?",answer:"MCP supports both, but <strong>the spec explicitly recommends HTTPS for any remote server</strong> \u2014 and running an MCP server over plain HTTP in production is a security vulnerability, not a configuration choice. MCP does not natively enforce encryption; TLS must be configured by the developer. The risk is concrete: MCP sessions carry tool call parameters and responses over a persistent connection \u2014 plain HTTP exposes that entire session to interception. Local development over localhost HTTP is acceptable (the connection doesn't leave the machine); any server exposed over a network must use HTTPS. The gap most teams hit: MCP tutorials run over HTTP for simplicity; developers copy that configuration to production and ship an insecure server without realizing the protocol never warned them."},{id:"q5-3",num:"5.3",question:"Does MCP use OAuth?",answer:`MCP standardizes <strong>OAuth 2.1 with PKCE</strong> for authentication in remote server connections \u2014 but this is not built into the base protocol. OAuth support was added to the MCP spec after initial launch, when real-world adoption revealed auth as the most common production gap. Implementing it requires developers to configure an OAuth 2.1 server, handle PKCE flows, and manage token refresh \u2014 none of which MCP handles automatically. Local MCP servers running over stdio typically require no auth because they run in a trusted local environment. Remote servers require auth, and OAuth 2.1 with PKCE is the spec-recommended approach. The practitioner reality: "MCP supports OAuth" means MCP defines how OAuth should work in its context \u2014 it doesn't mean your MCP server has OAuth until you build it.`},{id:"q5-4",num:"5.4",question:"Is an MCP server just an API?",answer:"An MCP server looks like an API from the outside but differs in three structural ways. First, it exposes a <strong>standard capabilities manifest via <code>tools/list</code></strong> \u2014 a traditional API has static docs; an MCP server has a live, machine-readable manifest the agent queries at runtime. Second, it maintains session state across multiple calls within a connection \u2014 REST APIs are stateless by design; MCP sessions are stateful. Third, it uses JSON-RPC 2.0 rather than REST conventions \u2014 request/response patterns, error codes, and method naming all follow JSON-RPC semantics, not HTTP verb conventions. <em>Calling an MCP server</em> and <em>calling a REST API</em> look superficially similar from a network perspective; they're architecturally different contracts that break in different ways when misused."}]},{id:"s6",num:"06",title:"MCP Ecosystem",titleItalic:"& What's Next",deck:"Power Automate, Google's investment, and why MCP is a layer above APIs rather than a replacement for them.",cards:[{id:"q6-1",num:"6.1",question:"What has replaced Power Automate?",answer:"Nothing has fully replaced Power Automate \u2014 it remains Microsoft's enterprise workflow product and continues to serve the structured, if-then automation use cases it was built for. <strong>What MCP and AI agents are taking from Power Automate is the dynamic, decision-driven tier of automation</strong> \u2014 tasks that require reasoning, not just routing. Power Automate routes data between systems based on rules a human writes; an MCP-connected agent can decide which tools to call, handle edge cases without pre-written rules, and adapt to inputs that a static workflow would reject. <em>Routing a support ticket to the right queue</em> is Power Automate's domain; <em>reading the ticket, checking the customer's history, drafting a response, and escalating if needed</em> is where MCP-connected agents outperform static workflow tools. The transition isn't replacement \u2014 it's a shift in which automation problems belong in which category."},{id:"q6-2",num:"6.2",question:"Does Google own 14% of Anthropic?",answer:"Google has invested significantly in Anthropic across multiple rounds, but <strong>the exact ownership percentage is not publicly disclosed</strong>. Reports indicate Google invested $300 million in a 2023 funding round and participated in subsequent rounds. The precise ownership stake \u2014 including whether it is or was 14% \u2014 has not been confirmed in any public filing. What is confirmed: Google Cloud and Anthropic have a partnership that includes MCP integration into Google's Gemini models and Google Cloud services. The relevant fact for MCP evaluation: Google's investment in Anthropic did not prevent Google from independently adopting MCP \u2014 both Google Gemini and Google Cloud services are listed among MCP clients, and adoption was driven by the protocol's open governance under the Linux Foundation, not by equity relationships."},{id:"q6-3",num:"6.3",question:"Is MCP basically an API?",answer:"MCP is best understood as a <strong>layer above APIs, not a replacement for them</strong>. Saying MCP is basically an API is like saying HTTP is basically a phone call \u2014 technically there's a connection, but the architectural purpose is different. Traditional APIs serve developers who know what they want to call and write the call in advance. MCP serves AI agents that discover what's available at runtime and decide what to call dynamically. The difference shows up at scale: an API breaks when you add a new endpoint that no existing code knows to call; an MCP server's <code>tools/list</code> response automatically surfaces the new capability to every connected agent on the next session. MCP Scraper is an example of what this makes possible \u2014 a web data tool built not for developers to integrate manually, but for AI agents to discover and call directly, the way MCP was designed to work."}]}]},{slug:"when-not-to-use-an-mcp",title:"When Not to Use an MCP: The Architectural Decision Guide",titleDisplay:"When Not to Use",titleDisplayItalic:"an MCP.",description:"Not complexity\u2014architecture. The three signals that disqualify MCP, the A2A and function-calling alternatives, and the protocol durability question for 2026.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["MCP","AI","protocol","architecture","agents"],category:"MCP Protocol",badge:"36 questions answered",fieldGuideLabel:"Field guide",deck:"The complexity calculus is the wrong frame. In 2026, the real question is whether MCP will still be the dominant protocol when your integration ships \u2014 and that requires reading the ecosystem, not the docs.",readTimeMinutes:15,stats:[{value:"5",label:"sections"},{value:"36",label:"questions"},{value:"3",label:"skip signals"},{value:"0",label:"fluff"}],ctaHeading:"Read the ecosystem with",ctaHeadingItalic:"live PAA data.",ctaBody:"MCP Scraper harvests People Also Ask questions at scale \u2014 so your MCP decision is based on what practitioners are actually searching right now, not what the docs say they should ask.",sections:[{id:"s1",num:"01",title:"What MCP",titleItalic:"Actually Is.",deck:"Every article about MCP tells you it's a protocol that lets AI models call tools. That definition is technically accurate and practically useless \u2014 it skips the problem MCP was invented to solve, which is the only thing that tells you whether you need it at all.",cards:[{id:"q1-1",num:"1.1",question:"What is an MCP in AI?",answer:`MCP (Model Context Protocol) is an open standard that solves the N\xD7M integration problem \u2014 the combinatorial explosion that happens when M AI models each need custom connectors to N external tools. Before MCP existed, connecting three AI models to ten tools required up to thirty custom integrations; MCP standardizes the interface so any compliant model can call any compliant server without custom work. <strong>The official description is "like a USB-C port for AI applications" \u2014 a single standardized connector that replaces a sprawl of proprietary cables.</strong> The current stable specification (2025-11-25) is built on JSON-RPC 2.0, defines three roles (Hosts, Clients, Servers), and is supported across Claude, ChatGPT, Visual Studio Code, and Cursor. The question to ask before adopting MCP is not "what is it" but "do I have an N\xD7M problem worth solving at the protocol layer" \u2014 if you have one model and one tool, you don't.`,source:"https://modelcontextprotocol.io"},{id:"q1-2",num:"1.2",question:"What is MCP in simple terms?",answer:`MCP is the standardized language that lets an AI model ask an external tool to do something \u2014 and get a structured answer back \u2014 without either side needing to know how the other was built. <strong>Think of it as a universal remote control for AI agents: instead of each AI building its own custom remote for each device it wants to control, MCP gives every device a standard input jack and every remote a standard output plug.</strong> In practice, an MCP server exposes a list of "tools" (callable functions) with descriptions the AI model reads. The model decides which tool to call and passes structured arguments; the server executes the operation and returns a result. What makes this worth a protocol is that the model doesn't need to be rewritten when the tool changes, and the tool doesn't need to be rewritten when a new AI model is added. A practitioner who has shipped MCP integrations will tell you the simplification is real at scale \u2014 and essentially invisible for single-tool, single-model use cases.`,source:"https://modelcontextprotocol.io"},{id:"q1-3",num:"1.3",question:"What problems does MCP solve?",answer:'MCP solves two structural problems that emerge when AI systems grow beyond a single model connected to a single tool: the N\xD7M connector problem and the capability-discovery problem. <strong>The N\xD7M connector problem is architectural: without a standard, every new AI model requires custom integration code for every tool it needs to call \u2014 the cost scales multiplicatively, not additively.</strong> The capability-discovery problem is subtler: before MCP, a model had to know at design time exactly what an external tool could do; with MCP, the server advertises its capabilities dynamically at session start, so the model can adapt to whatever tools are available. Both problems are irrelevant when you have one model and one tool with a stable interface \u2014 which is why "what problems does MCP solve" is also the correct frame for "when not to use MCP." If neither problem applies to your current system, the protocol layer adds overhead without payoff.',source:"https://modelcontextprotocol.io/specification/2025-11-25"},{id:"q1-4",num:"1.4",question:"What does the MCP do?",answer:"MCP manages the full lifecycle of a tool-calling session between an AI model and an external server: capability negotiation at connection, tool invocation during the session, and structured result delivery back to the model. <strong>The stateful session is MCP's defining feature \u2014 unlike a REST API call where each request is independent, an MCP session maintains context across multiple tool calls, so the model can use the output of one tool as the input to the next without the orchestration living inside the model itself.</strong> In concrete terms: when a Claude instance connects to an MCP server, the server first sends a list of available tools with descriptions. Claude reads those descriptions, decides which tool to call, sends a structured JSON-RPC request, and receives a structured result. The session stays open so Claude can call additional tools without re-authenticating or re-negotiating capabilities. For single-tool, single-call integrations, this session overhead is pure cost.",source:"https://modelcontextprotocol.io/specification/2025-11-25"},{id:"q1-5",num:"1.5",question:"Is MCP a tool or framework?",answer:`MCP is a protocol \u2014 not a tool, not a framework, and not a library. <strong>A protocol defines the rules for how two parties communicate; it does not prescribe how either party is implemented, which is what makes MCP portable across AI models and tool servers built in different languages and architectures.</strong> The practical consequence is that calling MCP a "framework" is a category error that leads to wrong architectural expectations: frameworks come with opinions about structure, abstractions, and project layout. MCP has none of those \u2014 it defines message formats, session lifecycle, and capability negotiation only. If you need a framework to build MCP clients or servers, you use an SDK (Anthropic provides SDKs for Python and TypeScript); the SDK is the framework layer on top of the protocol. The distinction matters when evaluating adoption cost: you're not adopting a framework with its opinionated structure, you're implementing a protocol that can live inside whatever structure you already have.`,source:"https://modelcontextprotocol.io/specification/2025-11-25"},{id:"q1-6",num:"1.6",question:"What is the difference between MCP and LLM?",answer:"An LLM (Large Language Model) is the AI system that reasons and generates text; MCP is the protocol that tells the LLM how to interact with external tools. <strong>The relationship is one-directional: LLMs use MCP \u2014 MCP does not use or require an LLM.</strong> An LLM without MCP can only work with information it was trained on and whatever appears in its context window. An LLM with MCP can call external servers to retrieve current data, execute code, query databases, and take actions in other systems \u2014 then incorporate those results into its reasoning. MCP adds the tool-use capability; the LLM supplies the reasoning about when and how to use those tools. The confusion between MCP and LLM typically indicates someone comparing a capability (tool-calling) with the system exercising that capability (the model) \u2014 the correct comparison is MCP vs. function calling (another tool-use mechanism built into model APIs), not MCP vs. LLM."}],interactiveHtml:`<div class="si si-quiz">
|
|
1
|
+
import{a as IE,b as Kh,c as RE,d as CE,e as TE,f as PE,g as xi,h as so,i as vp,j as ei,k as Ee,l as Ep,m as vy,n as Si,o as ao,p as kp,q as Ap,s as ST}from"./chunk-ZG52SKGD.js";import{a as Yh,b as qE,d as UE,e as no,f as Zn,g as iy,h as Ak,k as pp,m as mp,n as Ik,o as Rk,p as Ck}from"./chunk-IK5BG7MO.js";import"./chunk-XG6GCEUE.js";import{A as wk,B as _k,D as vk,E as up,G as xk,I as Sk,J as Ek,K as ro,L as rc,M as kk,a as nk,d as ik,e as ok,f as sk,g as ak,h as ck,i as ny,j as Ws,k as lk,m as dk,n as uk,o as _i,q as pk,r as mk,s as gk,t as fk,u as dp,v as hk,w as yk,y as bk,z as Ol}from"./chunk-5JKBFNYF.js";import{d as dy,j as Hk,l as Wk,m as zk}from"./chunk-LSFDB5XE.js";import{A as dA,B as uA,C as pA,D as mA,E as gA,d as Xk,e as Zk,h as Qn,i as Qk,j as eA,k as py,l as tA,m as rA,n as nA,p as Ko,q as iA,r as oA,s as Go,t as sA,u as zs,v as $l,w as aA,x as my,y as cA,z as lA}from"./chunk-X7GZJU5P.js";import{a as BA,b as by,c as HA,d as WA,e as wy,f as _y,i as xr,l as Sp}from"./chunk-NLKA6SHC.js";import{$ as RC,$a as yT,A as dC,Aa as ZC,B as uC,Ba as Xy,C as pC,Ca as QC,D as mC,Da as eT,E as gC,Ea as tT,F as Gy,Fa as rT,G as fC,Ga as Zy,H as hC,Ha as nT,I as Vy,Ia as iT,J as yC,Ja as oT,K as bC,Ka as sT,L as Jy,La as aT,M as wC,Ma as cT,N as _C,Na as Qy,O as vC,Oa as eb,P as xC,Pa as lT,Q as SC,Qa as dT,R as EC,Ra as uT,S as kC,Sa as pT,T as AC,Ta as tb,U as Up,Ua as rb,V as IC,Va as Bp,W as Yo,Wa as mT,X as Vs,Xa as gT,Y as Xo,Ya as sc,Z as Hl,Za as fT,_ as Yy,_a as hT,a as Tk,aa as CC,ab as bT,b as Pk,ba as TC,bb as wT,c as Nk,ca as PC,cb as nb,d as oy,da as NC,db as _T,e as fp,ea as jp,eb as vT,f as sy,fa as LC,g as Y,ga as MC,gb as xT,h as io,ha as OC,i as hp,ia as DC,j as ay,ja as qC,k as Lk,ka as UC,l as Re,la as jC,m as zo,ma as $C,n as CA,na as $p,o as TA,oa as FC,p as _p,pa as BC,q as PA,qa as HC,r as nc,ra as Fp,s as hy,sa as WC,t as tR,ta as zC,u as Ny,ua as KC,v as rR,va as GC,w as Ei,wa as VC,x as nR,xa as JC,y as iR,ya as YC,z as lC,za as XC}from"./chunk-AFFMCO7R.js";import{b as yA,c as bA,d as wA,e as _A,f as vA,g as xA,h as xp,i as yy,j as MA,l as OA,m as DA,n as qA,o as UA,p as jA,q as $A,r as Rn,s as Vo,t as FA}from"./chunk-IJF4VXJ7.js";import{$ as Mp,$a as aC,A as yR,Aa as WR,B as bR,C as wR,Ca as Wy,D as _R,Da as zR,E as vR,Ea as KR,F as Lp,Fa as GR,G as xR,Ga as qp,H as Uy,Ha as VR,I as SR,Ia as JR,J as ER,Ja as YR,K as kR,Ka as zy,L as AR,Ma as XR,Na as ZR,Oa as QR,P as IR,Pa as eC,Q as RR,Sa as tC,Ta as rC,Ua as nC,V as CR,Va as iC,Wa as oC,Xa as oc,Ya as Ky,Z as TR,Za as sC,_ as jy,a as Ly,aa as Op,b as oR,ba as $y,bb as cC,ca as PR,d as sR,da as NR,e as aR,f as cR,fa as Fy,g as lR,ga as LR,h as dR,ha as By,i as Jo,ia as MR,j as My,ja as Bl,k as Oy,ka as OR,l as uR,la as DR,m as pR,ma as qR,n as Dy,na as UR,o as mR,oa as jR,p as gR,pa as $R,q as J,qa as Dp,r as z,ta as FR,ua as BR,v as ic,w as Gs,x as fR,y as qy,ya as HR,z as hR,za as Hy}from"./chunk-LI7WHOII.js";import{a as gp,b as Dl,c as ql,e as SA,g as EA,h as gy,k as kA,m as AA,n as IA,o as fy,p as RA}from"./chunk-QTDLZTQ7.js";import{a as ln,b as gE,c as Ji,d as Yi,e as tp,f as fE,g as Ve,h as An,i as Us,j as Ja,k as hE,n as kl,o as yE,p as bE,q as rp,r as wE,s as _E,t as vE,u as np,v as wi,w as Wo}from"./chunk-M5VZVMPC.js";import{f as Gr}from"./chunk-M5QHXNFZ.js";import{$ as vI,Aa as KI,B as YA,Ba as Ks,C as XA,Ca as GI,D as ZA,Da as VI,E as QA,Ea as JI,F as eI,Fa as Np,G as tI,Ga as Py,H as rI,Ha as YI,I as nI,Ia as XI,J as iI,K as oI,Ka as ZI,L as sI,La as QI,M as aI,Ma as eR,N as cI,O as lI,P as dI,Q as uI,R as pI,S as mI,T as Ey,U as gI,V as fI,W as hI,X as yI,Y as bI,Z as wI,_ as _I,a as dn,aa as xI,b as dr,ba as SI,c as Jn,ca as EI,d as Gh,da as kI,e as yp,ea as AI,f as qk,fa as II,g as Uk,ga as RI,h as oo,ha as CI,i as bp,j as jl,ja as TI,k as vi,ka as PI,l as jk,la as Rp,m as $k,ma as NI,n as Kk,na as ky,o as Gk,oa as Cp,p as uy,pa as LI,q as Vk,qa as MI,r as Jk,ra as OI,s as Yk,sa as DI,t as wp,ta as Ay,u as Fl,ua as Iy,v as fA,va as Ty,w as hA,wa as WI,x as NA,xa as zI,y as LA,za as Pp}from"./chunk-4FGLXXZ2.js";import{b as xE,g as Hh,i as NE,j as Zi,k as Al}from"./chunk-DT2FYN6N.js";import{a as OE,b as DE,d as Vh,f as Jh}from"./chunk-W2BVJ7S2.js";import{a as $t,b as vt,c as Pt,d as un,e as pn,f as sp,g as Xa}from"./chunk-RK2VCTZI.js";import{b as Ya,c as js,d as ip,e as op,f as $s,h as jt,i as SE,j as EE,k as At,l as Wh,m as kE,n as Xi,o as AE,p as zh,q as Fs}from"./chunk-ORB4RHCK.js";import{a as to,b as Vr,d as Ml,e as rk,f as Xn}from"./chunk-TMB56NCA.js";import{c as cy,e as Mk,g as Ok,h as Dk}from"./chunk-HUV2WTRW.js";import{a as qI,b as UI,c as ti,f as jI,h as $I,j as FI,k as Ry,l as Cy,m as BI,n as Tp,q as HI}from"./chunk-YGBTTW5D.js";import{$ as lp,A as Zh,C as BE,D as Qh,E as Za,F as HE,G as WE,H as zE,I as KE,J as ap,K as ey,L as ty,M as ry,N as eo,O as GE,P as VE,Q as JE,R as Qa,S as YE,T as XE,U as Nl,V as cp,X as ZE,Y as Ll,_ as ce,a as D,aa as ec,b as Il,ba as N,c as jE,ca as QE,d as Hs,da as ek,e as rt,ea as tk,fa as Yn,g as Rl,ga as In,ha as tc,m as Qi,n as Xh,o as Cl,p as vr,q as Tl,r as Pl,s as $E,v as FE}from"./chunk-UN6FVDJQ.js";import{a as zu,b as ae,c as Ku,d as Gu}from"./chunk-SH5KB4P7.js";import{a as tr,b as qt,d as Ut,e as ep,f as mE,g as Vn,h as kn}from"./chunk-2SP57VCG.js";import{b as ve,c as we}from"./chunk-6DTXIZY2.js";import{a as Ul,b as zA,f as KA,g as Ip,h as xy,i as Sy,j as GA,k as VA}from"./chunk-W2T4NTCE.js";import{a as Fk,b as ly,c as Bk}from"./chunk-KJQXUZ4Y.js";import{a as Kn,b as Wu,d as iS,e as oS,f as sS,g as aS,h as cS,i as lS}from"./chunk-Y6MKMSOC.js";import{a as ME}from"./chunk-WO3N5FH2.js";import{a as LE,b as fe,c as Bs}from"./chunk-HE45FFBU.js";import{a as JA}from"./chunk-SWCAKYBP.js";import{$ as _S,$a as GS,$b as cE,A as uS,Aa as NS,Ab as Bh,B as pS,Ba as LS,C as mS,Ca as MS,D as gS,Da as OS,E as fS,Ea as DS,Eb as eE,Fa as Yu,Fb as tE,G as hS,Ga as Gn,Hb as rE,Ia as qS,Ja as Oh,Ka as Dh,La as qh,Lb as nE,Ma as US,Na as jS,Nb as ie,Oa as $S,Ob as iE,Pa as FS,Pb as Zu,Q as Ju,Qa as Xu,R as gt,Ra as BS,Rb as oe,S as Bo,Sa as Uh,Sb as _t,T as tt,Ta as HS,Tb as ze,Ub as Qu,Vb as oE,W as yS,Wa as WS,X as bS,Xb as Fr,Y as Ih,Ya as zS,Yb as sE,Z as bi,Za as KS,_ as wS,_a as jh,_b as aE,a as Vu,aa as Rh,ab as VS,ac as lE,ba as vS,bb as JS,bc as dE,ca as Ch,cb as YS,cc as uE,da as Th,db as XS,dc as pE,fa as Ph,fb as ZS,ga as xS,gb as QS,h as $r,ha as Nh,hb as Sl,i as xl,j as lr,jb as _r,ka as SS,l as Ds,lb as $h,ma as ES,na as kS,oa as Ho,pa as Vt,qa as AS,ra as IS,sa as Va,t as R,ta as Lh,ua as RS,va as CS,wa as Z,wb as qs,xa as TS,ya as PS,yb as El,z as dS,za as Mh,zb as Fh}from"./chunk-WJ4XFLS4.js";var Hp=[{slug:"ai-development-workflows",title:"AI Development Workflows: A Complete Guide",titleDisplay:"AI Development",titleDisplayItalic:"Workflows.",description:"What they are, how they're built, which tools actually matter, and the mistakes most teams make before they figure it out.",publishedAt:"2026-05-23",author:"MCP Scraper",authorInitials:"M",tags:["AI","workflows","development","automation"],category:"AI Workflows",badge:"49 questions answered",fieldGuideLabel:"Field guide",deck:"What they are, how they're built, which tools actually matter, and the mistakes most teams make before they figure it out.",readTimeMinutes:12,stats:[{value:"5",label:"sections"},{value:"49",label:"questions"},{value:"4",label:"core stages"},{value:"0",label:"fluff"}],ctaHeading:"Build faster with",ctaHeadingItalic:"real data.",ctaBody:"MCP Scraper gives your AI workflows the web intelligence they need \u2014 SERP data, People Also Ask harvests, page extraction, YouTube transcripts, and more. All via API or MCP server.",sections:[{id:"s1",num:"01",title:"What is an AI",titleItalic:"Workflow?",deck:"The definition most people skip, and why skipping it costs them three months of rework.",callout:{eyebrow:"Quick take",heading:"An AI workflow is not an AI tool.",body:"A tool does one thing. A workflow connects data, model, and action into a loop that runs without you. <strong>Most teams build tools. Winners build workflows.</strong>"},cards:[{id:"q1-1",num:"1.1",question:"What is an example of an AI workflow?",answer:"An AI workflow is a connected sequence where an AI model handles one or more steps in a larger process. A concrete example: a model reads an incoming invoice, extracts line items, cross-references them against a purchase order in your database, flags discrepancies, and routes it to the right approver \u2014 all automatically. <strong>The human only touches the exceptions.</strong> That's the leverage. Single-step automations (just a prompt, just a classification) are AI tools. Workflows string those steps together with state and decision logic."},{id:"q1-2",num:"1.2",question:"What is the basic workflow of AI?",answer:"The core components of any AI workflow are: <strong>agents</strong> that perform tasks or make decisions, <strong>data pipelines</strong> that feed those agents, <strong>tool integrations</strong> that let agents take action in external systems, and <strong>feedback loops</strong> that improve outputs over time. Strip away any one of those and you have a prototype, not a workflow. The feedback loop is where most teams cut corners \u2014 and it's the one that compounds."},{id:"q1-3",num:"1.3",question:"What are the four stages of an AI workflow?",answer:"The four stages are: <strong>(1) Data input</strong> \u2014 structured or unstructured data enters the system. <strong>(2) Processing and analysis</strong> \u2014 the model interprets, classifies, or extracts. <strong>(3) Decision-making</strong> \u2014 based on the model's output, the workflow branches. <strong>(4) Output with feedback</strong> \u2014 an action is taken and the result is logged so future runs improve. Most implementations nail stages 1\u20133 and forget 4. That's why they plateau."},{id:"q1-4",num:"1.4",question:"What are the four types of workflows?",answer:"<strong>Sequential</strong> \u2014 tasks run in a fixed order. Good for predictable processes. <strong>Parallel</strong> \u2014 multiple tasks run simultaneously. Good for speed. <strong>State machine</strong> \u2014 waits for an event to transition. Good for long-running or human-in-the-loop processes. <strong>Rules-driven</strong> \u2014 conditional logic branches based on data values. Good for compliance or tiered routing. Most real AI workflows combine two or three of these."},{id:"q1-5",num:"1.5",question:"What is the AI project cycle?",answer:"The AI Project Cycle runs: <strong>problem definition \u2192 data collection \u2192 model selection \u2192 evaluation \u2192 deployment \u2192 monitoring</strong>. Deployment and monitoring take longer than most teams budget \u2014 usually 60% of total project time. Skipping proper problem definition is where 85% of AI projects fail before they start."}]},{id:"s2",num:"02",title:"Stages of",titleItalic:"Development.",deck:"How AI systems mature from prototype to production \u2014 and the adoption curve most organizations get stuck on.",cards:[{id:"q2-1",num:"2.1",question:"What are the 5 stages of AI adoption?",answer:"Five stages: <strong>Aware</strong> (experimenting with prompts), <strong>Active</strong> (running pilots), <strong>Operational</strong> (AI is a production dependency), <strong>Systemic</strong> (AI shapes how teams are structured), and <strong>Transformational</strong> (the business model itself changes). Most teams stall at Operational \u2014 they have working AI but it hasn't changed how decisions get made."},{id:"q2-2",num:"2.2",question:"What are the 5 layers of AI development?",answer:"Infrastructure \u2192 Data \u2192 Model development and operations \u2192 Application \u2192 Cross-layer governance. The cross-layer governance piece is the one that gets ignored until there's an incident. When you have AI making decisions in production, you need an audit trail, rollback capability, and ownership assignments at every layer \u2014 before something goes wrong, not after."},{id:"q2-3",num:"2.3",question:"What are the 8 stages of a workflow?",answer:"Creation \u2192 Initiation \u2192 Execution \u2192 Review \u2192 Approval \u2192 Documentation \u2192 Archival \u2192 Iteration. The Iteration stage is where AI adds disproportionate value \u2014 a workflow that can learn from its own execution history will outperform a static one within weeks."},{id:"q2-4",num:"2.4",question:"What are the 4 types of AI?",answer:'<strong>Reactive</strong> (no memory), <strong>Limited memory</strong> (uses recent context \u2014 most modern LLMs), <strong>Theory of mind</strong> (not yet achieved), <strong>Self-aware</strong> (theoretical). Every AI in production today is limited memory. The "agentic AI" hype is largely about making limited-memory systems behave more like theory-of-mind ones through tool use and persistent context.'},{id:"q2-5",num:"2.5",question:"What are the 4 pillars of AI?",answer:"Across frameworks, the consistent pillars are: <strong>Data</strong> (quality and volume), <strong>Compute</strong> (infrastructure and cost), <strong>Algorithms</strong> (model architecture), and <strong>People</strong> (domain expertise to supervise and improve the system). Of these, People is the longest-lead bottleneck. You can rent compute and buy data. You can't rapidly acquire practitioners who know both the domain and the models."}]},{id:"s3",num:"03",title:"Tools &",titleItalic:"Platforms.",deck:"The honest rundown on what's actually useful versus what's just well-funded.",cards:[{id:"q3-1",num:"3.1",question:"What is the best AI workflow tool?",answer:"It depends on where you sit on the complexity curve. <strong>No-code</strong>: Zapier AI, Make, monday.com. <strong>Low-code</strong>: n8n, Pipedream. <strong>Code-first</strong>: LangChain, LlamaIndex, custom Claude/GPT API integrations. No-code gets you to 80% quickly and hits a wall. Code-first has no ceiling but requires engineering time. Most production teams end up hybrid."},{id:"q3-2",num:"3.2",question:"Can ChatGPT create workflows?",answer:"Yes \u2014 ChatGPT can design, describe, and write the code for workflows. It can also be a step inside a workflow via the API. The distinction matters: using ChatGPT to <em>build</em> a workflow is a productivity tool. Using the API as a <em>node</em> in a running workflow is an architectural decision. The latter is where teams underestimate latency and cost at scale."},{id:"q3-3",num:"3.3",question:"Which AI tool is most popular?",answer:"ChatGPT remains the most widely recognized AI tool. But popularity in a consumer context doesn't translate to best-in-class for workflows. <strong>Claude 3.5 Sonnet</strong> is widely considered superior for nuanced writing, coding, and reasoning tasks as of mid-2026. Gemini leads on multimodal and Google Workspace integration. Match the model to the task, not the brand recognition."},{id:"q3-4",num:"3.4",question:"What are the 7 components of AI?",answer:"For an AI agent architecture: <strong>goal definition, perception/input, memory, reasoning/planning, tool use, action execution, and output/feedback</strong>. This maps directly to workflow design. Goal = trigger condition. Perception = data ingestion. Memory = context + retrieval. Reasoning = the model call. Tool use = API integrations. Action = write to DB, send message. Feedback = log outcome for evaluation."}]},{id:"s4",num:"04",title:"Building",titleItalic:"Workflows.",deck:"The practical how \u2014 from blank canvas to something running in production.",callout:{eyebrow:"Before you build",heading:"Map the failure modes first.",body:"Draw the workflow. Then ask: what happens when the model returns garbage? What happens when the API is down? <strong>Every branch that leads to silent failure needs a fallback before you ship.</strong>"},cards:[{id:"q4-1",num:"4.1",question:"How do you generate a workflow using AI?",answer:"Connect a data source (email, form, webhook) \u2192 define what the AI model does with that data (classify, extract, generate) \u2192 wire the output to an action (update a record, send a message, trigger another step). The hard part is writing the prompt that's robust to edge cases \u2014 that takes iteration and logging, not just a clever initial draft."},{id:"q4-2",num:"4.2",question:"How do you develop workflows?",answer:"Define a clear, measurable goal. Map chronological tasks. Assign ownership at each step (human or AI). Select tools. Build the happy path first, then stress-test with edge cases. The most common mistake is building the automation before establishing the baseline metric \u2014 if you don't know your current error rate, you can't prove the AI improved it."},{id:"q4-3",num:"4.3",question:"What are the 4 C's of AI compliance?",answer:"In the context of responsible AI workflow design: <strong>Compliance</strong> (meets regulatory requirements), <strong>Confidence</strong> (can you quantify the model's certainty), <strong>Consistency</strong> (same behavior on similar inputs), and <strong>Clarity</strong> (can you explain the output). These aren't theoretical \u2014 they're the questions an auditor asks when a workflow makes a wrong decision at scale."},{id:"q4-4",num:"4.4",question:"What is L1 L2 L3 in AI workflows?",answer:"<strong>L1</strong> handles routine, rule-based tasks. <strong>L2</strong> handles exceptions with AI-assisted decision-making, escalating to humans when confidence is low. <strong>L3</strong> handles complex, judgment-intensive tasks where AI augments human expertise. Deploy L1 broadly (high ROI, low risk), L2 selectively, and L3 sparingly \u2014 not because L3 isn't valuable but because it requires the most oversight."}]},{id:"s5",num:"05",title:"Challenges &",titleItalic:"Best Practices.",deck:"Why 85% of AI projects fail \u2014 and what the 15% do differently.",cards:[{id:"q5-1",num:"5.1",question:"What is the biggest problem with AI?",answer:"In production workflows, the biggest problem is <strong>lack of transparency</strong> in how models make decisions. When a workflow produces a wrong output, you need to know which step failed and why. Without logging and explainability tooling built in from day one, debugging becomes archaeology. Second biggest: data quality. Models are amplifiers \u2014 they amplify good data into great outputs and bad data into confidently wrong ones."},{id:"q5-2",num:"5.2",question:"Why do 85% of AI projects fail?",answer:'Top reasons: vague problem definition (no measurable success condition), poor data quality, underestimating deployment and monitoring cost, building for the demo rather than the edge case, and lack of domain expertise on the team. Most projects "fail" by not reaching production, not by producing wrong results. Getting to production is an organizational problem more than a technical one.'},{id:"q5-3",num:"5.3",question:"What skills are needed to work in AI workflows?",answer:"Core: prompt engineering, API integration, basic Python or JavaScript, data cleaning fundamentals. Differentiating: systems thinking (understanding how components fail), domain expertise, evaluation methodology (how do you score outputs?), and cost modeling (how do you prevent runaway API spend?). The actual bottleneck in most organizations is people who can scope, build, and evaluate a workflow end to end."},{id:"q5-4",num:"5.4",question:"What do humans have that AI can never have?",answer:"Accountability. A model can produce an output; only a human can own the consequence of acting on it. This is the non-technical moat for human workers in AI-augmented workflows. Design your workflows with explicit human ownership of outcomes, not just human review of outputs."}]}]},{slug:"who-hallucinates-more-chatgpt-or-claude",title:"Who Hallucinates More: ChatGPT or Claude?",titleDisplay:"Who Hallucinates More:",titleDisplayItalic:"ChatGPT or Claude?",description:"Five benchmarks, two competing verdicts, and the one variable every comparison article gets wrong. Know which model to trust before the answer matters.",publishedAt:"2026-05-23",author:"MCP Scraper",authorInitials:"M",tags:["AI hallucination","ChatGPT","Claude","LLM accuracy","AI benchmarks"],category:"AI Accuracy",badge:"32 questions answered",fieldGuideLabel:"Field guide",deck:'Five benchmarks give five different winners \u2014 and every "Claude wins" article was benchmarked on a model that is no longer the default. Here is what the current data actually says, and what to do with it.',readTimeMinutes:14,stats:[{value:"5",label:"sections"},{value:"32",label:"questions answered"},{value:"5",label:"benchmarks compared"},{value:"0",label:"fluff"}],ctaHeading:"Verify before you ship with",ctaHeadingItalic:"live data.",ctaBody:"MCP Scraper gives you real-time SERP intelligence, PAA harvests, and page extraction so your AI workflows are grounded in current sources \u2014 not cached claims from articles written about models that no longer exist.",sections:[{id:"what-is-hallucination",num:"01",title:"What You're Actually Asking",titleItalic:"About.",deck:'Before comparing rates, you need to know what the word "hallucination" means \u2014 and it turns out no benchmark, no article, and no AI company uses the same definition. A 3% rate and a 15% rate can describe the same model on the same day.',callout:{eyebrow:"Terminology",heading:"Hallucination and confabulation are not the same thing \u2014 and the distinction explains why Claude and ChatGPT get different labels.",body:"Confabulation is the specific pattern of plausibly gap-filling missing knowledge with invented detail \u2014 the brain (or model) connecting dots that were never there. Hallucination is the broader term covering any confident false output. <strong>Claude's uncertainty-admission training was designed to interrupt confabulation specifically.</strong> ChatGPT's RLHF was tuned on human preference, which tends to reward confident, complete-sounding answers even when the model is uncertain. The same root behavior gets opposite training signals in each system."},cards:[{id:"q1-1",num:"1.1",question:"what is AI hallucination",answer:`<strong>AI hallucination is when a language model produces confident, fluent output that is factually wrong \u2014 a citation that doesn't exist, a date that never happened, a quote no one said.</strong> The term is borrowed loosely from psychiatry, where hallucination means perceiving something that isn't there. In practice, LLM hallucinations look less like delusions and more like plausible-sounding autocomplete: the model generates the statistically likely continuation of a sentence, not a grounded fact lookup. The critical word is "confident" \u2014 hallucinations are dangerous not because models are wrong, but because they are wrong without signaling any uncertainty. A practitioner's real concern is not hallucination frequency but hallucination detectability: a model that hallucinates rarely but never hedges is far more dangerous in production than one that hallucinates often and flags it.`},{id:"q1-2",num:"1.2",question:"why do AI chatbots hallucinate",answer:`<strong>AI chatbots hallucinate because they are trained to predict the most plausible next token, not to retrieve verified facts from a ground-truth database.</strong> The architecture is fundamentally generative \u2014 the model produces text that fits the statistical patterns in its training corpus, and sometimes those patterns lead it to fill gaps with invented specifics. Three compounding factors make hallucination worse: sparse coverage of a topic in training data (the model extrapolates), conflicting information in the corpus (the model blends), and RLHF reward signals that favor fluent, complete-sounding outputs over hedged ones (the model stops saying "I'm not sure"). The reason ChatGPT and Claude hallucinate at different rates on different tasks is not architecture alone \u2014 it is which of these three failure modes each system's training most aggressively corrects for. If your task exposes sparse training coverage (niche domain knowledge, recent events), neither model can save you without grounded retrieval.`},{id:"q1-3",num:"1.3",question:"what is confabulation in AI",answer:`<strong>Confabulation in AI is the specific pattern where a model fills a knowledge gap with invented-but-plausible detail rather than refusing or hedging</strong> \u2014 the model "connects the dots" that were never actually there. The clinical term comes from neurology, where patients with certain memory disorders produce false memories that feel entirely real to them. In LLMs, confabulation is the mechanism behind the most dangerous class of hallucinations: not random nonsense but well-constructed fabrications \u2014 a fake paper with a real author's name, a plausible-sounding legal citation, a drug dosage derived by averaging nearby real figures. The distinction matters for tooling: hallucination detectors that look for low confidence scores will often miss confabulation, because the model's internal confidence on a confabulated output can be high. Grounding against primary sources \u2014 not just asking the model to self-check \u2014 is the only reliable counter.`},{id:"q1-4",num:"1.4",question:"what is the difference between AI hallucination and confabulation",answer:"<strong>Hallucination is the broad category; confabulation is the specific failure mode where the model invents plausible gap-fills rather than flagging its own uncertainty.</strong> All confabulation is hallucination, but not all hallucination is confabulation \u2014 a model that confidently states a wrong date is hallucinating, but it isn't necessarily confabulating if the error traces to a corrupted training example rather than a gap-bridging inference. The distinction changes what interventions work: suppressing confabulation requires training models to recognize the edges of their own knowledge and refuse at those boundaries (which is what Constitutional AI's self-critique loop does for Claude). Suppressing hallucination more broadly requires grounding \u2014 retrieval-augmented generation, citation enforcement, source verification. Practitioners who use the words interchangeably will apply the wrong fix."},{id:"q1-5",num:"1.5",question:"are AI hallucinations the same as lying",answer:`<strong>No \u2014 hallucination is a failure of knowledge, not a failure of intent, which means the usual remedies for dishonesty (adversarial red-teaming, filtering, policy enforcement) don't reduce it.</strong> A lying agent knows the truth and conceals it; a hallucinating model has no ground-truth representation to conceal \u2014 it generates the output that fits the learned distribution, whether that output is accurate or not. This distinction is not just philosophical. Treating hallucination as lying leads organizations to apply trust-and-safety interventions (content moderation, output filtering) rather than epistemic interventions (grounding, uncertainty calibration, retrieval). The more practically damaging confusion is the reverse: treating hallucination as a fixable "bad behavior" that fine-tuning will eventually eliminate, rather than as a structural property of generative models that requires architectural solutions.`},{id:"q1-6",num:"1.6",question:"why does ChatGPT make things up",answer:`<strong>ChatGPT makes things up because its RLHF training consistently rewarded fluent, complete-sounding answers \u2014 and human raters often cannot tell in the moment whether a specific claim is true.</strong> When the model encounters a query at the edge of its training knowledge, it faces two options: produce a hedged, incomplete answer (which RLHF raters historically penalized as unhelpful) or produce a fluent, confident-sounding answer that fills the gap (which raters often rewarded as useful). Over millions of training examples, that signal compounds: the model learns that confident gap-filling is the preferred behavior. OpenAI's release notes for GPT-5.5 Instant specifically cite "reduces hallucination in sensitive areas such as law, medicine, and finance" as a named improvement \u2014 which is an implicit acknowledgment that prior versions were not calibrated to refuse when uncertain. The fix is not better knowledge; it is better uncertainty signaling.`,source:"https://techcrunch.com/2026/05/05/openai-releases-gpt-5-5-instant-a-new-default-model-for-chatgpt/"}]},{id:"the-verdict-depends",num:"02",title:"The Verdict",titleItalic:"Depends.",deck:"One proprietary test shows Claude hallucinating more than ChatGPT (15% vs. 12%). A different benchmark run on the same models the same year shows Claude with the lowest contradiction rate of five providers. Both studies are real. Neither is lying. The winner changes when the measurement changes \u2014 and no competitor article tells you which measurement matches your actual task.",callout:{eyebrow:"Deposition",heading:`Every "Claude wins" verdict was written against a different product than the one you're using today.`,body:'GPT-5.5 Instant became the default ChatGPT in May 2026. Claude Opus 4.7 is the current frontier Claude. <strong>The top SERP articles comparing hallucination rates were benchmarked primarily on GPT-4 Turbo and Claude 3 variants.</strong> The benchmark scores you are reading describe models that are no longer the default. This is not a minor caveat \u2014 task-type inversion, refusal-rate confounds, and methodology differences all compound when the model version gap is also wrong. The deposition question is not "which model wins?" It is: "Which benchmark, on which task type, on which model version, measured how?"'},cards:[{id:"q2-1",num:"2.1",question:"who hallucinates more ChatGPT or Claude",answer:`<strong>Neither model consistently hallucinates more \u2014 the winner changes based on the task type, benchmark methodology, and which model version is being measured.</strong> On BullshitBench v2, Claude Sonnet 4.6 hits a 3% hallucination rate with a 91% detection rate, while OpenAI GPT models are "stuck in the 55\u201365% range" for detection. On Vectara's harder enterprise dataset (February 2026), GPT-4.1 scores 5.6% versus Claude Sonnet 4.6 at 10.6% \u2014 a reversal. On AA-Omniscience, Claude Opus 4.1 achieves 0% hallucination (via refusal), while GPT-5.5 reaches 86% error on the same benchmark. The honest answer for practitioners: Claude tends to outperform on tasks requiring uncertainty calibration and open-recall; ChatGPT tends to outperform on grounded tasks with source material present. Your use case determines the verdict.`,source:"https://medium.com/@anyapi.ai/llm-hallucination-index-2026-why-claude-4-6-7b2d13ed9f0c"},{id:"q2-2",num:"2.2",question:"does Claude hallucinate less than ChatGPT",answer:`<strong>Claude hallucinates less than ChatGPT on open-recall and uncertainty-calibration benchmarks, but GPT models can outperform Claude on grounded generation tasks where source material is provided.</strong> On the Vectara HHEM original dataset (April 2025), GPT-5 scores 1.4% versus Claude-3.7-Sonnet at 4.4% \u2014 ChatGPT wins. On BullshitBench v2, Claude Sonnet 4.6 scores 3% with a 91% detection rate \u2014 Claude wins. On AA-Omniscience, Claude Opus 4.1 achieves 0% hallucination via confident refusal \u2014 Claude wins decisively. The most useful reframe: Claude tends to hallucinate less on tasks where "I don't know" is an acceptable output; ChatGPT can score lower on structured summarization tasks where the source material bounds the answer space.`,source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q2-3",num:"2.3",question:"which AI is more accurate ChatGPT or Claude",answer:"<strong>GPT models score higher on grounded factual accuracy when source material is present; Claude scores higher on calibration \u2014 knowing when not to answer.</strong> On FACTS Overall Scores (grounded generation), GPT-5 scores 61.8 versus Claude Opus 4.5 at 51.3. On AA-Omniscience, Claude Opus 4.1 achieves 0% hallucination while GPT-5.5 reaches 86% error. These are not contradictions \u2014 they measure different things. FACTS rewards producing correct answers given a source; AA-Omniscience rewards refusing answers when knowledge is uncertain. Accuracy in a production system means both: getting the answer right when you have the source, and refusing when you don't. No single model currently dominates both dimensions simultaneously.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q2-4",num:"2.4",question:"what is the hallucination rate of ChatGPT in 2026",answer:`<strong>ChatGPT's hallucination rate in 2026 ranges from 1.4% on Vectara's original RAG benchmark to 86% on AA-Omniscience's domain-knowledge open-recall test \u2014 the same model, different methodologies.</strong> On BullshitBench v2, OpenAI GPT models are "stuck in the 55\u201365% range" for hallucination detection. GPT-5 with thinking mode achieves 1.6% on HealthBench (medical domain). O3 hits 51% hallucination on SimpleQA; o4-mini reaches 79% on PersonQA. The number you see in any article reflects the benchmark used, not a universal accuracy property. The most applicable figure depends on your task: if you are doing RAG summarization, Vectara's 1.4% is relevant; if you are asking ChatGPT to recall domain-specific facts without source material, the AA-Omniscience figure is the honest baseline.`,source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q2-5",num:"2.5",question:"what is the hallucination rate of Claude in 2026",answer:`<strong>Claude's hallucination rate in 2026 spans from 0% (Claude Opus 4.1 on AA-Omniscience, via refusal) to 58% (Claude Opus 4.5 on the same benchmark when not configured to refuse) \u2014 a range that makes any single number misleading.</strong> On BullshitBench v2, Claude Sonnet 4.6 hits 3% with a 91% detection rate, making it the strongest performer in that benchmark class. On Vectara's enterprise dataset (February 2026), Claude Sonnet 4.6 scores 10.6% and Claude Opus 4.6 scores 12.2%. The spread is explained by task type: Claude's Constitutional AI training produces strong refusal behavior on uncertain factual questions, which collapses the hallucination rate on benchmarks that reward "I don't know" responses and inflates it on benchmarks that penalize non-answers.`,source:"https://medium.com/@anyapi.ai/llm-hallucination-index-2026-why-claude-4-6-7b2d13ed9f0c"},{id:"q2-6",num:"2.6",question:"which AI has the lowest hallucination rate in 2026",answer:`<strong>Gemini-2.0-Flash-001 holds the lowest published Vectara HHEM score at 0.7% on the original dataset \u2014 but that benchmark measures factual consistency in RAG summarization, not open-ended recall.</strong> On open-recall benchmarks, Claude Opus 4.1 achieves 0% on AA-Omniscience by refusing uncertain questions, while o3-mini-high scores 0.8% on Vectara. The "lowest hallucination rate" title changes with every benchmark and model release cycle; the more useful question is which model has the lowest hallucination rate on your specific task class. For enterprise RAG pipelines with provided source material, GPT-4.1 at 5.6% on the harder Vectara dataset is currently competitive. For open-domain factual recall with uncertainty, Claude's refusal behavior produces the lowest confirmed error rate.`,source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q2-7",num:"2.7",question:"how is AI hallucination measured",answer:"<strong>AI hallucination is measured by comparing model outputs against a verified ground-truth set and scoring the proportion of confident claims that are factually wrong \u2014 but the ground-truth set, task type, and scoring rules vary so widely across benchmarks that the resulting numbers are rarely comparable.</strong> Three methodology families dominate: RAG consistency tests (Vectara HHEM measures whether a summary stays faithful to the source document), factual recall tests (SimpleQA, PersonQA ask the model open questions with known correct answers), and calibration tests (AA-Omniscience scores how often a model produces wrong answers on questions it should refuse). The same model can score in the top tier on one family and bottom tier on another. Before citing a hallucination rate, the practitioner question is: what task type does this benchmark represent, and does that match what I'm actually asking the model to do?",source:"https://chatgptguide.ai/ai-hallucination-rates-report-gpt-claude-gemini/"}]},{id:"why-claude-behaves-differently",num:"03",title:"Why Claude Behaves Differently",titleItalic:"(And Why That's Complicated.)",deck:`Constitutional AI was built to interrupt confabulation at the output layer \u2014 not to make Claude more knowledgeable, but to make it refuse when it isn't. That design makes Claude's hallucination rate look better on open-recall benchmarks and worse on grounded tasks where refusing an answer is the wrong move. The "safer model" label hides a trade-off every competitor article misses.`,callout:{eyebrow:"Architecture",heading:"Claude's 0% hallucination score on AA-Omniscience is achieved by refusing to answer \u2014 GPT-5.5 attempts the same questions and scores 86% error.",body:`These are not equivalent failure modes. <strong>Claude's refusal behavior is a deliberate uncertainty-admission signal trained by Constitutional AI's self-critique loop.</strong> GPT-5.5's 86% error rate on that benchmark reflects RLHF training that rewards confident, complete-sounding output even under epistemic uncertainty. A practitioner choosing between them for a task where refusal is unacceptable \u2014 a legal brief, a diagnostic intake form, a real-time research summary \u2014 needs to know that "lower hallucination rate" may mean "higher refusal rate," not "more accurate answers."`},cards:[{id:"q3-1",num:"3.1",question:"what is Constitutional AI and does it reduce hallucinations",answer:`<strong>Constitutional AI is Anthropic's training methodology where Claude critiques and revises its own outputs against a set of principles \u2014 and yes, it reduces a specific class of hallucination: confabulation driven by overconfidence.</strong> The core mechanism is a self-critique loop: at training time, Claude is prompted to evaluate its own responses against a constitution of principles (including honesty norms) and revise outputs that violate them. Over millions of examples, this trains the model to flag uncertainty rather than elaborate plausibly over it. What it does not do is give Claude better knowledge \u2014 it makes the model more likely to output "I'm not sure" or refuse at the boundary of its knowledge. The result is measurably lower hallucination rates on open-recall benchmarks and, as a side effect, higher refusal rates on tasks where humans expect confident answers. The "safer model" framing is accurate but incomplete: Constitutional AI reduces dangerous confabulation, not all incorrect outputs.`},{id:"q3-2",num:"3.2",question:"why does Claude hallucinate less than ChatGPT",answer:"<strong>Claude hallucinates less than ChatGPT on uncertainty-sensitive tasks because Constitutional AI's self-critique training penalized overconfident outputs at the architectural level \u2014 not because Claude has better underlying knowledge.</strong> ChatGPT's RLHF training was tuned on human preference ratings, and human raters consistently prefer confident, complete-sounding answers to hedged, partial ones \u2014 even when the hedged answer is more accurate. That preference signal, applied at scale, teaches the model to fill gaps confidently. Constitutional AI's revision loop applies a different signal: outputs that violate honesty norms (overconfident claims under uncertainty) are scored negatively by the model itself and revised. On benchmarks like AA-Omniscience, this difference is dramatic: Claude Opus 4.1 achieves 0% hallucination by refusing uncertain questions; GPT-5.5 attempts the same questions and produces 86% error. The practical implication is that Claude's advantage narrows or reverses when the task context provides source material that bounds the answer \u2014 because grounding reduces the need for uncertainty calibration.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q3-3",num:"3.3",question:"why does Claude say I don't know more than ChatGPT",answer:`<strong>Claude says "I don't know" more than ChatGPT because uncertainty admission was a first-class design goal in Constitutional AI, not an afterthought in RLHF fine-tuning.</strong> Anthropic explicitly trained Claude to identify the edges of its own knowledge and signal them rather than bridge them. The constitutional principle "prefer accurate uncertainty estimates over confident wrong answers" was applied via the self-critique loop \u2014 meaning at training time, Claude learned to score its own overconfident outputs negatively. ChatGPT's training did the opposite: human raters penalized partial or hedged answers as unhelpful, pushing the model toward confident completeness. For practitioners, the implication is that Claude's "I don't know" is a calibration signal worth respecting \u2014 it correlates with genuine knowledge boundaries. ChatGPT's confident answers do not carry the same calibration signal and require independent verification more often.`},{id:"q3-4",num:"3.4",question:"why does Claude admit uncertainty more than ChatGPT",answer:"<strong>Claude admits uncertainty more than ChatGPT because its training reward function directly penalized overconfidence, while ChatGPT's reward function indirectly penalized uncertainty by preferring fluent, complete-sounding outputs.</strong> These are mirror-image training problems with mirror-image results. At the architectural level, both models have the same epistemic limitation: they cannot know what they do not know. What differs is how each handles that edge. Claude's Constitutional AI self-critique loop was designed to surface that edge and output it. ChatGPT's RLHF fine-tuning learned to smooth over it. The consequence for production use is that when Claude expresses uncertainty, it is more likely to be a genuine signal. When ChatGPT expresses certainty, it is less likely to be a reliable signal than the confident phrasing suggests. This asymmetry is the single most important behavioral difference between the two systems for high-stakes use cases."},{id:"q3-5",num:"3.5",question:"does Claude refuse to answer questions it doesn't know",answer:"<strong>Yes \u2014 Claude is trained to refuse or heavily hedge questions at the boundary of its knowledge, and this behavior is measurable: on AA-Omniscience, Claude Opus 4.1 achieves 0% hallucination by refusing uncertain domain-knowledge questions rather than attempting them.</strong> This refusal behavior is not a safety filter applied after generation \u2014 it is a trained output preference baked into the model via Constitutional AI's self-critique loop. The practical consequence is two-sided: Claude produces fewer confident wrong answers than ChatGPT, but it also produces more non-answers on questions where an attempt \u2014 even an imperfect one \u2014 would be useful. For tasks where a partial answer is better than no answer (brainstorming, hypothesis generation, exploratory research), ChatGPT's higher attempt rate is a feature. For tasks where a wrong answer causes real harm (legal, medical, compliance), Claude's refusal behavior is the more defensible default.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"}]},{id:"real-world-consequences",num:"04",title:"When Getting It Wrong",titleItalic:"Has Consequences.",deck:"ChatGPT has already fabricated legal citations in federal court, hallucinated drug dosages, and invented academic papers that passed first-pass review. The more unsettling risk is newer: when ChatGPT, Claude, and Gemini all hallucinate the same false claim, cross-checking them doesn't give you three independent sources. It gives you the same error three times.",callout:{eyebrow:"False Consensus Risk",heading:"If three major LLMs hallucinate the same lie about your business, it can become the new truth \u2014 and no benchmark measures this.",body:"The correlated hallucination problem emerges from shared training data, overlapping RLHF pipelines, and convergent fine-tuning on the same web corpus. <strong>When Claude, ChatGPT, and Gemini all reproduce the same unsupported claim, a practitioner who cross-checks across models gets false triangulation rather than independent verification.</strong> This risk is entirely absent from every current competitor article on hallucination rates \u2014 and it is most acute for entities (companies, people, products) that appear in training data in ways the entity cannot audit or correct."},cards:[{id:"q4-1",num:"4.1",question:"what are examples of ChatGPT hallucinations",answer:"<strong>The most documented ChatGPT hallucinations include fabricated legal case citations presented to federal courts, invented academic papers with real author names, and confident wrong drug dosages in medical queries.</strong> The legal citation failures are the best-documented: in the Mata v. Avianca case, a New York attorney submitted a brief citing multiple cases that ChatGPT invented wholesale \u2014 cases that did not exist anywhere in the legal record. Academic hallucinations are structurally similar: ChatGPT generates plausible-sounding paper titles, journal names, and DOIs that pass casual verification because all the component elements (author names, journal names, topic keywords) are real \u2014 only the assembled paper is fabricated. The pattern in all major cases is the same: ChatGPT produces hallucinations that are specifically designed, by the training distribution, to pass the first verification step a non-expert would apply."},{id:"q4-2",num:"4.2",question:"has ChatGPT hallucinated in court",answer:"<strong>Yes \u2014 the most consequential documented case is Mata v. Avianca, where an attorney used ChatGPT to research case law and submitted a brief citing multiple cases that did not exist, resulting in federal court sanctions.</strong> The cases ChatGPT generated were plausible: they had realistic docket numbers, party names consistent with real aviation litigation, and summaries that read as coherent legal precedent. None of them could be located by opposing counsel or the court because they were entirely fabricated. The attorney was sanctioned for failing to verify the citations. What makes this case significant beyond its notoriety is the mechanism: ChatGPT did not produce random nonsense \u2014 it confabulated, generating outputs that fit the expected form of real legal citations so closely that a trained attorney did not catch them on first read. This is the confabulation failure mode operating at the level of maximum real-world cost."},{id:"q4-3",num:"4.3",question:"did ChatGPT make up fake legal cases",answer:`<strong>Yes \u2014 in the Mata v. Avianca case, ChatGPT generated at least six fake legal cases that were submitted to a federal court as real precedent, making this the first major documented instance of LLM hallucination causing legal sanctions against a practicing attorney.</strong> The fabricated cases had realistic-looking citations: Varghese v. China Southern Airlines, Shaboon v. Egyptair, Zicherman v. Korean Air Lines, and others \u2014 all plausible-sounding aviation negligence precedents. When the court asked for copies of the actual decisions, the attorney could not produce them because they did not exist. The episode became a landmark not just for AI liability but for the broader question of what "verification" means when a model's hallucinations are structurally indistinguishable from real citations to a non-expert reader. Neither Claude nor any other model has a documented comparable case \u2014 but the mechanism exists in all models that generate legal text without grounding.`},{id:"q4-4",num:"4.4",question:"what happened in the Mata v. Avianca ChatGPT hallucination case",answer:"<strong>In Mata v. Avianca, a personal injury lawsuit filed in the Southern District of New York, attorney Steven Schwartz used ChatGPT to research aviation negligence precedents and submitted a court brief citing six cases that ChatGPT had fabricated.</strong> When opposing counsel could not locate the cited cases, the court ordered Schwartz to produce the actual decisions. He could not \u2014 the cases existed only in ChatGPT's output. The court sanctioned Schwartz and his firm. Schwartz's defense was that he was unfamiliar with ChatGPT's tendency to generate false information; the court found that reliance on an AI tool without verification constituted professional negligence. The case is now the canonical example cited in AI liability discussions because it makes concrete the abstract warning that LLM hallucinations have real-world costs \u2014 and because the mechanism was confabulation, not noise: the fake cases were structurally indistinguishable from real ones."},{id:"q4-5",num:"4.5",question:"can ChatGPT hallucinate medical information",answer:"<strong>Yes \u2014 ChatGPT can hallucinate medical information, and the risk is highest in exactly the scenarios where clinicians are most likely to use it: rare conditions, drug-drug interactions, and off-label dosages that are underrepresented in training data.</strong> GPT-4o scores 15.8% hallucination on HealthBench, a medical domain benchmark; GPT-5 with thinking mode reduces this to 1.6%, but the improvement is conditional on the thinking mode being enabled and the query being within the benchmark's scope. The practical risk for medical use is not just the headline hallucination rate \u2014 it is the confabulation pattern: ChatGPT generating specific-sounding dosages or protocol details that are plausible but wrong, in the confident register that clinical notes require. The consensus across medical AI research is that no current LLM should be used as a primary information source for treatment decisions without RAG grounding against validated clinical databases.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q4-6",num:"4.6",question:"does ChatGPT hallucinate more on academic citations",answer:"<strong>ChatGPT hallucinates on academic citations at a notably higher rate than on general-text tasks because academic citations combine several conditions that maximize confabulation risk: sparse training coverage of specific papers, high structural regularity (author, title, journal, year, DOI), and user verification behavior that rarely extends beyond checking the format.</strong> The model has learned that citations follow predictable patterns. When asked for a citation it does not have in training, it generates a citation that fits those patterns \u2014 assembling a plausible author name, a realistic journal, and a plausible year around a fabricated paper. Dedicated academic search integrations (like ChatGPT's search tool when enabled) significantly reduce citation hallucination by grounding against live databases. Without grounding, treating any LLM-generated academic citation as provisional and verifying it against Google Scholar, CrossRef, or a DOI resolver is not optional \u2014 it is the baseline standard of care."}]},{id:"choose-and-verify",num:"05",title:"Choose a Model.",titleItalic:"Verify the Answer.",deck:"The right model for your task depends on whether wrong answers or missing answers cost you more. The right verification method depends on whether you need real-time source grounding, prompt-level controls, or live SERP intelligence to know whether the benchmark you're relying on has already been superseded. Static articles can't give you that. Here's what can.",callout:{eyebrow:"MCP Scraper",heading:"Every hallucination benchmark is already measuring a different model than the one you're running today.",body:"GPT-5.5 Instant and Claude Opus 4.7 are the current defaults as of May 2026. <strong>The top SERP articles were benchmarked on GPT-4 Turbo and Claude 3 variants.</strong> Live PAA intelligence from MCP Scraper shows which benchmark claims are currently circulating in the SERP, which task-specific questions are going unanswered (legal, medical, enterprise, scientific writing), and whether the competitive landscape shifted while the static comparison articles were being written. A practitioner who needs the current answer \u2014 not a cached verdict \u2014 needs a live source, not another article that will be wrong in six months."},cards:[{id:"q5-1",num:"5.1",question:"how do you stop ChatGPT from hallucinating",answer:'<strong>You cannot stop ChatGPT from hallucinating entirely, but the four techniques that reliably reduce it are: retrieval-augmented generation with verified sources, chain-of-thought prompting with explicit uncertainty flagging, citation enforcement in the system prompt, and output verification against primary sources before use.</strong> RAG is the highest-leverage intervention: grounding responses against a controlled, verified document set eliminates the knowledge-gap confabulation that produces most dangerous hallucinations. Chain-of-thought prompting ("explain your reasoning step by step, and flag any step where you are less than certain") forces the model to surface uncertainty it would otherwise paper over. Adding "if you are unsure, say so explicitly rather than guessing" to system prompts has measurable effect on calibration. None of these eliminate the problem \u2014 they reduce it. For high-stakes outputs (legal, medical, financial), human verification against primary sources remains the standard. The tools help; they do not replace verification.'},{id:"q5-2",num:"5.2",question:"how to reduce AI hallucinations with prompt engineering",answer:`<strong>The prompt engineering techniques with the strongest documented effect on hallucination reduction are: explicit uncertainty instructions, step-by-step reasoning requirements, role-scoping, and source-citation enforcement \u2014 applied together, not independently.</strong> Explicit uncertainty instructions ("say 'I don't know' rather than guessing") improve calibration on both Claude and ChatGPT because both models are capable of signaling uncertainty; they need permission to do it. Step-by-step reasoning forces the model to commit to intermediate claims that can be checked, catching confabulation earlier in the chain. Role-scoping ("you are a fact-checker; do not include any claim you cannot source") narrows the output distribution toward the verified. Source citation enforcement ("provide a source URL for every factual claim") creates a verification trail. The compounding insight most engineers miss: these techniques are not additive linearly \u2014 models trained with uncertainty signals (like Claude) show larger improvements from uncertainty prompts than models trained against them (like older ChatGPT versions).`},{id:"q5-3",num:"5.3",question:"does asking ChatGPT to cite sources reduce hallucinations",answer:"<strong>Asking ChatGPT to cite sources reduces the frequency of unverifiable claims in the output, but it does not eliminate fabricated citations \u2014 and a fabricated citation with a plausible URL is harder to catch than a claim with no citation at all.</strong> The mechanism is real: source-citation prompting shifts the model toward outputs where citation is possible, which correlates with better-grounded claims. But ChatGPT can and does generate plausible-looking DOIs, arXiv IDs, and URLs that resolve to nothing. The citation requirement creates a false confidence layer \u2014 the output looks verified when it isn't. The correct workflow is source-citation prompting plus independent verification of every cited source before use. For enterprise pipelines, automated link-checking (do all cited URLs actually resolve?) is the minimum; content verification (does the cited source actually say what the LLM claims it says?) is the standard that eliminates the fabricated-citation failure mode."},{id:"q5-4",num:"5.4",question:"what is retrieval-augmented generation and does it stop hallucinations",answer:"<strong>Retrieval-augmented generation (RAG) is an architecture that grounds LLM outputs by retrieving relevant documents from a verified source set and providing them as context \u2014 and it is currently the most effective single intervention for reducing hallucination in production systems.</strong> Instead of asking a model to recall facts from training, a RAG system retrieves the relevant passage from a controlled document store and asks the model to summarize or synthesize it. On Vectara's HHEM benchmark, which specifically tests this summarization-from-source behavior, even older model versions achieve hallucination rates below 5% \u2014 because the knowledge gap confabulation mechanism is eliminated when the answer exists in the provided context. What RAG does not stop: hallucinations that occur when the retrieved document does not contain the answer and the model interpolates anyway, and hallucinations in the retrieval step itself (if a semantic search retrieves the wrong document). RAG reduces hallucination dramatically in bounded domains; it is not a universal cure for open-domain queries."},{id:"q5-5",num:"5.5",question:"which AI is more trustworthy ChatGPT or Claude",answer:`<strong>Claude is more trustworthy for tasks where calibrated uncertainty is the primary requirement; ChatGPT is more trustworthy for tasks where producing an answer \u2014 even an imperfect one \u2014 is the primary requirement.</strong> Trust is not a single dimension. On calibration trust (does the model's expressed confidence correlate with its actual accuracy?), Claude leads: Constitutional AI's uncertainty training produces hedges that track genuine knowledge gaps. On coverage trust (will the model attempt the question rather than refuse?), ChatGPT leads: RLHF training produces higher attempt rates on hard questions. The user perception data reflects calibration trust: 62% of verified ChatGPT (GPT-4/4o) users report "occasional confident inaccuracies" versus 24% of Claude 3.5 users. For practitioners, the framework is: trust Claude more when a wrong answer is worse than no answer; trust ChatGPT more when a partial answer is better than a refusal.`,source:"https://chatgptguide.ai/ai-hallucination-rates-report-gpt-claude-gemini/"},{id:"q5-6",num:"5.6",question:"which AI is safer to use for high-stakes tasks",answer:`<strong>For high-stakes tasks where a wrong answer has irreversible consequences, Claude's refusal-calibrated behavior makes it the safer default \u2014 but safe use of either model requires human verification against primary sources, not model selection alone.</strong> Claude's design advantage in high-stakes contexts is the refusal signal: when Claude says it is uncertain, that signal has been trained to track real knowledge boundaries. ChatGPT's confident outputs in the same situations are less reliably calibrated. However, "safer model" does not mean "reliable without verification" \u2014 Claude Opus 4.5 scores 58% hallucination on AA-Omniscience when not configured to refuse, and Claude Opus 4.7 scores 36%. The practical standard for high-stakes work is: use Claude for the calibration signal, ground with RAG against verified sources, and treat any AI-generated factual claim in a legal, medical, or financial context as provisional until independently confirmed.`,source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q5-7",num:"5.7",question:"which AI hallucinates less for enterprise use",answer:"<strong>For enterprise RAG pipelines where source documents are provided, GPT-4.1 currently outperforms Claude on Vectara's harder enterprise dataset (5.6% vs. Claude Sonnet 4.6's 10.6%) \u2014 but for enterprise use cases requiring open-domain knowledge retrieval or strict uncertainty signaling, Claude's refusal calibration produces fewer dangerous confident errors.</strong> The enterprise use case is split along the same task-type boundary that governs all hallucination comparisons: grounded generation with provided source material favors GPT models; open-domain recall with uncertainty requirements favors Claude. For enterprise deployments at the highest risk level (legal, medical, compliance), the architecture recommendation from available benchmark data is: pair Claude with a RAG pipeline, use Claude's uncertainty signal as a flag for human review, and verify any output where the model expresses high confidence without a cited source. The combination outperforms either model used alone.",source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"},{id:"q5-8",num:"5.8",question:"can you trust ChatGPT for medical or legal advice",answer:'<strong>No \u2014 neither ChatGPT nor any current LLM should be trusted as a primary source for medical or legal advice without independent verification against authoritative primary sources, and the documented failure cases make the risk concrete.</strong> In the Mata v. Avianca case, ChatGPT fabricated legal citations that passed initial attorney review, resulting in federal court sanctions \u2014 the most expensive hallucination failure mode documented in legal practice. On HealthBench, GPT-4o hallucinates medical information at a 15.8% rate; GPT-5 with thinking mode reduces this to 1.6%, but that reduction depends on task-specific configurations unavailable in standard ChatGPT use. For legal research, both ChatGPT and Claude should be used as research accelerators \u2014 identifying potentially relevant cases and concepts \u2014 with every specific citation independently verified against Westlaw, LexisNexis, or primary court documents before any professional use. The standard of care is not "use the safer model"; it is "verify every claim."',source:"https://suprmind.ai/hub/ai-hallucination-rates-and-benchmarks/"}]}]},{slug:"claude-code",title:"Claude Code: The Evaluation Guide Every Tutorial Skips",titleDisplay:"Claude Code",titleDisplayItalic:"Evaluated.",description:"The cost math, head-to-head comparisons, and trust answers that every Claude Code tutorial skips \u2014 so you can decide before you adopt.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["Claude Code","AI coding","developer tools","Anthropic","AI agents"],category:"Developer Tools",badge:"45 questions answered",fieldGuideLabel:"Field guide",deck:"Every article about Claude Code tells you it's an AI coding agent you just need to learn to use. This one tells you how to evaluate whether it's worth adopting \u2014 with the cost math, the head-to-head comparisons, and the trust answers that every tutorial skips.",readTimeMinutes:18,stats:[{value:"6",label:"sections"},{value:"45",label:"questions"},{value:"22",label:"gap questions covered"},{value:"0",label:"fluff"}],ctaHeading:"Scrape smarter with",ctaHeadingItalic:"real web data.",ctaBody:"MCP Scraper gives your Claude Code agents the live web intelligence they need \u2014 SERP data, People Also Ask harvests, competitor page extraction, and structured data feeds \u2014 without rate limits or browser fingerprinting.",sections:[{id:"what-is-claude-code",num:"01",title:"What Claude Code",titleItalic:"Actually Is",deck:'Most people who call Claude Code an "AI coding assistant" are describing a category it has already outgrown \u2014 it is closer to a junior engineer that runs in your terminal than a smarter autocomplete.',callout:{eyebrow:"Built-in definition",heading:"Claude Code operates on your full codebase, not just the file you have open.",body:"Unlike Copilot or Cursor's inline suggestions, Claude Code reads your entire project tree, executes shell commands, runs tests, and commits \u2014 making it an <strong>agentic system</strong>, not an autocomplete extension."},cards:[{id:"q1-1",num:"1.1",question:"What exactly is Claude Code?",answer:"<strong>Claude Code is an agentic coding tool that reads your codebase, edits files, runs commands, and integrates with your development tools</strong> \u2014 available in your terminal, IDE, desktop app, and browser. It is not a chat interface you paste code into; it operates on your local filesystem, executes shell commands with your authorization, and can spawn parallel sub-agents on separate subtasks. Unlike autocomplete tools that react to the file you have open, Claude Code takes a goal as input and works through the steps to achieve it across however many files that requires. The distinction matters for evaluation: you are not buying a smarter tab-completion, you are buying an agent that can misunderstand goals, degrade in long sessions, and occasionally do exactly what you said instead of what you meant."},{id:"q1-2",num:"1.2",question:"How does Claude Code work?",answer:"<strong>You describe a goal in natural language; Claude Code reads relevant files, writes or edits code, runs shell commands, and iterates until the task is complete \u2014 operating on your local filesystem throughout.</strong> The model (Claude Sonnet 4.6 by default, or Opus 4.7 for harder problems) holds up to 1M tokens of context, allowing it to reason across an entire medium-sized codebase in a single session. It runs a loop: read context, plan, execute, observe output, adjust \u2014 stopping when the goal is met or when it needs clarification. The practical implication is that prompt quality matters more than most tutorials admit: vague goals produce vague results, and Claude Code will complete a vague goal confidently."},{id:"q1-3",num:"1.3",question:"What can you do using Claude Code?",answer:"<strong>Write and refactor code, run and fix tests, review pull requests, create scripts, set up CI/CD pipelines, manage files, and spawn parallel agents on subtasks</strong> \u2014 across any language or framework. Beyond code editing, Claude Code integrates with GitHub Actions and GitLab CI/CD natively, can trigger pull requests from a Slack mention, connects to external tools via Model Context Protocol (MCP), and can be scheduled as a routine that runs on Anthropic-managed infrastructure while your computer is off. The surface area is broad enough that the more useful question is what it cannot do reliably \u2014 and that list appears in Section 06."},{id:"q1-4",num:"1.4",question:"What is Claude Code best for?",answer:"<strong>Multi-file refactors, greenfield scaffolding, complex debugging cycles, and any task that requires reading a large codebase to understand context before making changes</strong> are where Claude Code's 1M-token context window creates a genuine capability gap over file-local autocomplete tools. It is also well-suited for tasks that require coordination across multiple steps \u2014 setting up a test suite, migrating an API client, or auditing a codebase for a specific pattern \u2014 where the work is too spread out for a single prompt in a chat interface. Where it underperforms: novel algorithmic design, highly domain-specific regulatory code, and real-time systems \u2014 detailed in Section 06."},{id:"q1-5",num:"1.5",question:"Why is Claude Code so good at coding?",answer:"<strong>It scores 80.8% on SWE-bench Verified</strong> \u2014 the leading benchmark for autonomous software engineering \u2014 and uses models with a 1M-token context window, allowing it to hold an entire medium-sized codebase in memory at once. The SWE-bench score means it solves roughly 4 in 5 real-world GitHub issues autonomously; competing tools score lower on the same benchmark. Context window size is the second structural advantage: autocomplete tools reason over hundreds of tokens; Claude Code reasons over millions, which is the difference between fixing a function and fixing a system. The 20% failure rate on SWE-bench is the equally important number \u2014 Section 06 covers what failure looks like in practice."},{id:"q1-6",num:"1.6",question:"Why is everyone obsessed with Claude Code?",answer:"<strong>Claude Code is the first widely-adopted agentic coding tool that works at the project level rather than the file level</strong>, and it frequently completes tasks developers expected to take hours in under ten minutes \u2014 without requiring IDE changes or workflow restructuring. It runs in a terminal, which means it drops into any existing development environment without replacing the editor you already use. The combination of project-level context, shell execution, and a 1M-token window crossed a threshold where developers found themselves assigning real work to it rather than using it as a drafting aid \u2014 and that shift in how people use it, not any single feature, is what generated the adoption curve."},{id:"q1-7",num:"1.7",question:"How can Claude Code be so good?",answer:"<strong>The underlying models \u2014 Claude Sonnet 4.6 and Opus 4.7 \u2014 were trained with a 1M-token context window and ranked first on SWE-bench</strong>; paired with full filesystem access and shell execution, the capability gap over autocomplete tools is structural, not marginal. Most autocomplete tools use models optimized for next-token prediction in a small window; Claude Code uses models optimized for multi-step reasoning across large contexts \u2014 a different training objective that produces qualitatively different behavior. The important caveat for practitioners: benchmark performance reflects average case; your specific codebase, stack, and task distribution may diverge significantly from the benchmark distribution."}]},{id:"setup-and-surfaces",num:"02",title:"Setup, Surfaces, and",titleItalic:"Who Can Use It",deck:'You do not need to be a developer to start a session with Claude Code \u2014 but the gap between "starting a session" and "getting reliable results" is wider than any installation guide will tell you.',cards:[{id:"q2-1",num:"2.1",question:"How do you start using Claude Code?",answer:"<strong>Install via one curl command (`curl -fsSL https://claude.ai/install.sh | bash`), or use Homebrew (`brew install --cask claude-code`) on Mac or WinGet (`winget install Anthropic.ClaudeCode`) on Windows</strong>; authenticate with a Claude Pro subscription or an Anthropic API key, then run `claude` in your project directory. The tool runs on macOS (Intel and Apple Silicon), Windows (x64 and ARM64), Linux, and WSL. A desktop application for macOS and Windows was also released on April 14, 2026, which adds a GUI launcher and Git worktree support for isolated parallel sessions. The fastest path to a first working session is the API key route \u2014 no subscription required, first session costs cents."},{id:"q2-2",num:"2.2",question:"Can Claude Code start from scratch?",answer:"<strong>Yes \u2014 give it an empty directory and a description and it will scaffold the project structure, create files, initialize git, and write initial code without any existing codebase to read.</strong> This is one of the use cases where Claude Code's agentic loop is most visible: it plans the file structure, writes each file, runs an initial build to check for errors, and iterates \u2014 the same way a developer would approach a greenfield project. The caveat is that greenfield outputs still require review: Claude Code will make architectural decisions based on its training data, and those decisions may not match your team's standards, your target infrastructure, or your preferred dependencies."},{id:"q2-3",num:"2.3",question:"Do you need to be a programmer to use Claude Code?",answer:'<strong>Not for well-defined, bounded tasks</strong> \u2014 product managers and researchers have used it to run competitive analyses, clean data, and build simple automation \u2014 but verifying outputs and debugging failures still benefits from technical fluency. The gap that non-programmers run into is not starting Claude Code; it is recognizing when its output is wrong. Claude Code will generate syntactically valid code that does the wrong thing, use a deprecated API without flagging it, or misunderstand a requirement in a way that only becomes visible when the code runs. Knowing what "correct" looks like is a prerequisite for using any agentic coding tool reliably, and that knowledge does not come from the tool itself.'},{id:"q2-4",num:"2.4",question:"Can beginners use Claude Code?",answer:"<strong>Yes, and beginners can complete real tasks in a first session</strong> \u2014 the friction of setup is low, the natural language interface is accessible, and for tasks with clear success criteria (write a Python script that does X, convert this CSV to JSON), the output is often directly usable. The challenge for beginners is the review step: knowing whether a diff is correct requires enough understanding of the code to evaluate it. The practitioner pattern that works for non-expert users is to use Claude Code for tasks where the output is directly testable \u2014 write a test, run it, see if it passes \u2014 rather than for tasks where correctness requires reading and understanding the generated logic."},{id:"q2-5",num:"2.5",question:"Can I use Claude Code without coding knowledge?",answer:`<strong>For bounded, testable tasks \u2014 yes; for open-ended engineering work \u2014 no.</strong> The limiting factor is not the tool's interface but your ability to recognize when Claude has misunderstood the goal, which requires knowing what "correct" looks like for the specific task. Users without coding knowledge have successfully used Claude Code for data processing, file organization, simple script automation, and research tasks where the output is text or structured data they can evaluate directly. The failure mode is adopting it for work where correctness requires code comprehension \u2014 and then merging Claude's output without understanding it.`},{id:"q2-6",num:"2.6",question:"Can you use Claude Code for personal use?",answer:"<strong>Yes \u2014 there is no restriction to commercial or professional use</strong>; the Pro plan at $20/month or an Anthropic API key with pay-as-you-go billing both cover personal projects without restriction. Personal use cases that work well include automating repetitive file tasks, building personal utilities, learning a new language or framework by having Claude scaffold examples and explain them, and running competitive analysis scripts. The API key path is often more economical for personal use: if you spend under a few dollars per day on average, pay-as-you-go will cost less than the $20/month Pro subscription."},{id:"q2-7",num:"2.7",question:"Can you use Claude Code just to chat?",answer:"<strong>Technically yes, but it is an expensive and poorly-optimized path for conversation only</strong> \u2014 Claude Code is built for filesystem-aware, command-executing sessions, and for general conversation, Claude.ai is a better fit and may be cheaper depending on your plan. Running a chat-only session inside Claude Code consumes the same token budget as a coding session, which means you are burning rate-limit capacity on messages that would cost less (or nothing, on a free Claude.ai tier) in the standard interface. The one exception: if you are in the middle of a coding session and need to think through an architectural question without switching contexts, the terminal is a reasonable place to have that conversation."},{id:"q2-8",num:"2.8",question:"Can you prompt Claude Code?",answer:"<strong>Yes \u2014 you interact with Claude Code entirely through natural language prompts in the terminal</strong>, and you can store persistent instructions in a CLAUDE.md file in your project root so preferences, coding standards, and architectural decisions carry across sessions without re-stating them. The CLAUDE.md file is the most underused feature for teams: it allows you to encode your stack's conventions, preferred libraries, test patterns, and style rules once, so every Claude Code session starts with that context loaded. Practitioners who skip CLAUDE.md setup spend significantly more tokens re-explaining context that should be a project-level constant."}]},{id:"cost-math-and-evaluation",num:"03",title:"The Real Cost Math",titleItalic:"Before You Commit",deck:"The subscription price is not the most important number \u2014 the ratio between what you pay at Max 5x versus what the same usage costs on raw API tokens is 18-to-1, and almost no review article publishes it.",callout:{eyebrow:"Evaluation layer \u2014 what every tutorial skips",heading:"You can run Claude Code today without a paid subscription.",body:"An Anthropic API key unlocks full Claude Code functionality on a <strong>pay-as-you-go basis</strong> \u2014 no $20/month Pro plan required. The average developer spends about <strong>$6 per day</strong> at API rates; for light users, this path is cheaper than a monthly plan and removes the cost-before-commit barrier entirely."},cards:[{id:"q3-1",num:"3.1",question:"How to use Claude Code for free?",answer:'<strong>There is no free Claude Code plan</strong>, but you can use it without a subscription by providing an Anthropic API key and paying per token \u2014 and for light users, this often costs less per month than the $20/month Pro subscription. The Free plan on Claude.ai does not include Claude Code access. The API key path requires a funded Anthropic account but no minimum spend: a few evaluation sessions will cost a few dollars, not $20. The distinction matters: "no free plan" and "no way to try it without committing $20" are different conditions, and almost every article conflates them.'},{id:"q3-2",num:"3.2",question:"Can I run Claude Code locally for free?",answer:"<strong>Not for free with Claude models</strong>, but Claude Code can be configured to run against local Ollama-compatible models, which eliminates API costs entirely \u2014 at the cost of model quality relative to Claude Sonnet 4.6. Running Claude Code against a local Ollama model means all inference stays on your machine: no API call, no token spend, no data leaving your network. The trade-off is that local open-source models perform significantly below Claude Sonnet 4.6 on SWE-bench and similar coding benchmarks, so the output quality for complex tasks is not comparable. For privacy-sensitive experimentation or cost-zero prototyping, local Ollama is a legitimate path; for production coding work, the quality gap is real."},{id:"q3-3",num:"3.3",question:"Can I try Claude Code without paying?",answer:"<strong>The API key path requires a funded Anthropic account, but there is no minimum spend</strong> \u2014 you can run several evaluation sessions for a few dollars without committing to any monthly plan. Create an Anthropic account, add a small credit balance ($5\u2013$10 is enough for meaningful evaluation), generate an API key, and authenticate Claude Code with it. At Sonnet 4.6 rates ($3/MTok input, $15/MTok output), a few hours of coding sessions will cost well under $10. This is the evaluation path that no ranking article describes explicitly, which is why most searchers believe the choice is binary: $20/month Pro or nothing."},{id:"q3-4",num:"3.4",question:"Is Claude Code free now?",answer:'<strong>No \u2014 the Free plan does not include Claude Code access</strong>; the tool requires Pro ($20/month), Max ($100\u2013$200/month), Team Premium ($100\u2013$125/seat/month), or an Anthropic API key with pay-as-you-go billing. As of 2026-05-24, Anthropic has not announced a free tier for Claude Code. The "free" path that does exist is the API key route for low-volume users who spend less monthly than the Pro subscription would cost \u2014 which is free of subscription commitment but not free of per-token cost. If you are searching this question because you saw "free" mentioned somewhere, it likely refers to the absence of a required subscription for the API key path, not zero-cost access.'},{id:"q3-5",num:"3.5",question:"How much is Claude Code per month?",answer:"<strong>Pro: $20/month (or $17/month billed annually); Max 5x: $100/month; Max 20x: $200/month; Team Premium: $125/seat/month (or $100/seat/month annually), minimum 5 seats, Claude Code included; API key: pay-as-you-go, average approximately $6/developer/day.</strong> Team Standard ($25/seat/month) does not include Claude Code. The Max plans exist because Pro has a usage ceiling \u2014 approximately 44,000 tokens per 5-hour window \u2014 that active power users hit daily; Max 5x roughly doubles that to 88,000 tokens, and Max 20x reaches approximately 220,000 tokens per window. For developers who would otherwise pay API rates at those volumes, the Max plan is dramatically cheaper than the alternative."},{id:"q3-6",num:"3.6",question:"Is it worth it to pay for Claude for coding?",answer:"<strong>At Pro ($20/month), the break-even is roughly one hour of professional developer time saved per month</strong> \u2014 for developers using it daily on real tasks, the ratio is heavily favorable. The harder question is whether you need Max-tier throughput: if you hit the Pro usage ceiling regularly, you are spending time waiting for windows to reset instead of working, and the $80/month step-up to Max 5x pays for itself quickly. For occasional users \u2014 a few sessions per week on bounded tasks \u2014 the API key path at average $6/developer/day will cost less than $20/month and provides the same capability without the subscription commitment."},{id:"q3-7",num:"3.7",question:"Is Claude Code actually worth it?",answer:"<strong>For developers running multi-file tasks daily, yes \u2014 the Max plan is approximately 18x cheaper than equivalent API usage at full capacity</strong>; for occasional users, the API key path is more economical and the subscription adds no value. The 18x figure comes from the projected cost of purchasing the same token volume directly at Sonnet 4.6 rates ($3/MTok input): at Max 20x throughput sustained, the API equivalent would run approximately $3,650/month versus $200/month for the Max 20x plan. The honest framing for an evaluation decision: start with the API key path, measure your actual daily spend for two weeks, then decide whether the Pro or Max subscription saves money relative to your real usage pattern."},{id:"q3-8",num:"3.8",question:"How expensive is it to use Claude Code?",answer:"<strong>Light use on an API key: under $2/day; moderate use on Pro: $20/month flat; heavy agentic use on Max 5x: $100/month for approximately 88,000 tokens per 5-hour window</strong>, versus approximately $3,650/month if you paid API rates for the same volume. Ninety percent of API-path users spend under $12/day. Prompt caching reduces costs further for long sessions with repeated context: Sonnet 4.6 cache reads cost $0.30/MTok versus $3/MTok for fresh input \u2014 a 90% discount on context that is already in the cache. The Batch API adds a 50% discount across all token prices for non-real-time workloads. Heavy users who ignore caching and batching pay 2\u20133x more than necessary."},{id:"q3-9",num:"3.9",question:"Is Claude Code no longer pro?",answer:"<strong>Claude Code remains available on the Pro plan</strong> \u2014 the Max plans (5x and 20x) are higher-throughput tiers added for power users, not replacements for Pro. Pro was not removed or downgraded; it retains the same model access (Sonnet 4.6 and Opus 4.7) as Max, with tighter usage limits per 5-hour window (approximately 44,000 tokens). The confusion likely stems from Anthropic's introduction of the Max tier, which is marketed heavily to power users \u2014 but Pro is still the primary entry point for individual developers who do not consistently hit usage ceilings."}]},{id:"comparison-and-switching",num:"04",title:"Claude Code vs. the Tools",titleItalic:"You Already Use",deck:"The three tools most developers compare against Claude Code \u2014 Cursor, GitHub Copilot, and ChatGPT \u2014 answer different questions than Claude Code does, and picking the wrong framing makes the comparison meaningless.",callout:{eyebrow:"Decision-stage question the SERP ignores",heading:"Most professional teams use Claude Code alongside Cursor or Copilot, not instead of them.",body:"The most common production stack is <strong>Cursor for inline editing</strong> (72% autocomplete acceptance rate with Supermaven) <strong>+ Claude Code for complex multi-file tasks</strong> in the terminal \u2014 or Copilot in the IDE + Claude Code for architectural work. Picking one and dropping the other is a false choice."},cards:[{id:"q4-1",num:"4.1",question:"Why are people leaving ChatGPT and going to Claude?",answer:"<strong>Claude's models score higher on coding benchmarks (80.8% SWE-bench Verified) and have a substantially longer context window (1M tokens versus 128k for GPT-4o)</strong>, and Claude Code offers deeper filesystem integration than ChatGPT Codex or the GPT-4 API. For developers specifically, the context window difference is the most consequential: 1M tokens allows Claude Code to reason over an entire codebase; 128k limits competing tools to a subset of files. The SWE-bench gap is real but less dramatic than marketing implies \u2014 both tools fail a meaningful percentage of tasks, and the right comparison is not benchmark scores but how each tool behaves on your specific workload."},{id:"q4-2",num:"4.2",question:"Is ChatGPT or Claude better?",answer:'<strong>For agentic coding tasks, Claude Code leads on SWE-bench Verified at 80.8%</strong>; for general conversation, document analysis, and multimodal tasks, the gap between the two is smaller and depends on the specific benchmark. Neither is universally better: GPT-4o has advantages in certain multimodal contexts; Claude Sonnet 4.6 leads on long-context coding tasks. The evaluation question for a developer is not "which is better overall" but "which handles my workload better" \u2014 the two tools have different context window sizes, different pricing structures, and different agentic execution models, and those differences matter more than aggregate benchmark rankings.'},{id:"q4-3",num:"4.3",question:"Why are people switching to Claude?",answer:`<strong>The three primary reasons developers cite: longer context window (1M versus 128k), stronger agentic task performance on SWE-bench, and Claude Code's full-filesystem terminal approach versus chat-based alternatives.</strong> A secondary factor is Constitutional AI training, which produces a model that more often says "I don't know" or flags uncertainty rather than generating confidently wrong output \u2014 a meaningful difference for code review workflows where false confidence is costly. Developers who switched from ChatGPT-based workflows most commonly cite hitting GPT-4o's context limit on large codebase tasks as the triggering event.`},{id:"q4-4",num:"4.4",question:"Is Copilot cheaper than ChatGPT?",answer:"<strong>GitHub Copilot Pro at $10/month is the lowest-priced individual plan in this comparison</strong>: ChatGPT Plus is $20/month, Claude Pro is $20/month, and Cursor Pro is $20/month. Copilot also offers a team plan at $19/seat/month and enterprise at $39/seat/month. The price comparison is misleading without capability context: Copilot at $10/month provides IDE-integrated autocomplete and code chat; it does not include a standalone agentic coding tool equivalent to Claude Code. Developers who need both inline autocomplete (Copilot's strength) and multi-file agentic task completion (Claude Code's strength) are looking at $10 + $20 = $30/month minimum, not a choice between them."},{id:"q4-5",num:"4.5",question:"Who hallucinates more \u2014 ChatGPT or Claude?",answer:"<strong>Both hallucinate; the more useful comparison for coding tasks is SWE-bench score</strong>, which measures how often a model actually solves a real-world issue correctly rather than generating plausible-looking wrong code \u2014 and Claude Code leads at 80.8%. Claude's Constitutional AI training is designed to produce more calibrated uncertainty: the model is more likely to say it does not know something than to confabulate a confident but wrong answer. In practice, both tools will generate syntactically valid code that fails tests, fabricate library method names, and misread logic \u2014 the difference is in frequency and in how they signal uncertainty. For any coding output from either tool, running tests and reviewing diffs is non-optional."},{id:"q4-6",num:"4.6",question:"What is Claude Code and Ollama?",answer:"<strong>Ollama runs open-source language models locally on your machine; Claude Code can be configured to use Ollama-compatible models as its backend, eliminating API costs and keeping all data entirely local.</strong> This configuration is relevant for two use cases: cost-zero experimentation without API spend, and air-gapped or high-security environments where sending code context to an external API is not permissible. The trade-off is model quality: local open-source models perform significantly below Claude Sonnet 4.6 on SWE-bench and complex coding tasks. The Ollama path is worth knowing because most articles presenting Claude Code as requiring a paid subscription omit it entirely \u2014 for teams with strong privacy requirements or zero budget, it is the only viable evaluation path."}]},{id:"safety-trust-and-privacy",num:"05",title:"Safety, Data Handling, and",titleItalic:"What Claude Code Can Actually See",deck:"The most common trust concerns about Claude Code \u2014 screenshot access, code ownership, data leakage \u2014 have clear factual answers, and none of the ranking articles provide them.",callout:{eyebrow:"Enterprise and privacy-sensitive teams",heading:"Claude Code does not store your code on Anthropic's servers between sessions.",body:"Your files run locally on your machine; only the <strong>conversation context</strong> (your prompts and Claude's responses) is sent to Anthropic's API. The Enterprise plan adds <strong>HIPAA-ready data handling</strong>, audit logs, custom data retention controls, and SCIM provisioning \u2014 features that unblock adoption for regulated industries."},cards:[{id:"q5-1",num:"5.1",question:"Is it safe to use Claude Code?",answer:"<strong>For most professional use, yes \u2014 code executes locally, file access is local, and only prompt context traverses the API</strong>; sensitive or regulated data (HIPAA, PII) requires the Enterprise plan for compliant data handling. The practical risk surface for most developers is not data leakage but local command execution: Claude Code executes shell commands you authorize, and approving a destructive command without reviewing it produces real damage. For teams in regulated industries, the Enterprise plan adds HIPAA-ready data handling, audit logs, and custom data retention controls \u2014 the specific features required for compliant adoption in healthcare, finance, and similar sectors."},{id:"q5-2",num:"5.2",question:"Can you get banned from Claude Code?",answer:"<strong>Yes \u2014 Anthropic's usage policies apply to Claude Code, and automated or agentic misuse can trigger account suspension.</strong> The prohibited uses are the same as those that apply to Claude.ai: generating malware, automating scraping in violation of a site's terms, producing content that violates Anthropic's usage policy, and misrepresenting Claude-generated output as human-authored in contexts where that matters. Agentic tools that execute at scale create higher-velocity policy surface than chat interfaces \u2014 a Claude Code routine running unattended can produce policy violations faster than a human would catch them. Review Anthropic's usage policy before deploying Claude Code in automated, unmonitored pipelines."},{id:"q5-3",num:"5.3",question:"Does Claude Code leak?",answer:`<strong>No evidence of systemic data leakage exists</strong>; prompt context is transmitted to Anthropic's API as part of normal operation but is not persisted beyond the session by default under standard plans. What "leak" means technically: your code appears in the prompt context sent to the API for inference; it is not stored, indexed, or accessible to other users. Enterprise plans add explicit custom retention controls and data-use opt-outs for teams who need contractual data handling guarantees rather than policy-level assurances. The meaningful risk for most teams is not leakage to other users but the transmission of proprietary code to an external API at all \u2014 a policy question, not a technical vulnerability.`},{id:"q5-4",num:"5.4",question:"Is Claude Code unsafe?",answer:"<strong>The primary risk surface is local command execution, not network security</strong> \u2014 Claude Code executes shell commands you authorize on your machine, and approving a destructive command without reviewing it produces damage that is real and often irreversible. Claude Code's design requires your explicit approval before executing commands in most configurations, but users who approve commands quickly without reading the proposed action bypass the primary safety mechanism. The secondary risk is prompt injection in multi-agent configurations \u2014 a subtask agent receiving malicious instructions from an external source. Neither risk is exotic; both are manageable with standard review practices."},{id:"q5-5",num:"5.5",question:"Can I trust Claude Code?",answer:`<strong>Trust is context-dependent: for local development tasks, yes; for regulated data, only under the Enterprise plan; for security-sensitive environments, verify the data handling documentation before adopting.</strong> The relevant trust question for most practitioners is not "is Anthropic malicious" but "does sending my codebase context to an external API comply with my organization's data handling policy" \u2014 and that is a legal and compliance question, not a technical one. Enterprise teams evaluating Claude Code should request Anthropic's data processing agreement and review the HIPAA-ready configuration before making adoption decisions; the documentation exists and is specific.`},{id:"q5-6",num:"5.6",question:"Does Claude Code take screenshots?",answer:"<strong>No \u2014 Claude Code does not have screen capture capability.</strong> It reads and writes files on your filesystem and executes terminal commands, but it cannot capture your screen, access your clipboard, read data from applications outside the project directory you opened it in, or observe your browser activity. The question comes up because agentic AI tools are often conflated with general system-access tools; Claude Code's access is specifically scoped to your project directory and the shell commands you authorize it to run. If you are evaluating Claude Code for an environment where screen capture would be a security concern, that concern does not apply to this tool's architecture."},{id:"q5-7",num:"5.7",question:"Does ChatGPT own my code?",answer:"<strong>OpenAI's standard terms do not claim ownership of output code</strong>, and Anthropic's terms similarly do not claim ownership of code that Claude Code generates \u2014 but both companies' default terms may use your inputs to improve their models unless you opt out or upgrade to an enterprise plan. For most developers, code ownership is not the risk: the risk is whether code you submit as input context can be used in model training. Anthropic's Enterprise plan includes explicit data-use opt-outs; OpenAI's Enterprise plan does the same. On individual plans for either tool, review the current terms of service for training data opt-out provisions, which have changed multiple times across both platforms."},{id:"q5-8",num:"5.8",question:"Why is Claude being blacklisted?",answer:"<strong>Some organizations block Claude via firewall because it is an external API service \u2014 not because of a known security vulnerability in Claude specifically.</strong> IT departments that treat all AI API services as unauthorized external data connections will block Claude Code, ChatGPT, Copilot, and similar tools under the same policy, regardless of their individual security properties. The blocking is a data governance decision, not a technical finding against Claude. For enterprise teams trying to get Claude Code approved through IT, the relevant artifacts are Anthropic's data processing agreement, the HIPAA-ready Enterprise configuration documentation, and Anthropic's SOC 2 compliance status \u2014 not the tool's general reputation."}]},{id:"limits-and-skepticism",num:"06",title:"What Claude Code Gets Wrong",titleItalic:"(And What It Cannot Do)",deck:"The performance ceiling that marketing materials never publish: Claude Code hallucinates, degrades in long sessions, and will confidently generate plausible-looking wrong code \u2014 knowing the failure modes before you adopt matters more than knowing the benchmark score.",cards:[{id:"q6-1",num:"6.1",question:"What is Claude not good at?",answer:"<strong>Novel algorithmic design requiring mathematical proof, highly domain-specific regulatory code with no training signal, real-time systems where every millisecond matters, and tasks requiring external context it cannot access</strong> \u2014 production databases, proprietary internal documentation, undocumented internal APIs \u2014 are the consistent failure categories. Claude Code reasons over what is in its context window; anything that must be inferred from systems it cannot read produces hallucinated or superficially correct but functionally wrong output. The most expensive failure mode in practice is not obvious errors but plausible-looking code that passes a surface review and fails in production \u2014 which is why running tests is not optional."},{id:"q6-2",num:"6.2",question:"Is there a limit to how much you can use Claude Code?",answer:"<strong>Yes \u2014 rate limits vary by plan: Pro gets approximately 44,000 tokens per 5-hour window; Max 5x approximately 88,000; Max 20x approximately 220,000</strong>; hitting the ceiling pauses access until the window resets. These are approximate figures derived from usage reports; Anthropic does not publish the exact token limits by plan. The practical effect: heavy Pro users who run multiple long agentic sessions in a day will hit the ceiling and wait; Max 5x handles most power-user workflows without interruption. On the API key path, there are no usage windows \u2014 you pay per token with no ceiling, which is one reason the API path can be preferable for users who need uninterrupted long sessions."},{id:"q6-3",num:"6.3",question:"Is Claude Code getting dumber?",answer:'<strong>Claude Code is not being degraded \u2014 Anthropic has not reduced model capability</strong>; perceived quality drops in long sessions typically trace to context window saturation, not model downgrade. When a session accumulates enough turns that older context is compressed or dropped to fit within the context window, Claude Code loses access to earlier decisions, file states, and constraints \u2014 and the output quality degrades visibly. The fix is to start a new session for major new tasks rather than extending a single session indefinitely. The "getting dumber" perception is real; the cause is session management, not model regression.'},{id:"q6-4",num:"6.4",question:"Does Claude Code hallucinate?",answer:"<strong>Yes \u2014 it will generate incorrect code, fabricate library method names, and misread file logic</strong>, and the SWE-bench score of 80.8% means it fails roughly 1 in 5 real-world tasks by the benchmark's definition. Hallucination in coding contexts looks different from hallucination in conversation: the output is syntactically valid, the method names look plausible, and the logic structure appears correct \u2014 it fails when you run it. Always run tests and review diffs before merging Claude Code output; treating it as authoritative without validation is the most common source of expensive errors. The 80.8% score is the performance ceiling under benchmark conditions; your specific codebase, stack, and task distribution will produce a different empirical failure rate."},{id:"q6-5",num:"6.5",question:"Can you run Claude Code without internet?",answer:"<strong>Not with Claude models \u2014 all inference goes through Anthropic's API, which requires an active internet connection.</strong> The exception is running Claude Code configured against local Ollama-compatible models, which work fully offline with no API call required. For teams in air-gapped environments or with strict egress policies, the Ollama configuration is the only viable path to offline Claude Code use \u2014 at the cost of model quality. Anthropic does not currently offer an on-premises deployment option for Claude models equivalent to some enterprise AI vendors; the Enterprise plan provides stronger data handling guarantees but still routes inference through Anthropic's API."},{id:"q6-6",num:"6.6",question:"Is Claude Code actually useful?",answer:"<strong>For developers running multi-file refactors, complex debugging cycles, or greenfield scaffolding, yes \u2014 the SWE-bench benchmark performance and consistent practitioner time-savings reports are aligned</strong>; for users expecting zero-verification autonomous output, no. The honest evaluation frame: Claude Code is a force multiplier for developers who can review its output, not an autonomous agent that eliminates the need for developer judgment. Teams that have adopted it most successfully use it for the high-context, high-effort tasks that benefit most from 1M-token reasoning \u2014 not as a replacement for understanding the codebase."},{id:"q6-7",num:"6.7",question:"Is Claude Code still the best coding agent?",answer:'<strong>As of mid-2026, Claude Code leads on SWE-bench Verified at 80.8% and on context window size at 1M tokens</strong>; Cursor leads on inline autocomplete acceptance rate at 72% with Supermaven; the "best" answer depends entirely on whether you optimize for agentic task completion or IDE-native editing speed. The benchmark lead is real but not permanent: SWE-bench scores across competing tools have risen steadily, and the gap between leaders narrows with each model generation. The more durable evaluation criterion than benchmark ranking is which tool handles the specific failure modes that matter most in your codebase \u2014 which requires empirical testing, not reading rankings.'}]}]},{slug:"vibe-coding",title:"Vibe Coding: The Complete Honest Guide",titleDisplay:"Vibe",titleDisplayItalic:"Coding.",description:"What vibe coding actually is, which free tools work, whether you can get hired, and the data layer no tutorial mentions \u2014 30 questions answered without the hype.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["vibe coding","AI","development","tools","careers"],category:"AI Development",badge:"30 questions answered",fieldGuideLabel:"Field guide",deck:`Andrej Karpathy coined "vibe coding" on February 2, 2025. By November it was Collins Word of the Year. Most of what you'll read about it sells a tool. This is the version that tells you what works, what breaks, what it pays, and what you still have to build yourself after the prototype runs.`,readTimeMinutes:14,stats:[{value:"6",label:"sections"},{value:"30",label:"questions"},{value:"7",label:"tools compared"},{value:"0",label:"fluff"}],ctaHeading:"Your vibe-coded app needs",ctaHeadingItalic:"real data.",ctaBody:"MCP Scraper gives your AI-generated tools the live web data they need \u2014 SERP results, People Also Ask harvests, page extraction, and structured scraping \u2014 without writing a single line of custom scraping logic.",sections:[{id:"s1",num:"01",title:"What Is Vibe",titleItalic:"Coding?",deck:"The definition everyone links to, the origin story most posts get half-wrong, and the uncomfortable truth no one else in the top results will say.",callout:{eyebrow:"Origin story",heading:'"The hottest new programming language is English."',body:"Andrej Karpathy said that in 2023. On February 2, 2025, he named the practice: vibe coding. <strong>The tweet reached 4.5 million views in days.</strong> Merriam-Webster added the term on March 8, 2025. Collins named it Word of the Year on November 6, 2025. The definition matters because everyone is now selling their own version of it."},cards:[{id:"q1-1",num:"1.1",question:"Is vibe coding just having AI code for you?",answer:`<strong>Not exactly.</strong> In Karpathy's original framing, vibe coding is a specific mode: you describe what you want, you accept the code without reading it, and you treat bugs as a vibe to ride rather than a problem to debug. Merriam-Webster's listing makes the same point \u2014 coders don't need to understand how the code works and must accept that bugs will be present. That is narrower than "AI-assisted development," where you still review, accept, or reject every suggestion the model makes. Most products marketed as "vibe coding" today \u2014 Cursor, Copilot, Windsurf \u2014 are actually closer to AI-assisted development. <em>The distinction matters because the risk profile is completely different the moment you start reading the output.</em>`},{id:"q1-2",num:"1.2",question:"Is vibe coding difficult?",answer:`<strong>For prototypes, no. For anything you have to maintain, yes.</strong> The first working version of a tool \u2014 a price tracker, a Slack bot, a small dashboard \u2014 comes out of a vibe coding session in under an hour. The difficulty curve spikes the moment you try to extend it, debug a regression, or scale it past the original prompt. According to recent surveys, 66% of developers report spending more time fixing "almost-right" AI code than they save generating it. That gap is not in the tutorials. The skill that becomes hard is not writing code \u2014 it is reading code you didn't write, figuring out which AI suggestion to trust, and knowing when to throw the prototype away and rebuild it properly.`},{id:"q1-3",num:"1.3",question:"Is vibe coding a skill?",answer:"<strong>Yes \u2014 but not the skill most people expect.</strong> The transferable skill is prompt precision: describing system behavior clearly enough that the model produces the right thing on the first try. Add to that systems thinking (understanding how parts of an app talk to each other), and the judgment to spot when AI output is wrong before it ships. None of those require knowing syntax. All of them improve with practice. Vibe coders who succeed treat prompting as engineering \u2014 they iterate on prompts the way developers iterate on code, with versioned specs and explicit edge cases. Vibe coders who fail treat prompting as wishful thinking and re-prompt with the same vague description hoping for a better outcome."},{id:"q1-4",num:"1.4",question:"Can anyone be a vibe coder?",answer:`<strong>For personal tools and throwaway prototypes \u2014 yes.</strong> For anything with real users, the practical floor is higher than the marketing suggests. The clearest data point: 25% of Y Combinator's W25 cohort is running codebases that are almost entirely AI-generated, but every one of those teams has a technical founder who understands what the AI is producing. <em>The pattern in successful vibe-coded products is not "no engineering knowledge" \u2014 it is "engineering judgment without the syntax overhead."</em> If you have never thought about how data flows through a system, you can still ship a working prototype. You will struggle the first time it breaks in front of a real user, and that moment arrives faster than most tutorials let on.`},{id:"q1-5",num:"1.5",question:"What is the uncomfortable truth about vibe coding?",answer:`<strong>AI generates code that looks correct and isn't.</strong> The output reads cleanly, runs locally, and passes a casual review. It also reportedly contains SQL injection paths, leaked API keys in client-side bundles, overly permissive CORS configurations, and authentication logic that fails silently on edge cases. According to recent surveys, trust in AI code accuracy fell from 40% to 29% over the last year \u2014 not because the models got worse, but because more developers spent enough time with the output to see what it actually does. The "lower technical barrier" framing is honest for Day 0. It is actively misleading for Day 1 and beyond, when the cost of not understanding your own code starts showing up in production.`}]},{id:"s2",num:"02",title:"Vibe Coding",titleItalic:"Tools.",deck:"Seven tools ranked honestly \u2014 including which ones are free, which are worth paying for, and the one question the tool comparison tables never answer.",callout:{eyebrow:"Tool warning",heading:"No single tool handles Day 0 and Day 1+.",body:"Every tool is Day 0 optimized. None owns Day 1+. The emerging pattern: tools that let you build fast have weak maintenance stories. <strong>The moment you need to debug, audit, or extend AI-generated code, you're on your own.</strong> Pick your tool knowing this gap exists \u2014 and budget separately for the data, deployment, and review layers that none of them include."},cards:[{id:"q2-1",num:"2.1",question:"What is the best vibe coding platform for beginners?",answer:'<strong>Bolt.new and Replit are the two clearest entry points for non-developers.</strong> Both run entirely in the browser, both deploy a working app in one session, both require zero local setup. Bolt.new leans further toward "type what you want, see it run." Replit has a stronger long-term workspace because the project persists with files you can edit later. For developers who already have a code editor and want AI inside it, Cursor (2M+ users, $2B ARR) and Windsurf (1M+ active users) are the strongest options. The honest split: pick browser-based if you have never installed VS Code; pick Cursor or Windsurf if you have.'},{id:"q2-2",num:"2.2",question:"Are there any free vibe coding tools?",answer:"<strong>Yes. A definitive free-tier map.</strong> Bolt.new offers a free plan with daily token limits \u2014 enough to ship a small project, not enough to iterate heavily. Replit's free tier supports small public projects and runs in the browser. Aider is fully open-source and free, runs locally against any LLM you connect. Continue.dev is a free VS Code extension. Claude.ai and ChatGPT both offer free-tier chat that can generate complete code you paste into a runner. GitHub Copilot has a free tier for individuals. Cursor has a free tier with limited completions. <em>The only thing that is not free across all of these: heavy daily usage. Every product gates volume, not access.</em>"},{id:"q2-3",num:"2.3",question:"Which AI for vibe coding?",answer:'<strong>Claude and the GPT family are the two strongest code-generation models.</strong> For IDE-integrated use, two products dominate: GitHub Copilot leads on raw scale (20M total users, 4.7M paid subscribers, reportedly 42% of the AI coding assistant market) and Claude Code leads on satisfaction (91% CSAT in recent surveys, the highest of any AI coding tool measured). Developers using Copilot reportedly complete tasks 55% faster, with PR time dropping from 9.6 days to 2.4 days. The pragmatic answer: use Copilot if you live in VS Code and want the largest ecosystem; use Claude Code if you want the model most developers say produces the fewest "almost-right" answers.'},{id:"q2-4",num:"2.4",question:"Which AI agent is best for vibe coding?",answer:"<strong>For autonomous multi-step tasks, the leaders are Cursor's Composer, Windsurf's Cascade, and Claude Code.</strong> Cascade handles the most aggressive agentic workflows out of the box \u2014 multi-file refactors, end-to-end feature builds \u2014 and Windsurf was acquired by Cognition in December 2025 for reportedly ~$250M, which consolidated agentic capability under one roof. Claude Code is the highest-satisfaction agentic tool in developer surveys. <em>The honest tradeoff: more autonomy means less review, which means more time fixing surprises later.</em> Pick the level of autonomy you actually want to audit."},{id:"q2-5",num:"2.5",question:"Can you vibe code with ChatGPT?",answer:"<strong>Yes, but ChatGPT alone is a copy-paste workflow.</strong> You describe what you want, ChatGPT generates the code, you paste it into a runner \u2014 Replit, your local terminal, a CodeSandbox tab. That works, and it is genuinely free at the entry tier. The friction is everything between paste and run: you don't get inline edits, you don't get file-level context, and you re-paste the whole project every time you want a change. A better free path is ChatGPT plus Replit (ChatGPT writes, Replit runs and persists) or ChatGPT inside an IDE through a plugin. ChatGPT is a viable free-tier route into vibe coding. It is not the best long-term home for a real project."},{id:"q2-6",num:"2.6",question:"What's the best app for vibe coding?",answer:`<strong>"App" depends on the device.</strong> On desktop: Cursor is the most powerful for developers, Bolt.new is the fastest for non-coders, and Replit is the most balanced if you want both browser convenience and a real workspace. On mobile: Replit's iOS app is the only environment that supports full project builds from a phone. <em>Claude.ai and ChatGPT also run in mobile browsers and can generate code you deploy elsewhere, but neither is a complete vibe coding environment on its own.</em> The next section covers iPhone-specific workflows in detail \u2014 most posts skip this entirely, even though the question shows up clearly in search.`}]},{id:"s3",num:"03",title:"How to Start",titleItalic:"Vibe Coding.",deck:"A step-by-step path for non-developers, the one thing every tutorial skips (getting real data into your tool), and the iPhone question every other guide ignores.",callout:{eyebrow:"Critical gap",heading:"The prototype always works. The live version needs a data layer.",body:"Prompting AI to build a tool is 30 minutes. Getting it live data is where most projects die. Every useful vibe-coded tool eventually needs to read from the real web. Price trackers need prices. Research tools need pages. Lead generators need business data. <strong>That's where MCP Scraper enters the workflow.</strong>"},cards:[{id:"q3-1",num:"3.1",question:"How should I start vibe coding?",answer:'<strong>Three steps. In this order.</strong> First, pick a browser-based tool \u2014 Bolt.new or Replit \u2014 so you skip every "install this, configure that" trap that kills momentum on day one. Second, describe one specific thing you want to build, not a general app idea. "A tool that emails me when the price of these three Amazon products drops" beats "an e-commerce price tracker" because the model can build the first one in one pass. Third, test the output immediately and iterate with precise corrections \u2014 name the file, the function, and the exact behavior you want changed. <em>The most common beginner failure is prompting vaguely and then re-prompting with more vagueness, hoping the model figures it out.</em>'},{id:"q3-2",num:"3.2",question:"Can you learn coding from vibe coding?",answer:"<strong>Yes for systems thinking. No for syntax.</strong> Vibe coding teaches you to think in terms of inputs, outputs, state, and failure modes \u2014 the conceptual layer that makes a developer effective. It teaches debugging logic, because you spend real time reading errors and asking the model what they mean. It teaches prompt precision, which transfers to spec-writing in any technical role. What it does not teach is syntactic fluency, language-specific idioms, or low-level architecture decisions. If your goal is to be hired as a traditional software engineer at a company that interviews on syntax, vibe coding is an accelerator for concepts and a poor substitute for fundamentals. If your goal is to ship products, the concepts matter more."},{id:"q3-3",num:"3.3",question:"Can I vibe code on my iPhone?",answer:"<strong>Yes, with real limitations.</strong> Replit has a functional iOS app that supports full project builds, file editing, and deployment from your phone. Claude.ai and ChatGPT both run in Safari and can generate complete code you paste into Replit or another runner. Bolt.new is browser-accessible on mobile, though the desktop layout is the supported experience. No current iOS tool matches the desktop IDE experience for serious projects. For prototyping a small tool, drafting a Slack bot, or sketching the first version of an app idea, your iPhone is a viable vibe coding environment \u2014 and it is the only environment most travelers have on day one of an idea."},{id:"q3-4",num:"3.4",question:"Does Apple ban vibe coded apps?",answer:"<strong>No.</strong> Apple's App Store Review Guidelines do not categorically ban AI-generated code. What they ban \u2014 and have always banned \u2014 are thin wrappers around web content, spam apps with no original functionality, and apps that violate content or privacy policies. None of those rules are AI-specific. The accountability standard is unchanged: the developer is responsible for the app's behavior, regardless of how the code was produced. <em>A vibe-coded native app that genuinely does something useful for the user, handles data responsibly, and meets the same review bar as any other app will pass review.</em> A vibe-coded app that wraps a website in a webview and adds nothing will not \u2014 and would not have passed review in 2018 either."},{id:"q3-5",num:"3.5",question:"Is vibe coding good?",answer:`<strong>For prototyping and personal tools \u2014 unambiguously yes. For production at scale \u2014 only with engineering oversight.</strong> The data supports both halves. 84% of developers use or plan to use AI coding tools, with average savings of 3.6 hours per week. 25% of Y Combinator's W25 cohort runs nearly all-AI-generated codebases. At the same time, 66% of developers report spending more time fixing "almost-right" AI code than they save generating it, according to recent surveys. The honest read: vibe coding is excellent for the 80% of ideas that never needed production-grade code in the first place, and it is genuinely risky for the 20% that do. The skill is knowing which one you're building.`}]},{id:"s4",num:"04",title:"The Real",titleItalic:"Limits.",deck:"Security vulnerabilities, the Day 0 vs. Day 1+ cliff, what happens when AI-generated code hits production \u2014 and how to protect yourself.",callout:{eyebrow:"Industry stat",heading:"66% of developers reportedly spend more time fixing AI code than generating it.",body:"This is not an argument against AI coding tools. It is an argument for using them with your eyes open. <strong>The productivity gains are real \u2014 if you know when to trust the output and when to audit it.</strong>"},cards:[{id:"q4-1",num:"4.1",question:"Do vibe coders understand their code?",answer:"<strong>Most do not, by design.</strong> Karpathy's original framing is explicit: you give in to the vibes, you accept the code without reading it, you let the model handle bugs by reprompting rather than debugging. Merriam-Webster's definition repeats the same idea \u2014 practitioners don't need to understand how the code works and must accept bugs will be present. That is fine for personal tools you can throw away. It is a serious liability for anything deployed to real users: you cannot debug a system you do not understand, you cannot catch security issues you never look for, and you cannot tell a user with confidence what your app actually does with their data. The Day 0 advantage becomes the Day 1+ liability."},{id:"q4-2",num:"4.2",question:"What are the security risks of vibe coding?",answer:'<strong>AI-generated code commonly introduces SQL injection vulnerabilities, insecure API key handling, overly permissive CORS configurations, and authentication logic that fails silently on edge cases.</strong> None of these are visible to a non-developer reviewing the output, because the code looks correct. The mitigation is not "review your code more carefully" \u2014 that asks vibe coders to do exactly what they came here not to do. The mitigation is an automated security scanner. Snyk, GitHub Advanced Security, and Semgrep each run continuously and flag the common AI-generated mistakes before deploy. <em>Pair every vibe coding session with a scanner that runs on commit, and you eliminate the most common production-breaking class of mistake without learning to read every line.</em>'},{id:"q4-3",num:"4.3",question:"When should you not vibe code?",answer:"<strong>Three hard cases.</strong> First, systems handling personal data under GDPR, HIPAA, PCI-DSS, or similar \u2014 AI-generated code needs auditing you cannot DIY, and the regulatory penalty for getting it wrong is higher than the time you saved. Second, financial transaction logic \u2014 silent rounding errors, race conditions, and authorization gaps compound quickly and quietly. Third, anything you need to maintain for more than 12 months without a developer \u2014 AI-generated codebases become unmaintainable faster than hand-written code because the structure was never designed for change. <em>Vibe code freely when the cost of being wrong is small. Bring in engineering when the cost of being wrong is asymmetric.</em>"},{id:"q4-4",num:"4.4",question:"What is the Day 1+ problem in vibe coding?",answer:"<strong>Every vibe coding tool is optimized for the first working build. None of them is optimized for what comes after.</strong> Day 1+ problems are predictable: adding a second feature without breaking the first, debugging a regression you cannot trace, scaling to enough users that the original architecture cracks, and integrating third-party APIs whose contracts change. The tools do not advertise this gap because Day 0 demos sell better than maintenance demos. One developer-practitioner survey concluded that no single tool today can build and maintain an entire application end-to-end. <em>Plan for the Day 1+ moment before it arrives. Pick a tool with file-level access, version control, and an escape hatch into real code review.</em>"},{id:"q4-5",num:"4.5",question:"Can a vibe-coded app go viral?",answer:"<strong>Yes \u2014 with documented examples.</strong> One builder shipped more than ten vibe-coded apps that were used reportedly close to a million times in total before scaling and maintenance pressure forced a pause. Kevin Roose's LunchBox Buddy \u2014 a fridge-photo-to-meal-suggestion tool \u2014 was cited in the New York Times as an early vibe coding demo. Refetch, an open-source Hacker News alternative, was reportedly built in 15 hours of vibe coding on Appwrite Cloud. The pattern across these cases is consistent: vibe-coded apps scale to early traction with no problem, and the crisis arrives when traffic, data volume, or feature scope exceeds what the original AI-generated architecture was built for. That crisis is solvable. It just doesn't solve itself."}]},{id:"s5",num:"05",title:"Career &",titleItalic:"Hiring.",deck:"Six questions no competitor will answer \u2014 whether vibe coding is a real job, what it pays, and how to position it when you're applying.",callout:{eyebrow:"Hiring signal",heading:"The fastest-growing startups have already decided vibe coding is production-ready.",body:"25% of Y Combinator's current cohort runs codebases that are almost entirely AI-generated. IBM cites the stat. Google ignores it. Medium skips it. None asks the obvious next question: what does that mean for the person reading this article? <strong>The job market is catching up.</strong>"},cards:[{id:"q5-1",num:"5.1",question:"Is vibe coding a real job now?",answer:`<strong>It is becoming one \u2014 fastest at startups, slowest at large enterprises.</strong> Early-stage companies increasingly list "AI-assisted developer," "AI product builder," "prompt engineer," and "no-code/AI builder" roles. Freelance platforms like Upwork and Fiverr have active vibe coding service categories with steady project volume. At enterprise scale, formal "vibe coder" titles are still rare \u2014 but the practice is embedded in 18% of developers' day-to-day work according to JetBrains' January 2026 survey, and at the YC startups where 25% of codebases are nearly all AI-generated, it is the default daily practice. <em>The title is lagging the work by roughly 18 months. The work is already mainstream.</em>`},{id:"q5-2",num:"5.2",question:"Do companies hire vibe coders?",answer:'<strong>Early-stage startups and solo-founder companies actively do.</strong> Common titles include "growth engineer," "founding engineer," "AI product builder," and "technical founder in residence." The hiring signal is clearest in YC-backed companies and Series A startups where shipping speed matters more than code purity. Traditional enterprise software companies have been slower \u2014 their hiring processes are designed to test syntax and system-design fundamentals, which vibe coders often have not formally studied. The market direction is consistent with broader adoption: the AI coding tools market reportedly reached $7.37 billion in 2025, and 84% of developers use or plan to use AI tools. <em>The companies hiring fastest are the ones building fastest.</em>'},{id:"q5-3",num:"5.3",question:"Do vibe coders get hired?",answer:"<strong>If you can show working products \u2014 yes. The portfolio matters more than the title.</strong> Demonstrating that you shipped a functional tool used by real people is more compelling to early-stage hiring managers than a CS degree or a coding bootcamp certificate. A live URL with real users beats a GitHub repo with no traction. The friction point is larger companies whose engineering interviews are designed around whiteboard syntax problems vibe coders have not drilled on. The pragmatic path: build five shippable tools, get real users for at least one, document what you built and what you learned, and apply to companies whose hiring is portfolio-driven rather than interview-driven. The first job is the hardest. After the first job, the portfolio compounds."},{id:"q5-4",num:"5.4",question:"How much do vibe coders make?",answer:`<strong>Ranges vary by context. Approximate market signal as of 2026:</strong> Freelance project rates on Upwork and Fiverr for AI-built tools reportedly range from $500 to $5,000 per project, depending on scope and the client's budget. Full-time "AI-assisted developer," "AI product builder," and "founding engineer" roles at startups reportedly range from $80K to $140K base depending on seniority and location, with equity on top. Indie hackers shipping revenue-generating vibe-coded apps have publicly reported product revenue from $1K to $20K per month, with outliers higher. <em>The ceiling scales with what you build, not what you know. Salaried roles cap your upside; products do not.</em> Treat these as ballparks \u2014 no single survey aggregates them yet, and the market is moving monthly.`},{id:"q5-5",num:"5.5",question:"Is vibe coding a real job?",answer:`<strong>It depends on which job market you are targeting.</strong> In the startup and indie developer ecosystem \u2014 yes, building with AI is a marketable practice with paying roles and growing demand. In regulated industries (healthcare, finance, defense) and large enterprise environments \u2014 not yet as a standalone role, because compliance and code-audit requirements still demand engineers who can read every line. The trajectory is clearly toward normalization: AI reportedly writes about 41% of all new code today, 84% of developers are using or planning to use AI tools, and 25% of YC's current cohort runs nearly-all-AI-generated codebases. <em>The current state is early but real. The forward curve points toward "yes" being the default answer within two to three years.</em>`}]},{id:"s6",num:"06",title:"The Future of",titleItalic:"Coding.",deck:"Whether AI will replace coders, what the realistic 2026\u20132040 arc looks like, and the one skill that becomes more valuable as AI writes more code.",callout:{eyebrow:"Key take",heading:"AI replaces code-writing. It doesn't replace problem-solving.",body:"The developers who will struggle are those whose value is in typing code fast. <strong>The developers who will thrive are those whose value is in knowing what to build, how systems should behave, and when AI output is wrong.</strong> Vibe coding is the fastest way to find out which one you are."},cards:[{id:"q6-1",num:"6.1",question:"Will coders be replaced by AI?",answer:'<strong>Rote code-writing is already being replaced.</strong> AI reportedly writes about 41% of all code today, and 84% of developers use or plan to use AI coding tools. The role-level effect is more specific than "coders are obsolete": developers who specialize in syntax and straightforward implementation are most exposed; developers who specialize in architecture, system design, and problem framing are least exposed. The practice of writing code is being automated. The judgment of what to build, why to build it, and how to know whether it works is not \u2014 and there is no current evidence that it will be soon. <em>"Coder" is becoming a smaller part of "software developer." The other parts are growing.</em>'},{id:"q6-2",num:"6.2",question:"Will AI replace coders by 2040?",answer:"<strong>By 2040, AI will likely generate the majority of code by volume.</strong> What survives \u2014 and what the labor market will pay for \u2014 is the meta-skill: defining problems precisely, evaluating AI output critically, and directing systems toward intended outcomes. The vibe coder of 2026 who develops those meta-skills is better positioned than the traditional developer who ignores them. The career risk over the next 15 years is not AI itself. It is the choice not to adapt to AI. The 25% YC cohort statistic is the leading indicator: the fastest-growing companies are already operating at the model that the rest of the market will reach. Plan accordingly."},{id:"q6-3",num:"6.3",question:"What skill becomes most valuable as AI writes more code?",answer:"<strong>Prompt precision.</strong> The ability to describe complex system behavior clearly enough that AI produces the right output on the first try is the emerging premium skill. Closely paired with it: the ability to audit AI output for correctness, security, and architectural soundness \u2014 not by reading every line, but by knowing which questions to ask and which automated checks to run. Both are learnable by non-developers. Neither requires fluency in any programming language. <em>The developers who treat prompting as engineering \u2014 versioned, specified, tested \u2014 are already pulling ahead of the developers who treat it as a chat.</em> That gap will widen."},{id:"q6-4",num:"6.4",question:"What does a vibe-coded app need to work in the real world?",answer:"<strong>Three things \u2014 and the middle one is where most vibe coders stall.</strong> First, a deployment target: Vercel, Replit, Railway, Fly.io. Second, real data access \u2014 every useful live tool eventually needs to read from the web, and writing custom scrapers, handling JavaScript-rendered pages, and maintaining selectors as sites change is the wall most prototypes hit. Third, a feedback loop with actual users. The data access layer is the most underestimated step in the entire vibe coding workflow. MCP Scraper is the data layer vibe coders reach for when the prototype needs to start consuming real-world inputs \u2014 SERP results, People Also Ask trees, page extraction, YouTube transcripts \u2014 without writing and maintaining the scraping code themselves."}]}]},{slug:"people-also-ask-seo",title:"People Also Ask SEO: The Complete 2026 Guide",titleDisplay:"People Also Ask",titleDisplayItalic:"SEO.",description:"What PAA boxes are, how Google generates them, which tools harvest them at scale, and why the manual two-step loop breaks the moment you need programmatic data.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["SEO","people also ask","content strategy","PAA","keyword research"],category:"SEO",badge:"31 questions answered",fieldGuideLabel:"Field guide",deck:"PAA strategy ends where programmatic begins. Every competitor guide shows you how to expand boxes \u2014 none shows you how to harvest them at scale via API and pipe the data directly into your content system.",readTimeMinutes:12,stats:[{value:"85%",label:"of Google searches show PAA boxes"},{value:"13.6%",label:"CTR for purchase-intent PAA clicks"},{value:"31",label:"questions answered in this guide"},{value:"90%",label:"of AI Overviews include PAA boxes"}],ctaHeading:"Stop copying questions from a SERP accordion.",ctaHeadingItalic:"MCP Scraper delivers structured PAA trees via API \u2014 one call, any keyword, directly into your pipeline.",ctaBody:"MCP Scraper is the PAA harvesting API for developers and SEOs who need more than 4 visible questions. Extract hundreds of People Also Ask questions from any query via REST or MCP.",sections:[{id:"s1",num:"01",title:"What Is People",titleItalic:"Also Ask?",deck:"PAA mechanics, Google behavior, and what the AI-generation shift means for the feature \u2014 the foundational questions most guides leave unanswered.",callout:{eyebrow:"Key take",heading:"PAA is not a content placement. It is a real-time map of query intent.",body:"Google surfaces PAA on roughly 85% of searches. Winning a placement is useful. Understanding why Google shows the questions it does \u2014 and how 12.6% of answers are now AI-generated \u2014 is what separates a tactic from a strategy. <strong>Optimize the placement. Understand the system.</strong>"},cards:[{id:"q1-1",num:"1.1",question:"What do people also ask in SEO?",answer:`<strong>People Also Ask (PAA) is a dynamic SERP accordion Google introduced in 2016</strong> that surfaces related questions predicted to follow a searcher's original query. It now appears on roughly <strong>85% of searches</strong> \u2014 not a niche placement opportunity but a near-universal feature that functions as Google's real-time map of query intent for every topic. Early testing began in April 2015 as "Related questions" before the official July 2016 naming. For SEOs, PAA is two things simultaneously: a visibility placement to win, and a research signal showing which follow-up questions Google believes matter most to your searchers.`},{id:"q1-2",num:"1.2",question:"How does Google generate people also ask?",answer:'Google generates PAA algorithmically from <strong>co-occurrence patterns in search sessions</strong> \u2014 which queries follow which, and what content satisfies them. One significant recent shift: <strong>12.6% of PAA answers are now AI-generated by Google itself</strong>, not pulled from any web page. This means winning a PAA placement no longer guarantees your content is the displayed source. The answer Google shows may be synthesized entirely from its own models, with your page as a citation at best. Understanding this shift changes the strategic goal from "win the placement" to "be the authoritative source Google trusts when it generates the answer."'},{id:"q1-3",num:"1.3",question:"What does it mean when Google says people also searched for?",answer:'<strong>"People also searched for" is a distinct SERP feature</strong> \u2014 it appears after a user clicks a result and returns to the search page, surfacing refinement queries based on what searchers do next. <strong>PAA appears before the click</strong>, predicting follow-up intent from the query itself. They signal different moments in the search session and target different optimization strategies. "People also searched for" is a post-click refinement signal; PAA is a pre-click intent prediction. Conflating the two leads to misaligned content strategy \u2014 targeting refinement queries when your page needs to answer the predicted follow-ups, or vice versa.'},{id:"q1-4",num:"1.4",question:"How do Google people also ask work?",answer:"<strong>PAA boxes are infinite-scroll</strong>: expanding one question loads 2\u20134 additional questions, allowing Google to map an entire topic graph from a single seed query. The block appears after the first organic result in more than <strong>58% of SERPs</strong>, making it a top-three SERP element in roughly two-thirds of all searches. Featured answers average <strong>40\u201350 words</strong>. The infinite-scroll mechanism is the most operationally important detail for SEOs: a single seed keyword can branch into hundreds of related questions, which is why harvesting PAA programmatically returns a fundamentally different volume of data than manually expanding a visible accordion."},{id:"q1-5",num:"1.5",question:"How to find people also ask?",answer:"Three methods, in order of scale: <strong>manual SERP expansion</strong> \u2014 open Google, type your keyword, click the accordion arrows (free, works at 1\u20135 keywords); <strong>UI tools like AlsoAsked</strong>, which automate question collection and return visual question trees ($12\u2013$47/mo, works at 5\u20131,000 keywords); and <strong>programmatic API access via MCP Scraper</strong>, which returns structured PAA data for any keyword without a browser (works at any scale). The right method depends entirely on how many keywords you need to cover. The manual path is not inferior \u2014 it is simply volume-limited, and that limit arrives faster than most SEOs expect."},{id:"q1-6",num:"1.6",question:"What type of questions can Google not answer?",answer:"PAA avoids four structural categories: <strong>highly time-sensitive breaking news</strong> (where no authoritative answer has stabilized), <strong>deeply personal queries</strong> (medical, legal, financial specifics tied to individual circumstance), <strong>normative controversies</strong> where no consensus source exists, and <strong>queries where the intent is too ambiguous</strong> for any single question formulation to make sense. Understanding these boundaries helps SEOs identify where PAA placements are structurally off the table \u2014 not due to competition, but due to feature design. If your topic falls into one of these categories, optimize for other SERP features rather than PAA."}],interactiveHtml:`<div class="si si-calculator"><span class="si-heading">How many PAA questions does one keyword unlock?</span><div class="si-calc-inputs"><label class="si-label">Seed keywords<div class="si-slider-row"><input type="range" class="si-range" id="sip1-s" min="1" max="50" value="3"><span class="si-range-val" id="sip1-sv">3</span></div></label><label class="si-label">Expansion depth (click levels)<div class="si-slider-row"><input type="range" class="si-range" id="sip1-d" min="1" max="3" value="2"><span class="si-range-val" id="sip1-dv">2</span></div></label></div><div class="si-calc-output"><div class="si-stat"><span class="si-stat-num" id="sip1-m">12</span><span class="si-stat-label">questions visible manually</span></div><div class="si-stat si-stat-accent"><span class="si-stat-num" id="sip1-f">252</span><span class="si-stat-label">questions in the full PAA tree</span></div></div><script>(function(){var s=document.getElementById('sip1-s'),d=document.getElementById('sip1-d'),sv=document.getElementById('sip1-sv'),dv=document.getElementById('sip1-dv'),mo=document.getElementById('sip1-m'),fo=document.getElementById('sip1-f');function upd(){var ss=+s.value,dd=+d.value;sv.textContent=ss;dv.textContent=dd;mo.textContent=ss*4;fo.textContent=ss*(dd===1?28:dd===2?84:252);}s.addEventListener('input',upd);d.addEventListener('input',upd);})()</script></div>`},{id:"s2",num:"02",title:"PAA as an",titleItalic:"SEO Strategy",deck:"Intent mechanics, keyword discovery, and why PAA is a top-5 SEO strategy because of AI Overviews \u2014 not despite them.",callout:{eyebrow:"The number that changes the strategy",heading:"Purchase-intent PAA queries drive 13.6% interaction rates. Overall PAA: 3%.",body:"The 4\xD7 gap is not a curiosity \u2014 it is the entire prioritization framework. <strong>Which questions are worth targeting first is a data problem, and the data problem requires harvesting tools, not intuition.</strong>"},cards:[{id:"q2-1",num:"2.1",question:"What are the 4 types of intent in SEO?",answer:"The four intent types \u2014 <strong>informational, navigational, commercial, transactional</strong> \u2014 are not equally valuable in PAA strategy. Purchase-intent queries drive a <strong>13.6% PAA interaction rate</strong>, versus 3% for searches overall. That 4\xD7 gap means intent classification is not just a content exercise \u2014 it is the highest-leverage variable in deciding which PAA questions are worth targeting first. Most SEO teams apply intent classification to keyword strategy but never extend it to PAA prioritization. Applying it there is the arbitrage: harvest PAA for purchase-intent and commercial queries first, and the ROI per question answered separates immediately from the pack."},{id:"q2-2",num:"2.2",question:"What are the 3 C's of search intent?",answer:"Content type, content format, and content angle \u2014 the <strong>3 C's</strong> \u2014 determine structural eligibility for a PAA placement before Google evaluates topical relevance. <strong>Format matters most</strong>: PAA answers average 40\u201350 words, which means a 2,000-word section cannot win a placement regardless of its quality. PAA requires an extractable, self-contained answer at the right word count. The content angle determines whether your framing matches the specific question Google is showing \u2014 a page that answers the general topic but not the exact question in the accordion is not eligible. The 3 C's applied to PAA become a pre-qualification checklist, not an afterthought."},{id:"q2-3",num:"2.3",question:"What is the 80/20 rule in SEO?",answer:"Applied to PAA, the 80/20 rule holds in the data: the <strong>13.6% interaction rate for purchase-intent queries versus 3% overall</strong> suggests that a small subset of PAA questions drives disproportionate engagement and commercial value. Identifying that subset \u2014 by intent type and query category \u2014 is precisely what PAA harvesting tools solve, because the questions that matter most are rarely obvious from the seed keyword alone. A human reviewing a SERP accordion sees 4 questions. A programmatic harvest of the full PAA tree for that seed can surface 200. The 80/20 rule applies to that full dataset, not to the visible 4."},{id:"q2-4",num:"2.4",question:"What are the best SEO tools?",answer:"For PAA strategy specifically, a complete stack looks like: <strong>Google Search Console</strong> (free, shows existing PAA appearances for your domain), <strong>AlsoAsked</strong> (UI-based question trees, $12\u2013$47/mo, best for manual research up to 1,000 seeds), <strong>Semrush</strong> (broad SERP feature tracking), and <strong>MCP Scraper</strong> (programmatic PAA API for pipeline integration). The right combination depends on whether your workflow is UI-based or data-pipeline-based. Teams doing content at scale need both tiers \u2014 the UI tool for ad-hoc research and the API for systematic collection. The two are complementary, not competing."},{id:"q2-5",num:"2.5",question:"Which is the best free SEO tool for beginners?",answer:"For PAA research specifically, <strong>Google Search Console</strong> is the strongest free starting point \u2014 it shows which PAA features your site already appears in, at no cost. <strong>AlsoAsked offers free monthly credits</strong> without requiring a registered account. MCP Scraper is not the beginner recommendation; it requires API literacy and suits developers moving into structured data workflows, not first-time SEOs. The honest path for a beginner: start with Search Console to see what you already have, use AlsoAsked free credits to map the questions you're missing, and add the API layer when manual research becomes the rate-limiting step in your workflow."},{id:"q2-6",num:"2.6",question:"What are the top 5 SEO strategies?",answer:"PAA optimization earns a top-5 slot not because its direct interaction rate is high (3% overall) but because of its relationship with AI Overviews: <strong>PAA co-appears with AI Overviews in 90% of cases</strong>. Winning a PAA placement now doubles as qualifying content to be sourced by AI Overviews \u2014 a compound visibility return that traditional link-building cannot replicate. Optimizing for PAA is, in practice, optimizing for AI Overview sourcing eligibility. The 90% co-appearance figure means these two features share the same content signal. Teams that ignore PAA are leaving the most reliable AI Overview proxy on the table."}],interactiveHtml:`<div class="si si-decision"><span class="si-heading">Is your PAA strategy leaving data on the table?</span><div class="si-questions"><div class="si-q"><p class="si-q-text">Do you prioritize PAA questions by intent type \u2014 targeting purchase-intent questions before informational ones?</p><div class="si-q-opts"><label class="si-opt"><input type="radio" name="sip2-q0" value="1"> Yes \u2014 intent drives my question selection</label><label class="si-opt"><input type="radio" name="sip2-q0" value="0"> No \u2014 I target all PAA questions equally</label></div></div><div class="si-q"><p class="si-q-text">Do you refresh PAA-targeted content at least once per quarter?</p><div class="si-q-opts"><label class="si-opt"><input type="radio" name="sip2-q1" value="1"> Yes \u2014 freshness is part of my process</label><label class="si-opt"><input type="radio" name="sip2-q1" value="0"> No \u2014 I optimize once and move on</label></div></div><div class="si-q"><p class="si-q-text">Is your target keyword list larger than 50 keywords?</p><div class="si-q-opts"><label class="si-opt"><input type="radio" name="sip2-q2" value="1"> Yes \u2014 50 or more keywords</label><label class="si-opt"><input type="radio" name="sip2-q2" value="0"> No \u2014 fewer than 50</label></div></div></div><div class="si-result" id="sip2-result" hidden><p class="si-result-label" id="sip2-rl"></p><p class="si-result-body" id="sip2-rb"></p></div><script>(function(){var results=[{label:'Manual research fits your current scale.',body:'Your keyword list is manageable and your process is structured. No tool changes needed yet \u2014 the manual path handles your volume.'},{label:'You are at the edge of what manual research supports.',body:'One gap in intent prioritization or freshness cadence is limiting your results. Focusing on purchase-intent PAA first is the highest-ROI next step.'},{label:'Your PAA strategy needs a data layer.',body:'At 50+ keywords with systematic intent filtering and quarterly freshness requirements, programmatic access is no longer optional \u2014 manual research is your rate-limiting step.'}];function check(){var score=0,answered=0;for(var i=0;i<3;i++){var r=document.querySelector('input[name="sip2-q'+i+'"]:checked');if(r){score+=+r.value;answered++;}}if(answered<3)return;var res=score>=2?results[2]:score===1?results[1]:results[0];var el=document.getElementById('sip2-result');document.getElementById('sip2-rl').textContent=res.label;document.getElementById('sip2-rb').textContent=res.body;el.hidden=false;}['sip2-q0','sip2-q1','sip2-q2'].forEach(function(n){document.querySelectorAll('input[name="'+n+'"]').forEach(function(r){r.addEventListener('change',check);});});})()</script></div>`},{id:"s3",num:"03",title:"AI and the",titleItalic:"Future of PAA",deck:"Is SEO dead? Will AI replace it? Committed answers backed by data \u2014 including why AI Overviews make PAA more important, not less.",callout:{eyebrow:"Counter-intuitive finding",heading:"AI Overviews make PAA more important, not less.",body:"PAA co-appears with AI Overviews in 90% of searches. Winning a PAA placement is currently the most reliable proxy for AI Overview sourcing eligibility. <strong>The teams abandoning PAA because of AI are removing themselves from the exact feature that signals AI citation readiness.</strong>"},cards:[{id:"q3-1",num:"3.1",question:"Is SEO dead or evolving in 2026?",answer:"<strong>Structurally shifting, not dying</strong> \u2014 but the shift is significant. 58.5% of US Google searches already end without a click. AI Overviews reduce position-1 CTR by up to 58%, and searches triggering AI Overviews show an 83% zero-click rate. <strong>PAA co-appears with AI Overviews in 90% of cases.</strong> The game has moved from generating clicks to being cited as an answer source \u2014 visibility without click-through is the new baseline. The SEOs who treat this as a crisis are measuring the wrong thing. The SEOs who treat it as a repositioning opportunity are already optimizing for sourcing frequency, not position rank."},{id:"q3-2",num:"3.2",question:"Will SEO be replaced by AI?",answer:"<strong>AI is not replacing SEO</strong> \u2014 it is automating the parts that were manual and amplifying the parts that require data access. The practitioners who win are those using AI for content generation while feeding it with live, structured SEO data: PAA trees, SERP features, intent signals. The ones who lose are optimizing one page at a time while their competitors run data pipelines that update daily. The replacement narrative confuses the tool with the discipline. SEO as structured analysis of how content earns visibility in search systems is not going anywhere. The execution layer is being automated. The strategic layer is becoming more valuable, not less."},{id:"q3-3",num:"3.3",question:"What is SEO being replaced by?",answer:"The emerging term is <strong>answer-engine optimization (AEO)</strong> \u2014 structuring content so AI systems cite it as a source, not just rank it in blue links. PAA boxes are currently the most reliable signal for what questions AI Overviews will answer from a given domain, because the two features co-appear in <strong>90% of cases</strong>. Winning PAA now is training data sourcing practice for the AI-first SERP. AEO is not a replacement for SEO \u2014 it is an extension of it toward the citation layer. The content signals that earn PAA placements and the signals that earn AI Overview citations overlap significantly. PAA optimization is AEO in its most accessible form."},{id:"q3-4",num:"3.4",question:"Can ChatGPT write SEO articles?",answer:"<strong>ChatGPT can draft content but cannot perform PAA research</strong> \u2014 it has no live SERP access, returns no real-time question clusters, and cannot tell you which questions Google is surfacing for your keyword today. PAA data is a live signal that requires querying the SERP, not a language model. Any AI-assisted SEO workflow that skips a live data layer is building content strategy on stale assumptions. The correct architecture is: live PAA data (from a harvesting API) feeding structured inputs to the LLM, with the LLM handling drafting and the data layer handling research. Skipping the data layer produces well-written answers to questions no one is actually asking."},{id:"q3-5",num:"3.5",question:"Can ChatGPT do an SEO audit?",answer:"<strong>ChatGPT can review content structure</strong>, flag missing headers, and suggest improvements to on-page copy \u2014 but it cannot audit live SERP features, check which PAA questions your competitors currently hold, or identify PAA placement gaps across your keyword set. Those tasks require real-time structured data pulled from the SERP itself, not pattern-matching against training data. The distinction is not about writing quality \u2014 it is about data access as a category. LLM-based audits are useful for qualitative content review. They are structurally incapable of competitive PAA analysis, which requires live harvesting at query time."},{id:"q3-6",num:"3.6",question:"Which AI is best for SEO?",answer:'<strong>No single AI model is "best for SEO"</strong> \u2014 the question frames it wrong. The winning configuration is: a live data API (PAA harvesting, SERP feature extraction) feeding structured inputs to an LLM for content generation and optimization. The AI model handles language; the data layer handles reality. MCP Scraper occupies the data layer slot \u2014 it supplies the question data that makes content decisions defensible rather than intuitive. Conflating the two is why AI-assisted SEO often produces content that reads well but targets the wrong questions. The model choice matters far less than whether your workflow has a live data layer at all.'}],interactiveHtml:`<div class="si si-quiz"><span class="si-heading">Quick check</span><div class="si-quiz-q" data-correct="1"><p class="si-quiz-q-text">PAA co-appears with AI Overviews in what percentage of searches?</p><div class="si-quiz-opts"><button class="si-quiz-opt" data-idx="0">23% of searches</button><button class="si-quiz-opt" data-idx="1">90% of searches</button><button class="si-quiz-opt" data-idx="2">58% of searches</button></div><p class="si-quiz-explanation" hidden>The 90% co-appearance rate is the number that changes everything \u2014 winning a PAA placement is currently the most reliable proxy for AI Overview sourcing eligibility. Teams abandoning PAA because of AI are removing themselves from the exact signal that governs AI citations.</p></div><script>(function(){document.querySelectorAll('.si-quiz .si-quiz-q').forEach(function(qBlock){var correct=+qBlock.dataset.correct;var explanation=qBlock.querySelector('.si-quiz-explanation');qBlock.querySelectorAll('.si-quiz-opt').forEach(function(btn){btn.addEventListener('click',function(){if(qBlock.dataset.answered)return;qBlock.dataset.answered='1';var idx=+btn.dataset.idx;qBlock.querySelectorAll('.si-quiz-opt').forEach(function(b,i){b.disabled=true;if(i===correct)b.classList.add('si-quiz-correct');else if(i===idx&&idx!==correct)b.classList.add('si-quiz-wrong');});explanation.hidden=false;});});});})()</script></div>`},{id:"s4",num:"04",title:"PAA Tools",titleItalic:"Compared",deck:"AlsoAsked vs. Semrush vs. MCP Scraper \u2014 honest tradeoffs, including where MCP Scraper is and is not the right choice.",cards:[{id:"q4-1",num:"4.1",question:"How does AlsoAsked work?",answer:"<strong>AlsoAsked crawls Google PAA boxes for a given keyword</strong> and builds a branching tree of related questions, which it presents as a visual map, PNG export, or CSV. Bulk upload processes up to 1,000 seed terms in a single job, returning a large question set from recursive PAA expansion. It includes multi-region and multi-language support, API access, and webhook integration across all paid tiers. AlsoAsked is best positioned for UI-based workflows where a researcher is manually reviewing and selecting questions. The limit of the approach is pipeline integration: CSV exports require a human in the loop, and the API, while available on all paid tiers, is designed around the same single-query model as the UI."},{id:"q4-2",num:"4.2",question:"Is AlsoAsked free?",answer:"<strong>AlsoAsked offers free monthly credits</strong> for non-registered users, with no credit card required. Paid plans start at $12/mo (Basic, 100 credits) and go to $47/mo (Pro, 1,000 credits); the $23/mo Lite tier (300 credits) is listed as most popular. Annual billing saves 20%. <strong>All paid tiers \u2014 including Basic \u2014 include API access</strong>, so the API is not gated behind a premium plan. The free tier is genuinely useful for occasional PAA research. The paid tiers are priced for practitioners doing regular question harvesting. The $12 Basic plan is a legitimate entry point for SEOs who need more than the free credits allow but are not yet at pipeline scale."},{id:"q4-3",num:"4.3",question:"What are the 4 pillars of SEO?",answer:'Technical, on-page, off-page, and content \u2014 the <strong>traditional four pillars</strong> \u2014 are all affected by PAA strategy. But the content pillar increasingly depends on the technical pillar for data access: a content team that cannot harvest PAA programmatically is relying on manual research that caps out at dozens of keywords. The teams closing content at scale have connected their data layer directly to their publishing pipeline. PAA strategy now bridges two pillars simultaneously. The content question ("which questions should we answer?") is answered by the technical infrastructure ("what does the PAA API return for this seed keyword?"). That bridge is the competitive gap most content teams have not crossed.'},{id:"q4-4",num:"4.4",question:"Can a beginner do SEO?",answer:"A beginner can start PAA research immediately \u2014 <strong>free tier on AlsoAsked</strong> (no account required), Google Search Console for existing rankings, and manual SERP expansion for small keyword sets. The step-change to programmatic PAA requires basic API literacy, not advanced SEO expertise. <strong>Developers entering content teams are often better positioned</strong> for the API path than experienced SEOs who have never worked with structured data outputs. The entry barrier is not SEO knowledge \u2014 it is familiarity with REST APIs and JSON. A developer who has never done SEO can integrate the MCP Scraper API faster than an experienced SEO who has never touched an API endpoint."},{id:"q4-5",num:"4.5",question:"Can I do SEO by myself?",answer:"Solo SEOs can run effective PAA strategy \u2014 the <strong>manual workflow handles 1\u20135 target keywords well</strong>. The friction hits at roughly 50 keywords: at that volume, manually expanding PAA trees and logging questions becomes the rate-limiting step, not the content writing. The programmatic API path is the unlock at scale, not a requirement for getting started. The honest threshold: if your keyword list fits on one spreadsheet page and you update it quarterly, manual PAA research is sufficient. If your keyword list is dynamic, multi-locale, or feeds an automated content system, the API path pays for itself in the first week of research time saved."}],interactiveHtml:`<div class="si si-comparison"><span class="si-heading">Pick a tool to compare</span><div class="si-tabs" role="tablist"><button class="si-tab si-tab-active" role="tab" aria-selected="true" data-panel="sip4-p0">AlsoAsked</button><button class="si-tab" role="tab" aria-selected="false" data-panel="sip4-p1">Semrush</button><button class="si-tab" role="tab" aria-selected="false" data-panel="sip4-p2">MCP Scraper</button></div><div class="si-panels"><div class="si-panel" id="sip4-p0" role="tabpanel"><p class="si-panel-title">AlsoAsked \u2014 UI-based PAA question trees</p><ul class="si-panel-pros"><li>Visual question tree with PNG and CSV export</li><li>Free tier with no account required</li><li>Bulk upload up to 1,000 seeds, multi-region support</li></ul><ul class="si-panel-cons"><li>CSV export requires human review in the loop</li><li>API mirrors the single-query UI model \u2014 not designed for pipeline volume</li></ul><p class="si-panel-verdict">Best for UI-based research at up to 1,000 seeds per month. The right tool when a researcher is manually reviewing and selecting questions.</p></div><div class="si-panel" id="sip4-p1" role="tabpanel" hidden><p class="si-panel-title">Semrush \u2014 Full SEO suite with PAA tracking</p><ul class="si-panel-pros"><li>PAA alongside rankings, backlinks, and site audit in one platform</li><li>Historical SERP feature data for trend analysis</li></ul><ul class="si-panel-cons"><li>PAA is not its primary strength \u2014 depth limited vs. dedicated tools</li><li>Expensive if PAA is your only use case</li><li>No programmatic extraction of full PAA trees</li></ul><p class="si-panel-verdict">Right if you already use Semrush for keyword research and want PAA visibility added. Overkill for PAA-only workflows.</p></div><div class="si-panel" id="sip4-p2" role="tabpanel" hidden><p class="si-panel-title">MCP Scraper \u2014 Web dashboard + API + MCP server</p><ul class="si-panel-pros"><li>Dashboard at mcpscraper.dev: run PAA, SERP, Maps, YouTube, and Facebook Ads from one UI</li><li>Same data available via REST API and as MCP tools for Claude, Cursor, and Copilot</li><li>Full PAA tree per seed, results as cards or structured JSON/Markdown export</li></ul><ul class="si-panel-cons"><li>Covers seven surfaces \u2014 more than you need if PAA is your only use case</li><li>Credit-based billing: each surface costs credits, not a flat subscription per tool</li></ul><p class="si-panel-verdict">Right when you need PAA alongside SERP data, Maps intelligence, or competitive ad research \u2014 and when you want the same data accessible to both your team and your AI agents.</p></div></div><script>(function(){document.querySelectorAll('.si-comparison').forEach(function(comp){comp.querySelectorAll('.si-tab').forEach(function(tab){tab.addEventListener('click',function(){comp.querySelectorAll('.si-tab').forEach(function(t){t.classList.remove('si-tab-active');t.setAttribute('aria-selected','false');});comp.querySelectorAll('.si-panel').forEach(function(p){p.hidden=true;});tab.classList.add('si-tab-active');tab.setAttribute('aria-selected','true');document.getElementById(tab.dataset.panel).hidden=false;});});});})()</script></div>`},{id:"s5",num:"05",title:"Scaling PAA",titleItalic:"Extraction",deck:"The programmatic case \u2014 why the manual UI loop breaks at scale, and what a PAA data pipeline actually looks like.",callout:{eyebrow:"The scale threshold",heading:"PAA questions shift by location, device, language, and time. Scraping them once is not a strategy.",body:"Google mines search sessions continuously. PAA is a live signal, not a static dataset. <strong>Recurring programmatic harvesting on a schedule is the correct workflow for any live content operation \u2014 not a one-time manual pull followed by a spreadsheet filed away.</strong>"},cards:[{id:"q5-1",num:"5.1",question:"What are the 5 important concepts of SEO?",answer:"Applied to PAA at scale, five concepts that drive results: <strong>query intent mapping</strong> (which questions signal purchase-readiness), <strong>zero-volume question discovery</strong> (PAA surfaces questions that keyword tools miss entirely), <strong>PAA tree traversal</strong> (one seed keyword branches into hundreds of related questions), <strong>freshness signaling</strong> (content updated within 90 days appears 4.3\xD7 more frequently in PAA features), and <strong>structured data delivery</strong>. The last four require programmatic access. The freshness multiplier is the most underused lever in PAA strategy \u2014 most teams optimize the answer once and move on, missing the ongoing recency advantage that recurring updates deliver."},{id:"q5-2",num:"5.2",question:"What are the 3 pillars of SEO?",answer:"At programmatic scale, the three operational pillars become <strong>data acquisition, content production, and distribution</strong> \u2014 not the traditional crawlability, content, and authority. PAA harvesting via API sits at the data-acquisition layer, upstream of every content and publishing decision. MCP Scraper operates at that layer: it does not write content or build links, it supplies the question data that makes content decisions defensible rather than intuitive. The scope boundary is a trust signal: a tool that claims to do everything does nothing well. The data acquisition layer is the one most content teams have not built yet \u2014 and it is the layer that compounds."},{id:"q5-3",num:"5.3",question:"What are the 3 C's of SEO?",answer:'Content, code, and credibility \u2014 but in a programmatic PAA workflow, <strong>"content" starts upstream with machine-readable question data</strong>, not a brainstorming session. The gap between "which questions should we answer" and "draft created" collapses when PAA API output feeds directly into a content brief template or LLM prompt. The manual research phase that typically takes days becomes a <strong>sub-second API call</strong>. The "code" pillar is what enables this \u2014 a REST endpoint that accepts a seed keyword and returns a structured PAA tree is not a luxury for large teams; it is the unlock that makes content operations at any scale less dependent on individual research time.'},{id:"q5-4",num:"5.4",question:"What is the Google 20% rule?",answer:"Google's 20% rule \u2014 the practice of giving engineers discretionary time for side projects \u2014 produced features including Gmail and Google Maps. <strong>PAA itself emerged from the same underlying logic</strong>: Google continuously mines search session data to predict follow-up intent. That mining is live and ongoing, which is why PAA questions shift by location, device, language, and time \u2014 and why scraping them once and filing the results is not a strategy. <em>Recurring harvesting on a schedule is.</em> The operational implication: build a workflow that pulls PAA data for your priority keyword set on a monthly or weekly cadence, and treat the outputs as a live editorial signal, not a one-time research deliverable."}],interactiveHtml:`<div class="si si-checklist"><span class="si-heading">PAA pipeline setup</span><div class="si-check-progress-row"><div class="si-check-bar-wrap"><div class="si-check-bar" id="sip5-bar" style="width:0%"></div></div><span class="si-check-count" id="sip5-count">0 / 5 done</span></div><ul class="si-check-list"><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Define your seed keyword list \u2014 start with your top 50 pages by traffic</label></li><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Choose your extraction method: manual for ≤10 keywords, API for 50+</label></li><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Classify PAA output by intent type \u2014 prioritize purchase-intent questions first</label></li><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Set a freshness schedule \u2014 monthly for competitive topics, quarterly for stable ones</label></li><li class="si-check-item"><label><input type="checkbox" class="sip5-chk"> Connect PAA output to your content brief template or LLM prompt</label></li></ul><script>(function(){var checks=document.querySelectorAll('.sip5-chk'),bar=document.getElementById('sip5-bar'),count=document.getElementById('sip5-count'),total=checks.length;function upd(){var done=document.querySelectorAll('.sip5-chk:checked').length;bar.style.width=(done/total*100)+'%';count.textContent=done+' / '+total+' done';}checks.forEach(function(c){c.addEventListener('change',upd);});})()</script></div>`},{id:"s6",num:"06",title:"Practical PAA",titleItalic:"Tactics",deck:"Quick wins, freshness mechanics, anti-bot realities, and why the programmatic path is where SEO income separates.",cards:[{id:"q6-1",num:"6.1",question:"What are basic SEO skills?",answer:"The baseline PAA skill set is: <strong>reading query intent accurately</strong>, <strong>writing concise Q&A answers</strong> (40\u201350 words is the PAA sweet spot), <strong>implementing FAQ and HowTo schema markup</strong>, and <strong>maintaining content freshness</strong> \u2014 pages updated within 90 days appear 4.3\xD7 more frequently in PAA features than stale content. The advanced skill is API integration for teams transitioning to programmatic workflows. The basics above are achievable without any tooling beyond Google Search Console, and they compound: intent-matched 40-word answers with correct schema and a quarterly update cadence will outperform longer, less-structured content on PAA placements across nearly every category."},{id:"q6-2",num:"6.2",question:"Is SEO a high income skill?",answer:'PAA-driven SEO is splitting into two earning brackets. <strong>Practitioners who optimize content manually</strong> \u2014 identifying questions, formatting answers, checking rankings \u2014 are doing work that AI tools increasingly replicate, which compresses rates. <strong>Practitioners who build PAA data pipelines</strong> and integrate them into content systems are doing technical-strategic work that remains rare. The programmatic path is where income separates, not because it is harder to learn, but because few people have crossed the data-engineering threshold yet. The "data-engineering threshold" is lower than it sounds: REST API literacy, basic JSON handling, and familiarity with content pipeline tooling. The gap is not a skills gap \u2014 it is an awareness gap.'},{id:"q6-3",num:"6.3",question:"Why is Google checking if I'm human?",answer:"CAPTCHA and anti-bot verification appear when collecting PAA data without proper infrastructure. <strong>Headless browsers sending high-frequency requests</strong> without realistic session behavior, residential proxy coverage, or rate limiting trigger Google's bot-detection systems. MCP Scraper's API handles the anti-bot layer on the infrastructure side \u2014 the caller passes a keyword and receives structured PAA output; the detection friction never reaches the application layer. This is one of the most underestimated friction points in programmatic PAA collection: teams that build their own scrapers spend disproportionate engineering time on detection evasion rather than on the content strategy that the data enables."},{id:"q6-4",num:"6.4",question:"Am I being monitored by Google?",answer:"Google monitors search behavior continuously to update PAA questions \u2014 <strong>session patterns, query sequences, click data, and location signals</strong> all feed the PAA algorithm in real time. This is why the same keyword returns different PAA questions depending on location, language, device, and time of day. It also explains why a static PAA dataset goes stale: the questions Google surfaces this week may differ meaningfully from last month's harvest. <em>Recurring collection is the correct workflow for any live content operation.</em> The monitoring is a feature, not a surveillance concern \u2014 it means PAA is a continuously refreshed intent signal, and the teams harvesting it regularly have a permanently updated editorial dataset that teams relying on one-time research do not."}],interactiveHtml:`<div class="si si-codegen"><span class="si-heading">Generate your PAA API request</span><div class="si-codegen-inputs"><label class="si-label">Keyword<input class="si-text-input" id="sip6-kw" value="people also ask SEO" placeholder="e.g. best CRM software"></label><label class="si-label">Max questions<input class="si-text-input" id="sip6-mq" value="50" placeholder="50" type="number" min="1" max="200"></label></div><div class="si-codegen-output-wrap"><pre class="si-codegen-pre"><code id="sip6-out"></code></pre><button class="si-copy-btn" id="sip6-copy">Copy curl</button></div><script>(function(){function getKw(){return document.getElementById('sip6-kw').value||'your keyword';}function getMq(){return+(document.getElementById('sip6-mq').value||50);}function generate(){return JSON.stringify({query:getKw(),maxQuestions:getMq()},null,2);}function upd(){document.getElementById('sip6-out').textContent=generate();}document.querySelectorAll('#sip6-kw,#sip6-mq').forEach(function(i){i.addEventListener('input',upd);});document.getElementById('sip6-copy').addEventListener('click',function(){var body=JSON.stringify({query:getKw(),maxQuestions:getMq()});var cmd="curl -X POST https://mcpscraper.dev/harvest/sync -H 'Content-Type: application/json' -H 'x-api-key: YOUR_API_KEY' -d '"+body+"'";navigator.clipboard.writeText(cmd).then(function(){var btn=document.getElementById('sip6-copy');btn.textContent='Copied!';setTimeout(function(){btn.textContent='Copy curl';},1500);});});upd();})()</script></div>`}]},{slug:"what-is-an-mcp",title:"What Is an MCP? The Honest Developer's Guide",titleDisplay:"What Is",titleDisplayItalic:"an MCP?",description:"MCP explained for developers who need to decide \u2014 not just understand. Which platforms adopted it, when to skip it, and how it compares to REST, Zapier, and Copilot.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["MCP","model context protocol","AI tools","developer guide","Claude"],category:"AI Development",badge:"30 questions answered",fieldGuideLabel:"Field guide",deck:"Every article about what is an MCP tells you it's a universal plug-and-play connector for AI. This one tells you which platforms actually adopted it, when to skip it entirely, and why the USB-C analogy hides the decision your architecture actually requires.",readTimeMinutes:11,stats:[{value:"6",label:"sections"},{value:"30",label:"questions answered"},{value:"97M",label:"monthly MCP SDK downloads"},{value:"0",label:"fluff"}],ctaHeading:"Build for the agent era.",ctaHeadingItalic:"MCP Scraper is an MCP-native tool \u2014 extract PAA data the way AI agents actually work.",ctaBody:"MCP Scraper gives AI agents the web data they need \u2014 PAA questions, SERP results, and page content via REST or MCP. Built for the agent-native stack.",sections:[{id:"s1",num:"01",title:"MCP",titleItalic:"Defined",deck:"What MCP actually is, what it does, and the architectural distinction every explainer skips.",callout:{eyebrow:"Quick take",heading:"MCP is not USB-C.",body:"USB-C is a hardware connector. MCP is an architectural decision \u2014 a protocol that determines whether your AI agent can discover and call tools at runtime or must be hardcoded by a developer. <strong>The analogy hides the trade-off. This section names it.</strong>"},cards:[{id:"q1-1",num:"1.1",question:"What is an MCP in AI?",answer:"MCP \u2014 <strong>Model Context Protocol</strong> \u2014 is an open standard launched by Anthropic on November 25, 2024, that gives AI agents a single, standardized interface to connect with external tools, data sources, and workflows. Before MCP, every AI model and every external tool required a custom integration written by a developer. MCP eliminates that custom work by defining a shared communication contract that any agent and any tool can implement once and then use interchangeably. <em>An agent calling GitHub, Postgres, and Slack</em> no longer needs three bespoke connectors \u2014 it needs one MCP client. The practical implication most explainers miss: MCP is not primarily a developer convenience \u2014 it's what makes AI agents autonomous enough to run without a human in the loop on every tool call."},{id:"q1-2",num:"1.2",question:"What does the MCP do?",answer:"MCP enables AI agents to <strong>discover, invoke, and receive results from external tools at runtime</strong> \u2014 without a developer pre-configuring every possible tool call. The mechanism is a standard called <code>tools/list</code>: an agent sends this request to any MCP server and gets back a live manifest of every capability that server exposes, including names, descriptions, and input schemas. The agent then selects the right tool, calls it, and receives structured output \u2014 all without leaving its session. <em>Connecting an agent to a CRM, a code repo, and a web scraper</em> used to mean three separate integrations; MCP makes them all callable through the same protocol. The deeper point: MCP doesn't just save integration work \u2014 it makes tool use something the agent decides at runtime rather than something a developer hardcodes at build time."},{id:"q1-3",num:"1.3",question:"What is MCP in simple terms?",answer:`MCP is a <strong>shared language that any AI agent and any tool can both speak</strong>. Once a tool publishes an MCP server, every MCP-compatible agent \u2014 Claude, ChatGPT, Cursor, GitHub Copilot \u2014 can call it without a custom connector. The analogy that actually holds: imagine if every REST API automatically understood every other API's auth, parameter format, and response schema without any glue code. That's what MCP does for AI agents. <em>A Postgres database, a Slack workspace, and a web data service</em> all become callable through the same interface once they expose an MCP server. The distinction that matters in practice: "simple" doesn't mean effortless \u2014 the tool still has to implement the MCP server; MCP just ensures that implementation only has to happen once.`},{id:"q1-4",num:"1.4",question:"Is MCP a tool or framework?",answer:`MCP is neither \u2014 it's a <strong>protocol</strong>, which means a specification for how two systems communicate, not software you install or a framework you build on top of. Tools implement MCP (a Postgres connector that speaks MCP is a tool). Frameworks integrate MCP clients (LangChain, AutoGen). MCP itself is the rulebook those implementations follow \u2014 specifically, a JSON-RPC 2.0 message format plus a defined session lifecycle. IBM's documentation makes this distinction explicitly: MCP is not an agent framework. The practical consequence: asking "which MCP should I choose?" is like asking "which HTTP should I use?" \u2014 the protocol is the same; your choice is which server or client library implements it.`},{id:"q1-5",num:"1.5",question:"What problems does MCP solve?",answer:"MCP solves the <strong>N\xD7M integration problem</strong>: without it, connecting N AI models to M external tools produces N\xD7M custom integrations, each built and maintained separately. MCP collapses this to M+N \u2014 implement the protocol once on each side and every combination works. The secondary problems it solves flow from the same root: inconsistent auth patterns across integrations, brittle hardcoded tool calls that break when APIs change, and the inability for AI agents to discover new tools at runtime. <em>An enterprise with 10 AI models and 50 internal tools</em> faces 500 custom integrations without MCP and 60 protocol implementations with it. What no explainer says plainly: MCP doesn't reduce the number of tools you build \u2014 it reduces the number of connectors that break when either side changes."},{id:"q1-6",num:"1.6",question:"What is Anthropic?",answer:"Anthropic is the AI safety company that created Claude and launched MCP on November 25, 2024. <strong>In December 2025, Anthropic donated MCP governance to the Agentic AI Foundation under the Linux Foundation</strong> \u2014 a move that made MCP formally vendor-neutral and accelerated adoption by removing the perception that the protocol was an Anthropic-controlled standard. Anthropic was founded in 2021 by former OpenAI researchers, including Dario Amodei and Daniela Amodei. Its two primary contributions to AI infrastructure are Claude (a family of large language models) and MCP (the protocol this post covers). The governance transfer is the detail most explainers skip: Anthropic no longer controls MCP \u2014 the Linux Foundation body does, which is why OpenAI, Microsoft, and Google were willing to adopt it."}]},{id:"s2",num:"02",title:"MCP vs",titleItalic:"Everything",deck:"MCP vs REST, HTTP, Zapier, Copilot, and LLMs \u2014 the comparisons that actually determine whether MCP belongs in your stack.",callout:{eyebrow:"Key distinction",heading:"REST serves developers. MCP serves AI agents.",body:"That one sentence determines whether MCP belongs in your architecture. If the caller is a developer writing code, use REST. If the caller is an agent making runtime decisions, MCP earns its overhead. <strong>The mistake is treating them as competing choices rather than different interface layers.</strong>"},cards:[{id:"q2-1",num:"2.1",question:"What is an MCP vs API?",answer:"REST APIs serve developers; <strong>MCP serves AI agents</strong> \u2014 and that distinction determines which belongs in your system. A REST API is a stateless HTTP endpoint with static documentation a human reads and then hardcodes calls against. MCP is JSON-RPC 2.0 over a persistent session, with a live tool manifest the agent reads at runtime and uses to decide what to call. The critical difference: with REST, a developer writes the integration once and it breaks when the API changes; with MCP, the agent re-discovers the tool manifest on every session and adapts. MCP doesn't replace REST \u2014 MCP servers use REST internally. What changes is the interface layer above: REST is for humans integrating systems; MCP is for agents choosing tools."},{id:"q2-2",num:"2.2",question:"Why MCP instead of REST API?",answer:"Use MCP instead of REST when the consumer of the API is an AI agent making runtime decisions \u2014 not a developer writing hardcoded calls. <strong>MCP's <code>tools/list</code> endpoint lets an agent discover what a server can do without any human-written glue code</strong>. With REST, someone has to read the OpenAPI spec and write the integration; with MCP, the agent reads the live capability manifest and writes the call itself. The answer changes if you control both ends: if you own the API and the code calling it, REST is simpler and faster. MCP earns its overhead when the caller is an autonomous agent that needs to self-direct across a changing tool landscape \u2014 the moment you want the agent to decide which tool to use, not just execute the one you told it to."},{id:"q2-3",num:"2.3",question:"Why use MCP instead of HTTP?",answer:"Raw HTTP has no standard for <strong>tool discovery, session state, or AI-native authentication</strong> \u2014 MCP adds all three on top of HTTP. An agent calling raw HTTP endpoints has to know the URL, method, parameters, and auth scheme in advance, hardcoded. An MCP server exposes a <code>tools/list</code> manifest so the agent discovers capabilities dynamically, maintains state across a session, and uses a standardized OAuth 2.1 flow for auth rather than each API's bespoke scheme. The practical failure mode of raw HTTP at scale: when the tool landscape changes \u2014 a new endpoint, a deprecated parameter \u2014 every hardcoded HTTP call breaks silently; MCP's session-level capability manifest surfaces changes the agent can adapt to. Building on raw HTTP for agents is not wrong at toy scale \u2014 it fails at production scale when the number of tools grows past what any developer can maintain manually."},{id:"q2-4",num:"2.4",question:"Is MCP like Zapier?",answer:"MCP and Zapier solve adjacent problems with different audiences. <strong>Zapier automates workflows between apps for non-technical users; MCP is a protocol for AI agents to call tools programmatically</strong>. Zapier's model is: a human configures a trigger-and-action workflow once; Zapier runs it. MCP's model is: an AI agent discovers available tools at runtime and decides which ones to call. The two are not mutually exclusive \u2014 Zapier built MCP support on top of its library of 9,000+ apps and 30,000+ actions, meaning an AI agent with an MCP client can now reach every Zapier-connected app without the human workflow-configuration step. The practitioner nuance: Zapier MCP is useful when you want AI agent access to Zapier's app coverage without building individual MCP servers for each app."},{id:"q2-5",num:"2.5",question:"What is the difference between MCP and Copilot?",answer:`GitHub Copilot is an AI coding assistant that now operates as an <strong>MCP client</strong> \u2014 MCP is the protocol Copilot uses to reach external tools, not a competitor to Copilot. The relationship: Copilot is software running in VS Code or a browser; when Copilot needs to call an external tool (a Jira ticket, a GitHub repo, a web search), it does so through MCP. Before MCP, Copilot's tool integrations were custom connectors maintained by Microsoft. With MCP, any tool that publishes an MCP server becomes callable by Copilot without Microsoft writing a dedicated integration. The distinction developers miss: evaluating "MCP vs. Copilot" is a category error \u2014 Copilot is an adopter of MCP, not an alternative to it.`},{id:"q2-6",num:"2.6",question:"What is the difference between MCP and LLM?",answer:"An LLM is the <strong>reasoning engine</strong> \u2014 the model that reads input, generates text, and makes decisions. MCP is the protocol that gives that engine hands. Without MCP (or a comparable integration layer), an LLM can only work with what's in its context window \u2014 text and pre-loaded data. With MCP, the LLM can call tools at runtime: retrieve a live database record, execute a search query, write a file, trigger a workflow. <em>Claude 3.5 Sonnet reasoning about a customer support ticket</em> is an LLM at work; Claude calling a CRM to retrieve the customer's history mid-conversation is MCP at work. The frame that matters: LLMs decide; MCP acts. A sophisticated AI agent needs both."},{id:"q2-7",num:"2.7",question:"Is MCP just a JSON?",answer:`MCP uses <strong>JSON-RPC 2.0</strong> as its message format, but calling it "just JSON" misses the protocol. JSON-RPC 2.0 defines the envelope \u2014 request IDs, method names, parameters, error codes. MCP adds on top of that a defined session lifecycle: initialization handshake \u2192 capability negotiation \u2192 tool discovery \u2192 tool calls \u2192 termination. JSON alone specifies none of that structure. The distinction matters when debugging: a malformed MCP session isn't a JSON syntax error \u2014 it's a lifecycle state error, which requires understanding the protocol's state machine, not just validating JSON. The practitioner test: if your MCP server returns valid JSON but ignores the initialization handshake, every MCP client will reject it \u2014 not because the JSON is wrong, but because the protocol contract is broken.`}]},{id:"s3",num:"03",title:"MCP Adoption \u2014 Who",titleItalic:"Actually Uses It",deck:"Which platforms adopted MCP, which haven't, and the one misconception that sends most developers to the wrong conclusion.",callout:{eyebrow:"Highest-value misconception",heading:"MCP is not only for Claude.",body:"MCP has 300+ clients as of March 2026 \u2014 including ChatGPT, GitHub Copilot, VS Code, Cursor, and Google Gemini. Anthropic created it. The Linux Foundation now governs it. <strong>Write one MCP server. Every major AI platform can call it.</strong>"},cards:[{id:"q3-1",num:"3.1",question:"Is MCP only for Claude?",answer:"No \u2014 and this is the highest-value misconception in the MCP ecosystem. <strong>MCP has 300+ clients as of March 2026</strong>, including ChatGPT, GitHub Copilot, VS Code, Cursor, Windsurf, AWS Bedrock, Google Gemini, and JetBrains IDEs. Anthropic created MCP but donated its governance to the Agentic AI Foundation under the Linux Foundation in December 2025, making it a vendor-neutral standard no single company controls. OpenAI formally adopted MCP in March 2025 \u2014 four months after Anthropic launched it. The practical implication for developers: an MCP server you build today is reachable by Claude, ChatGPT, and GitHub Copilot without any modification \u2014 write the server once, serve every major agent platform."},{id:"q3-2",num:"3.2",question:"Does ChatGPT use MCP?",answer:"Yes \u2014 <strong>OpenAI formally adopted MCP in March 2025</strong> and added MCP support to ChatGPT apps in September 2025. OpenAI's adoption removed the last credible argument that MCP was a Claude-exclusive or Anthropic-controlled standard. When OpenAI adopted MCP, the protocol had an estimated 22 million monthly SDK downloads; by March 2026 that figure reached 97 million. ChatGPT is now one of more than 300 MCP clients \u2014 meaning any MCP server you build is natively callable from ChatGPT without any OpenAI-specific integration work. The sequence developers should know: Anthropic launched \u2192 OpenAI adopted \u2192 Microsoft integrated \u2192 Google followed \u2014 MCP's cross-vendor adoption happened in under 18 months."},{id:"q3-3",num:"3.3",question:"Does Microsoft use MCP?",answer:"Yes \u2014 Microsoft integrated MCP into <strong>GitHub Copilot, Visual Studio Code, and Copilot Studio</strong>. The GitHub Copilot extension in VS Code is among the most widely used MCP clients in the developer tooling ecosystem. Microsoft's adoption matters structurally: it brought MCP into enterprise environments at scale, since VS Code and GitHub Copilot are standard tooling in most engineering organizations. The consequence for MCP server builders: publishing an MCP server means your tool is callable from VS Code's AI features \u2014 the IDE that runs on more developer machines than any other. Microsoft's integration also means MCP is no longer a decision individual developers make; it's a platform decision that enterprise engineering teams inherit from their tooling."},{id:"q3-4",num:"3.4",question:"Is Apple using MCP?",answer:"Apple has not made a public MCP announcement as of May 2026. <strong>The honest answer is: unknown, but structurally likely.</strong> MCP clients that run on Apple platforms \u2014 Claude Desktop, VS Code, Cursor \u2014 are in wide use on macOS. If Apple builds agentic AI features into iOS or macOS, adopting MCP would give it instant interoperability with every existing MCP server ecosystem rather than requiring Apple to build a proprietary tool integration standard from scratch. The precedent: every other major AI platform that initially appeared absent from MCP (OpenAI, Microsoft, Google) has since formally adopted it. The practitioner read: Apple's silence is not a rejection \u2014 it's the gap between enterprise announcement cycles and protocol adoption reality."},{id:"q3-5",num:"3.5",question:"Is Zapier MCP free?",answer:`Yes \u2014 <strong>Zapier MCP is included on all Zapier plans, including the Free tier</strong>, at no additional cost. There is no separate product SKU. The only cost is task consumption: each MCP tool call uses 2 tasks from your existing Zapier task quota. A developer on Zapier's free plan can connect an AI agent to Zapier's 9,000+ app library and 30,000+ actions today, using their existing task allocation. The nuance that changes the math: "free" means no incremental charge, not zero cost \u2014 if an agent makes 500 MCP tool calls in a month, that consumes 1,000 tasks from your quota. High-volume agent workflows will exhaust free-tier quotas quickly and require a paid plan.`}]},{id:"s4",num:"04",title:"When MCP",titleItalic:"Breaks Down",deck:"When not to use MCP, why production projects stall, and the honest comparison to adjacent tools.",callout:{eyebrow:"Practitioner test",heading:"MCP has no native auth.",body:"The spec recommends OAuth 2.1 with PKCE. Your MCP server only has it if you built it. <strong>Every production MCP project that stalled did so at auth, not at the protocol itself.</strong>"},cards:[{id:"q4-1",num:"4.1",question:"When not to use MCP?",answer:"Skip MCP when your system is <strong>developer-to-API rather than agent-to-tool</strong>. If a human developer is writing the integration code, REST is simpler and adds no session-management overhead. Skip MCP when you control both ends of the integration and don't need runtime discovery \u2014 if you own the calling code and the tool, you already know what the tool does; the <code>tools/list</code> handshake is unnecessary overhead. Skip MCP when latency is the primary constraint \u2014 persistent sessions add round-trip initialization costs that stateless REST calls don't. Skip MCP when your AI use case is inference-only: if the model generates text without calling any external system, MCP adds complexity with zero benefit. The practitioner test: if a human could write the integration code once and it would never need to change, use REST."},{id:"q4-2",num:"4.2",question:"Why are people moving away from MCP?",answer:`The most common friction points are auth complexity, debugging difficulty, and server quality variance. <strong>MCP has no native authentication</strong> \u2014 developers must implement OAuth 2.1 with PKCE themselves, and the gap between "runs locally" and "ships securely to production" is where most MCP projects stall. Debugging is harder than REST because persistent sessions have lifecycle state: a broken MCP connection isn't a failed HTTP request \u2014 it's a state machine that failed at initialization, capability negotiation, or mid-session, and the error may not surface clearly. The 10,000+ public MCP servers have wildly inconsistent quality \u2014 some are maintained production services; many are experimental projects with no uptime guarantees. "Moving away" overstates it: developers who understand MCP's limits ship successfully; the ones who expected plug-and-play get surprised by the operational requirements.`},{id:"q4-3",num:"4.3",question:"Is A2A dead?",answer:"A2A \u2014 Google's Agent-to-Agent protocol \u2014 is not dead, but it has not achieved the ecosystem density MCP has. <strong>MCP and A2A address different layers</strong>: MCP handles tool access (an agent calling an external capability); A2A handles agent coordination (one agent delegating a task to another agent). They are complementary, not competing. The adoption gap is real: MCP has 97 million monthly SDK downloads and 300+ clients as of March 2026; A2A's ecosystem is smaller by every public measure. The honest framing: A2A is the correct protocol for multi-agent orchestration problems; MCP is the correct protocol for tool-calling problems. A developer who needs both can use both \u2014 and the most sophisticated agent architectures will."},{id:"q4-4",num:"4.4",question:"Will MCP replace API?",answer:`No \u2014 <strong>MCP wraps APIs rather than replacing them</strong>. MCP servers use REST APIs internally; they expose MCP above and call REST below. The correct model: REST remains the implementation layer; MCP becomes the interface layer that AI agents interact with. This isn't a philosophical position \u2014 it's how production MCP servers are built. <em>An MCP server for Stripe</em> doesn't replace Stripe's REST API; it wraps it, adding the <code>tools/list</code> manifest and session management that AI agents expect. The question developers should ask instead: not "will MCP replace REST?" but "at which layer does MCP belong in my stack?" \u2014 the answer is always above your existing APIs, never instead of them.`},{id:"q4-5",num:"4.5",question:"Why are people against Copilot?",answer:"Copilot critics object to Microsoft's <strong>pricing model, data training practices, and the perception that Copilot is a productivity layer rather than a reasoning upgrade</strong>. Common complaints include the cost per seat relative to perceived productivity gains, concerns about code written in Copilot being used to train future models, and the view that Copilot autocompletes rather than reasons. These criticisms are entirely separate from MCP \u2014 MCP is the protocol Copilot uses to reach external tools; it doesn't change Copilot's pricing, training data policies, or reasoning depth. The relevant clarification for developers evaluating both: being against Copilot doesn't mean avoiding MCP \u2014 every other major AI platform (Claude, ChatGPT, Cursor) also uses MCP and has none of Copilot's specific controversies."}]},{id:"s5",num:"05",title:"MCP Architecture",titleItalic:"& Security",deck:"Transport layers, encryption, auth, and the gap between what MCP promises and what your server actually ships.",callout:{eyebrow:"Security gap",heading:"MCP won't refuse an HTTP connection.",body:"The spec recommends HTTPS. Enforcement is the developer's job. Most MCP tutorials run over HTTP for simplicity \u2014 developers copy that configuration to production and ship an insecure server without any warning from the protocol. <strong>TLS is your responsibility, not MCP's.</strong>"},cards:[{id:"q5-1",num:"5.1",question:"Is MCP built on HTTP?",answer:"MCP supports two transport layers: <strong>stdio for local in-process communication and HTTP with Server-Sent Events (SSE) for remote connections</strong>. Local MCP servers \u2014 the kind that run on a developer's machine alongside Claude Desktop \u2014 use stdio, which is faster and simpler because the agent and server are in the same process space. Remote production MCP servers use HTTP/SSE, which enables cross-network communication but requires TLS since MCP has no native encryption layer. Most production MCP servers use the HTTP transport \u2014 it's what makes an MCP server callable from any agent on any machine. The choice isn't either-or: a single MCP server implementation can support both transports, but most developers building for production start with HTTP/SSE."},{id:"q5-2",num:"5.2",question:"Does MCP use HTTP or HTTPS?",answer:"MCP supports both, but <strong>the spec explicitly recommends HTTPS for any remote server</strong> \u2014 and running an MCP server over plain HTTP in production is a security vulnerability, not a configuration choice. MCP does not natively enforce encryption; TLS must be configured by the developer. The risk is concrete: MCP sessions carry tool call parameters and responses over a persistent connection \u2014 plain HTTP exposes that entire session to interception. Local development over localhost HTTP is acceptable (the connection doesn't leave the machine); any server exposed over a network must use HTTPS. The gap most teams hit: MCP tutorials run over HTTP for simplicity; developers copy that configuration to production and ship an insecure server without realizing the protocol never warned them."},{id:"q5-3",num:"5.3",question:"Does MCP use OAuth?",answer:`MCP standardizes <strong>OAuth 2.1 with PKCE</strong> for authentication in remote server connections \u2014 but this is not built into the base protocol. OAuth support was added to the MCP spec after initial launch, when real-world adoption revealed auth as the most common production gap. Implementing it requires developers to configure an OAuth 2.1 server, handle PKCE flows, and manage token refresh \u2014 none of which MCP handles automatically. Local MCP servers running over stdio typically require no auth because they run in a trusted local environment. Remote servers require auth, and OAuth 2.1 with PKCE is the spec-recommended approach. The practitioner reality: "MCP supports OAuth" means MCP defines how OAuth should work in its context \u2014 it doesn't mean your MCP server has OAuth until you build it.`},{id:"q5-4",num:"5.4",question:"Is an MCP server just an API?",answer:"An MCP server looks like an API from the outside but differs in three structural ways. First, it exposes a <strong>standard capabilities manifest via <code>tools/list</code></strong> \u2014 a traditional API has static docs; an MCP server has a live, machine-readable manifest the agent queries at runtime. Second, it maintains session state across multiple calls within a connection \u2014 REST APIs are stateless by design; MCP sessions are stateful. Third, it uses JSON-RPC 2.0 rather than REST conventions \u2014 request/response patterns, error codes, and method naming all follow JSON-RPC semantics, not HTTP verb conventions. <em>Calling an MCP server</em> and <em>calling a REST API</em> look superficially similar from a network perspective; they're architecturally different contracts that break in different ways when misused."}]},{id:"s6",num:"06",title:"MCP Ecosystem",titleItalic:"& What's Next",deck:"Power Automate, Google's investment, and why MCP is a layer above APIs rather than a replacement for them.",cards:[{id:"q6-1",num:"6.1",question:"What has replaced Power Automate?",answer:"Nothing has fully replaced Power Automate \u2014 it remains Microsoft's enterprise workflow product and continues to serve the structured, if-then automation use cases it was built for. <strong>What MCP and AI agents are taking from Power Automate is the dynamic, decision-driven tier of automation</strong> \u2014 tasks that require reasoning, not just routing. Power Automate routes data between systems based on rules a human writes; an MCP-connected agent can decide which tools to call, handle edge cases without pre-written rules, and adapt to inputs that a static workflow would reject. <em>Routing a support ticket to the right queue</em> is Power Automate's domain; <em>reading the ticket, checking the customer's history, drafting a response, and escalating if needed</em> is where MCP-connected agents outperform static workflow tools. The transition isn't replacement \u2014 it's a shift in which automation problems belong in which category."},{id:"q6-2",num:"6.2",question:"Does Google own 14% of Anthropic?",answer:"Google has invested significantly in Anthropic across multiple rounds, but <strong>the exact ownership percentage is not publicly disclosed</strong>. Reports indicate Google invested $300 million in a 2023 funding round and participated in subsequent rounds. The precise ownership stake \u2014 including whether it is or was 14% \u2014 has not been confirmed in any public filing. What is confirmed: Google Cloud and Anthropic have a partnership that includes MCP integration into Google's Gemini models and Google Cloud services. The relevant fact for MCP evaluation: Google's investment in Anthropic did not prevent Google from independently adopting MCP \u2014 both Google Gemini and Google Cloud services are listed among MCP clients, and adoption was driven by the protocol's open governance under the Linux Foundation, not by equity relationships."},{id:"q6-3",num:"6.3",question:"Is MCP basically an API?",answer:"MCP is best understood as a <strong>layer above APIs, not a replacement for them</strong>. Saying MCP is basically an API is like saying HTTP is basically a phone call \u2014 technically there's a connection, but the architectural purpose is different. Traditional APIs serve developers who know what they want to call and write the call in advance. MCP serves AI agents that discover what's available at runtime and decide what to call dynamically. The difference shows up at scale: an API breaks when you add a new endpoint that no existing code knows to call; an MCP server's <code>tools/list</code> response automatically surfaces the new capability to every connected agent on the next session. MCP Scraper is an example of what this makes possible \u2014 a web data tool built not for developers to integrate manually, but for AI agents to discover and call directly, the way MCP was designed to work."}]}]},{slug:"when-not-to-use-an-mcp",title:"When Not to Use an MCP: The Architectural Decision Guide",titleDisplay:"When Not to Use",titleDisplayItalic:"an MCP.",description:"Not complexity\u2014architecture. The three signals that disqualify MCP, the A2A and function-calling alternatives, and the protocol durability question for 2026.",publishedAt:"2026-05-24",author:"MCP Scraper",authorInitials:"M",tags:["MCP","AI","protocol","architecture","agents"],category:"MCP Protocol",badge:"36 questions answered",fieldGuideLabel:"Field guide",deck:"The complexity calculus is the wrong frame. In 2026, the real question is whether MCP will still be the dominant protocol when your integration ships \u2014 and that requires reading the ecosystem, not the docs.",readTimeMinutes:15,stats:[{value:"5",label:"sections"},{value:"36",label:"questions"},{value:"3",label:"skip signals"},{value:"0",label:"fluff"}],ctaHeading:"Read the ecosystem with",ctaHeadingItalic:"live PAA data.",ctaBody:"MCP Scraper harvests People Also Ask questions at scale \u2014 so your MCP decision is based on what practitioners are actually searching right now, not what the docs say they should ask.",sections:[{id:"s1",num:"01",title:"What MCP",titleItalic:"Actually Is.",deck:"Every article about MCP tells you it's a protocol that lets AI models call tools. That definition is technically accurate and practically useless \u2014 it skips the problem MCP was invented to solve, which is the only thing that tells you whether you need it at all.",cards:[{id:"q1-1",num:"1.1",question:"What is an MCP in AI?",answer:`MCP (Model Context Protocol) is an open standard that solves the N\xD7M integration problem \u2014 the combinatorial explosion that happens when M AI models each need custom connectors to N external tools. Before MCP existed, connecting three AI models to ten tools required up to thirty custom integrations; MCP standardizes the interface so any compliant model can call any compliant server without custom work. <strong>The official description is "like a USB-C port for AI applications" \u2014 a single standardized connector that replaces a sprawl of proprietary cables.</strong> The current stable specification (2025-11-25) is built on JSON-RPC 2.0, defines three roles (Hosts, Clients, Servers), and is supported across Claude, ChatGPT, Visual Studio Code, and Cursor. The question to ask before adopting MCP is not "what is it" but "do I have an N\xD7M problem worth solving at the protocol layer" \u2014 if you have one model and one tool, you don't.`,source:"https://modelcontextprotocol.io"},{id:"q1-2",num:"1.2",question:"What is MCP in simple terms?",answer:`MCP is the standardized language that lets an AI model ask an external tool to do something \u2014 and get a structured answer back \u2014 without either side needing to know how the other was built. <strong>Think of it as a universal remote control for AI agents: instead of each AI building its own custom remote for each device it wants to control, MCP gives every device a standard input jack and every remote a standard output plug.</strong> In practice, an MCP server exposes a list of "tools" (callable functions) with descriptions the AI model reads. The model decides which tool to call and passes structured arguments; the server executes the operation and returns a result. What makes this worth a protocol is that the model doesn't need to be rewritten when the tool changes, and the tool doesn't need to be rewritten when a new AI model is added. A practitioner who has shipped MCP integrations will tell you the simplification is real at scale \u2014 and essentially invisible for single-tool, single-model use cases.`,source:"https://modelcontextprotocol.io"},{id:"q1-3",num:"1.3",question:"What problems does MCP solve?",answer:'MCP solves two structural problems that emerge when AI systems grow beyond a single model connected to a single tool: the N\xD7M connector problem and the capability-discovery problem. <strong>The N\xD7M connector problem is architectural: without a standard, every new AI model requires custom integration code for every tool it needs to call \u2014 the cost scales multiplicatively, not additively.</strong> The capability-discovery problem is subtler: before MCP, a model had to know at design time exactly what an external tool could do; with MCP, the server advertises its capabilities dynamically at session start, so the model can adapt to whatever tools are available. Both problems are irrelevant when you have one model and one tool with a stable interface \u2014 which is why "what problems does MCP solve" is also the correct frame for "when not to use MCP." If neither problem applies to your current system, the protocol layer adds overhead without payoff.',source:"https://modelcontextprotocol.io/specification/2025-11-25"},{id:"q1-4",num:"1.4",question:"What does the MCP do?",answer:"MCP manages the full lifecycle of a tool-calling session between an AI model and an external server: capability negotiation at connection, tool invocation during the session, and structured result delivery back to the model. <strong>The stateful session is MCP's defining feature \u2014 unlike a REST API call where each request is independent, an MCP session maintains context across multiple tool calls, so the model can use the output of one tool as the input to the next without the orchestration living inside the model itself.</strong> In concrete terms: when a Claude instance connects to an MCP server, the server first sends a list of available tools with descriptions. Claude reads those descriptions, decides which tool to call, sends a structured JSON-RPC request, and receives a structured result. The session stays open so Claude can call additional tools without re-authenticating or re-negotiating capabilities. For single-tool, single-call integrations, this session overhead is pure cost.",source:"https://modelcontextprotocol.io/specification/2025-11-25"},{id:"q1-5",num:"1.5",question:"Is MCP a tool or framework?",answer:`MCP is a protocol \u2014 not a tool, not a framework, and not a library. <strong>A protocol defines the rules for how two parties communicate; it does not prescribe how either party is implemented, which is what makes MCP portable across AI models and tool servers built in different languages and architectures.</strong> The practical consequence is that calling MCP a "framework" is a category error that leads to wrong architectural expectations: frameworks come with opinions about structure, abstractions, and project layout. MCP has none of those \u2014 it defines message formats, session lifecycle, and capability negotiation only. If you need a framework to build MCP clients or servers, you use an SDK (Anthropic provides SDKs for Python and TypeScript); the SDK is the framework layer on top of the protocol. The distinction matters when evaluating adoption cost: you're not adopting a framework with its opinionated structure, you're implementing a protocol that can live inside whatever structure you already have.`,source:"https://modelcontextprotocol.io/specification/2025-11-25"},{id:"q1-6",num:"1.6",question:"What is the difference between MCP and LLM?",answer:"An LLM (Large Language Model) is the AI system that reasons and generates text; MCP is the protocol that tells the LLM how to interact with external tools. <strong>The relationship is one-directional: LLMs use MCP \u2014 MCP does not use or require an LLM.</strong> An LLM without MCP can only work with information it was trained on and whatever appears in its context window. An LLM with MCP can call external servers to retrieve current data, execute code, query databases, and take actions in other systems \u2014 then incorporate those results into its reasoning. MCP adds the tool-use capability; the LLM supplies the reasoning about when and how to use those tools. The confusion between MCP and LLM typically indicates someone comparing a capability (tool-calling) with the system exercising that capability (the model) \u2014 the correct comparison is MCP vs. function calling (another tool-use mechanism built into model APIs), not MCP vs. LLM."}],interactiveHtml:`<div class="si si-quiz">
|
|
2
2
|
<span class="si-heading">Quick check \u2014 what did you just read?</span>
|
|
3
3
|
<div class="si-quiz-q" data-correct="1">
|
|
4
4
|
<p class="si-quiz-q-text">What was broken before MCP existed that MCP is designed to fix?</p>
|
|
@@ -1627,7 +1627,7 @@ RULES:
|
|
|
1627
1627
|
WHERE id = ?`,args:[JSON.stringify(r),t]})}async failSiteAuditJob(t,r){await this.db.execute({sql:`UPDATE site_audit_jobs
|
|
1628
1628
|
SET status = 'failed', error = ?, updated_at = datetime('now')
|
|
1629
1629
|
WHERE id = ?`,args:[r,t]})}async logSiteAuditPhaseComplete(t,r,n){await this.db.execute({sql:`INSERT OR IGNORE INTO site_audit_phase_log (job_id, phase, output_summary, completed_at)
|
|
1630
|
-
VALUES (?, ?, ?, datetime('now'))`,args:[t,r,JSON.stringify(n)]}).catch(i=>{console.warn("[site-audit-repository] logSiteAuditPhaseComplete failed:",i instanceof Error?i.message:String(i))})}};import*as dm from"child_process";import*as Xl from"path";import*as NP from"util";import{fileURLToPath as _W}from"url";var TP=NP.promisify(dm.execFile),PP=Xl.dirname(_W(import.meta.url)),lm=class{graphMetricsScript;locationClassifierScript;constructor(){try{dm.execFileSync("python3",["-c","import networkx"])}catch{throw new Error("SiteAuditPythonRunner: networkx not available")}this.graphMetricsScript=Xl.join(PP,"scripts","compute_graph_metrics.py"),this.locationClassifierScript=Xl.join(PP,"scripts","location_classifier.py")}async runGraphMetrics(t){let r=_P.parse(t),n,i;try{({stdout:n,stderr:i}=await TP("python3",[this.graphMetricsScript,"--input",JSON.stringify(r)],{timeout:12e4}))}catch(o){let s=o;throw new Error(`python-subprocess-001: ${s.stderr??String(o)}`)}return vP.parse(JSON.parse(n))}async runLocationClassifier(t){let r=xP.parse(t),n,i;try{({stdout:n,stderr:i}=await TP("python3",[this.locationClassifierScript,"--input",JSON.stringify(r)],{timeout:12e4}))}catch(o){let s=o;throw new Error(`python-subprocess-001: ${s.stderr??String(o)}`)}return SP.parse(JSON.parse(n))}};function Xr(){let e=process.env.DEEPINFRA_API_KEY;if(!e)throw new Error("DEEPINFRA_API_KEY is required");let t=new im(e),r=process.env.OPENROUTER_API_KEY?new sm(t,new om(process.env.OPENROUTER_API_KEY)):t,n=new mc,i=new am,o=new lm;return new cm({repo:n,llm:r,http:i,python:o})}import LP from"p-limit";async function Tn(e,t,r){switch(t){case"phase1-ingest":return vW(e,r);case"phase2-build-graph":return xW(e,r);case"phase3-classify":return SW(e,r);case"phase4-compare":return EW(e,r);case"phase5-synthesize":return kW(e,r);default:{let n=t;throw new Error("Unknown phase: "+String(n))}}}async function vW(e,t){let r=pP.parse(t);await e.runIngestPhase(r.jobId,r),await e.runEnrichPhase(r.jobId,r)}async function xW(e,t){let r=mP.parse(t),i=[LP(3)(()=>e.runBuildGraphPhase(r.jobId,r))],o=await Promise.allSettled(i);for(let s of o)s.status==="rejected"&&console.warn("[phases] handleBuildGraphPhase: phase failed for job",r.jobId,s.reason instanceof Error?s.reason.message:String(s.reason))}async function SW(e,t){let r=gP.parse(t),i=[LP(3)(()=>e.runClassifyPhase(r.jobId,r))],o=await Promise.allSettled(i);for(let s of o)s.status==="rejected"&&console.warn("[phases] handleClassifyPhase: phase failed for job",r.jobId,s.reason instanceof Error?s.reason.message:String(s.reason))}async function EW(e,t){let r=fP.parse(t),i=(await e.runComparePhase(r.jobId,r)).linkRecommendations??[],s=iP(i,{resolveFlags:(a,c)=>[]});console.warn("[phases] handleComparePhase: guardrail check complete \u2014 passed:",s.passed.length,"flagged:",s.flagged.length,"(context-dependent flags require site-graph state not yet exposed by runComparePhase)")}async function kW(e,t){let r=hP.parse(t);await e.runSynthesizePhase(r.jobId,r)}var MP=at.createFunction({id:"site-audit",triggers:[{event:"site-architecture-auditor/audit.requested"}]},async({event:e,step:t})=>{let{phase:r,payload:n}=e.data;return await t.run("dispatch-phase",async()=>{let i=Xr();await Tn(i,r,n)}),{phase:r,status:"done"}});var AW=2e3,OP=512*1024,um=180;function IW(e,t=10){let r=Math.max(1,Math.min(10,Math.floor(Number.isFinite(e)?e:1))),n=Math.max(1,Math.min(100,Math.floor(Number.isFinite(t)?t:10)));return Math.min(20,r*Math.min(2,n))}function RW(e){return e==="complete"||e==="partial"||e==="failed"}async function DP(e,t){let r=`extract-site:${e}`,n=await Fs(r).catch(()=>null),i=n?.attempts[0];i?.status==="running"&&await $s({attemptId:String(i.id),status:t,durationMs:Math.max(0,Date.now()-Date.parse(String(i.started_at)))}).catch(()=>{}),n?.run.status==="running"&&await js({runId:r,status:t,failureClass:t==="failed"?"site_extract_failed_or_partial":null}).catch(()=>{})}function CW(e,t){let r=ur(e.url);return r?{...e,archivedUrl:e.url,archiveTimestamp:r.timestamp,archiveRequestedMonth:t??null,originalUrl:r.originalUrl,url:r.originalUrl,finalUrl:r.originalUrl}:e}async function Pb(e){if(e.userId==null||e.billedMc!=null)return;if(e.options.billingClass===ny){if(!await kk(e.id))throw new Error("xray_extract_zero_charge_finalization_rejected");e.publicError&&await ro(e.id,{...e.publicError,charge_status:"not_charged"});return}let t=Number(e.options.heldMc??0),r=await up(e.id),n=Math.min(r*D.page_scrape,t),i="site";try{i=new URL(e.startUrl).hostname}catch{}await rc(e.id,e.userId,t-n,n,i),e.publicError&&await ro(e.id,{...e.publicError,charge_status:n===0?"refunded":"charged"})}function TW(e,t="refund_pending"){let{error:r,...n}=Zn(no(e),{chargeStatus:t});return n}var qP=at.createFunction({id:"site-extract",retries:2,triggers:[{event:"mcp-scraper/extract.requested"}],onFailure:async({event:e})=>{let t=e?.data?.event?.data?.jobId;if(!t)return;let r=String(e?.data?.error?.message??"crawl failed"),n=/FUNCTION_INVOCATION_FAILED|out of memory|heap/i.test(r)?"Result assembly failed on our side \u2014 your crawl data is safe and the job can be re-finalized without re-crawling. Contact support if this persists.":r;await Sk(t,n,TW(new Error(n)));let i=await _i(t);if(i&&i.userId!=null&&i.billedMc==null)try{await Pb(i)}catch(o){console.error("[site-extract/onFailure] settlement pending reconciliation:",o instanceof Error?o.message:String(o))}await DP(t,"failed")}},async({event:e,step:t})=>{let r=e.data.jobId,n=await t.run("load-job",()=>_i(r));if(!n)return{jobId:r,status:"missing"};if(RW(n.status))return n.billedMc==null&&await t.run("settle-terminal-job",()=>Pb(n)),{jobId:r,status:n.status,artifacts:n.artifacts??[]};let i=ae(),o=null,s=null;try{o=(await Ya({id:`extract-site:${r}`,ownerScope:`user:${n.userId}`,userId:n.userId,tool:"extract_site",idempotencyKey:`job:${r}`,requestFingerprint:r,jobId:r,normalizedFlags:{maxPages:n.options.maxPages??null},deploymentCommit:process.env.VERCEL_GIT_COMMIT_SHA?.trim()||process.env.GIT_COMMIT_SHA?.trim()||null})).id;let B=await Fs(o);s=B?.attempts[0]?.id==null?await ip({runId:o,attemptNumber:1,method:"site_crawl_root",normalizedFlags:{maxPages:n.options.maxPages??null}}):String(B.attempts[0].id);let F=typeof n.options.debitKey=="string"?n.options.debitKey:null;F&&Number(n.options.heldMc??0)>0&&await Xi({runId:o,billingTable:"billing_debits",billingKey:F,relation:"debit",amountMc:Number(n.options.heldMc??0)})}catch(B){console.warn(JSON.stringify({event:"site_operation_start_failed",job_id:r,message:B instanceof Error?B.message:String(B)}))}let a={op:"extract_site",probeRunId:n.id,userId:n.userId,subOp:"background_site_extract",operationRunId:o,operationRootAttemptId:s,operationAccountScope:"production-default"},c=Number(n.options.maxPages??1e4),l=Number(n.options.concurrency??1),d=Number(n.options.urlsPerBrowser??n.options.rotateProxyEvery??10),u=n.options.captureRenderedDom===!0,p=n.options.renderJavaScript===!0||n.options.semanticSimilarity===!0||u,m=IW(l,d),g=ur(n.startUrl),f=n.options.waybackTimeline,h=n.options.disableLinkDiscovery===!0||!!g||!!f,y=await t.run("discover",async()=>{if(f){let U=await ve(a,()=>Pt("site_discovery",()=>UT({rootUrl:f.rootUrl,timeline:f.timeline,maxPagesPerSnapshot:f.maxPagesPerSnapshot,maxCaptures:c}))),W=U.captures.map(ee=>ee.rawReplayUrl),K=U.requestedUrls?U.months.length*U.requestedUrls.length:W.length+U.missing.length;return await Ol(r,Math.max(W.length,K)),{urls:W,sitemapUrls:[],waybackMonthsByUrl:Object.fromEntries(U.captures.map(ee=>[ee.rawReplayUrl,ee.requestedMonth??""]))}}if(g){let W=(await ve(a,()=>Pt("site_discovery",()=>Yp(g,c)))).map(K=>K.rawReplayUrl);return await Ol(r,W.length),{urls:W,sitemapUrls:[]}}let B=await ve(a,()=>QT(n.startUrl,c,i,{spiderFallback:!1})),F=B.urls.length?B.urls:[n.startUrl];return await Ol(r,F.length),{urls:F,sitemapUrls:B.sitemapUrls}}),b=Array.isArray(y)?{urls:y,sitemapUrls:[],waybackMonthsByUrl:void 0}:y,_=b.waybackMonthsByUrl??{},S=(()=>{try{return new URL(n.startUrl).hostname.replace(/^www\./,"")}catch{return""}})(),x=B=>{try{let F=new URL(B);return F.hash="",(F.origin+F.pathname).replace(/\/+$/,"")+F.search}catch{return B}},E=B=>{try{return new URL(B).hostname.replace(/^www\./,"")===S}catch{return!1}},v=b.sitemapUrls.length>0?new Set(b.sitemapUrls.map(x)):null,k=[],A=new Set,C=0;for(let B of b.urls){let F=x(B),U=Buffer.byteLength(F);F.length>es||U>es||A.has(F)||C+U>Zp||(A.add(F),C+=U,k.push(B))}let T=0;for(let B=0;T<c&&T<k.length;B++){let F=k.slice(T,Math.min(T+m,c));if(!F.length)break;let U=Math.max(0,c-A.size),W=await t.run(`crawl-batch-${B}`,async()=>{let ee=await ve(a,()=>Eb(F,{kernelApiKey:i,concurrency:l,urlsPerBrowser:d,forceBrowserRender:p,captureRenderedDom:u})),ue=ee.map(ne=>CW({...ne,acquiredHtml:ne.acquiredHtml,mainHtml:ne.mainHtml,inSitemap:v?v.has(x(ne.url)):null},_[ne.url]));if(await wk(r,ue),h)return[];let pe=new Set,he=0,Je=Math.min(U,AW);for(let ne of ee.filter(rs)){let be=ne.discoveryLinks??(ne.outlinks??[]).filter(se=>se.internal).map(se=>se.href);for(let se of be){if(pe.size>=Je)break;if(!se||!E(se)||pe.has(se))continue;let Ne=Buffer.byteLength(se);if(he+Ne>OP)break;pe.add(se),he+=Ne}if(pe.size>=Je||he>=OP)break}return[...pe]});T+=F.length;let K=!1;for(let ee of W){let ue=x(ee),pe=Buffer.byteLength(ue);!A.has(ue)&&A.size<c&&ue.length<=es&&pe<=es&&C+pe<=Zp&&(A.add(ue),C+=pe,k.push(ee),K=!0)}K&&await Ol(r,Math.min(A.size,c))}let I=Array.isArray(n.options.formats)&&n.options.formats.includes("branding")?await t.run("branding",()=>ve(a,()=>Qp(n.startUrl,i).catch(()=>null))):null,L=Array.isArray(n.options.formats)&&n.options.formats.includes("images")?await t.run("image-audit",async()=>{let B=await vk(r,um+1);return DE(B.slice(0,um),{concurrency:12,timeoutMs:12e3,max:um,sampleTruncated:B.length>um})}):null,H=await t.run("finalize",async()=>{let{assembleExtractArtifacts:B}=await import("./extract-bundle-MQOAKQDV.js"),F=await _i(r);if(!F)throw new Error("extract job disappeared before finalization");let U=await ve(a,()=>B(F,{branding:I,imageAudit:L})),W=Ws(F),K=lk(F,W.creditTruncated),ee=K==="failed"?`No pages were extracted successfully (${F.failedUrls} failed, ${F.remainingUrls} remaining).`:K==="partial"?`${F.successfulUrls} pages succeeded; ${F.failedUrls} failed and ${F.remainingUrls} remain.${W.creditTruncated?` Credit availability limited this request from ${W.requestedMaxPages} to ${W.effectiveMaxPages} pages.`:""}`:null,ue=K==="failed"?$r({errorCode:"extraction_failed",retryable:!0,chargeStatus:"refund_pending"}):null;if(!await xk(r,U,K,ee,!1,ue))throw new Error("site extract job state changed before finalization committed");return{artifacts:U,status:K}});return await t.run("settle",async()=>{let B=await _i(r);B&&await Pb(B)}),await DP(r,H.status==="complete"?"succeeded":"failed"),{jobId:r,status:H.status,artifacts:H.artifacts}});import{randomUUID as PW}from"crypto";var UP="image-sources/",NW=10080*60*1e3,LW=900*1e3,jP=20*1024*1024,MW=new Map([["image/jpeg","jpg"],["image/png","png"],["image/webp","webp"],["image/gif","gif"]]);function OW(){return process.env.IMAGE_SOURCE_ARTIFACT_READ_WRITE_TOKEN?.trim()||process.env.PRIVATE_ARTIFACT_READ_WRITE_TOKEN?.trim()||process.env.CONNECTED_DATA_READ_WRITE_TOKEN?.trim()||process.env.CONNECTED_DATA_BLOB_READ_WRITE_TOKEN?.trim()||null}function $P(){return{prefix:UP,artifactTtlMs:NW,downloadTtlMs:LW,token:OW()}}function DW(e){return to(e,UP)}async function FP(e){let t=MW.get(e.contentType);if(!t)throw new Error("image_source_type_unsupported");if(e.content.length===0||e.content.length>jP)throw new Error("image_source_size_invalid");let r=PW().replaceAll("-","");return Vr({policy:$P(),ownerId:e.ownerId,artifactKey:`${r}.${t}`,createdAt:e.createdAt??new Date,filename:e.filename??`mcp-scraper-image-${r.slice(0,12)}.${t}`,contentType:e.contentType,content:e.content})}async function pm(e){return DW(e.artifactId)!==e.ownerId?null:Xn({policy:$P(),artifactId:e.artifactId,maxBytes:jP})}import{createHash as qW,randomBytes as UW}from"crypto";import{homedir as jW}from"os";import{basename as $W,extname as FW,join as BP}from"path";import BW from"p-limit";import{ZipFile as HW}from"yazl";var mm="page-media/",WW=10080*60*1e3,zW=900*1e3,Nb=10*1024*1024,KW=45*1024*1024,GW=15e5,VW=25e5,JW=6,HP=5;function zP(){return process.env.PAGE_MEDIA_ARTIFACT_READ_WRITE_TOKEN?.trim()||process.env.PRIVATE_ARTIFACT_READ_WRITE_TOKEN?.trim()||process.env.CONNECTED_DATA_READ_WRITE_TOKEN?.trim()||process.env.CONNECTED_DATA_BLOB_READ_WRITE_TOKEN?.trim()||null}function KP(){return{prefix:mm,artifactTtlMs:WW,downloadTtlMs:zW,token:zP()}}function YW(e,t){if(e.length>=3&&e[0]===255&&e[1]===216&&e[2]===255)return{mimeType:"image/jpeg",extension:"jpg"};if(e.length>=8&&e[0]===137&&e[1]===80&&e[2]===78&&e[3]===71)return{mimeType:"image/png",extension:"png"};if(e.length>=12&&String.fromCharCode(...e.slice(0,4))==="RIFF"&&String.fromCharCode(...e.slice(8,12))==="WEBP")return{mimeType:"image/webp",extension:"webp"};if(e.length>=6&&["GIF87a","GIF89a"].includes(String.fromCharCode(...e.slice(0,6))))return{mimeType:"image/gif",extension:"gif"};if(e.length>=12&&String.fromCharCode(...e.slice(4,12)).match(/^ftyp(?:avif|avis)$/))return{mimeType:"image/avif",extension:"avif"};let r=Buffer.from(e.slice(0,Math.min(e.length,1024))).toString("utf8").replace(/^\uFEFF/,"").trimStart();return(t==="image/svg+xml"||r.startsWith("<svg")||/^<\?xml[\s\S]*?<svg/i.test(r))&&/<svg[\s>]/i.test(r)?{mimeType:"image/svg+xml",extension:"svg"}:null}function XW(e,t,r){if(r==="image")return YW(e,t);if(e.length>=12&&String.fromCharCode(...e.slice(4,8))==="ftyp")return{mimeType:"video/mp4",extension:"mp4"};if(e.length>=4&&e[0]===26&&e[1]===69&&e[2]===223&&e[3]===163)return{mimeType:"video/webm",extension:"webm"};if(e.length>=4&&String.fromCharCode(...e.slice(0,4))==="OggS")return r==="audio"?{mimeType:"audio/ogg",extension:"ogg"}:{mimeType:"video/ogg",extension:"ogv"};if(r==="audio"&&e.length>=3&&String.fromCharCode(...e.slice(0,3))==="ID3")return{mimeType:"audio/mpeg",extension:"mp3"};if(t&&Yh(t)===r){let n=t.split("/")[1]?.replace("jpeg","jpg").replace("mpeg","mp3").replace("svg+xml","svg");if(n&&/^[a-z0-9]+$/i.test(n))return{mimeType:t,extension:n}}return null}function ZW(e,t){let r=Buffer.from(e);if(t==="image/png"&&r.length>=24)return{width:r.readUInt32BE(16),height:r.readUInt32BE(20)};if(t==="image/gif"&&r.length>=10)return{width:r.readUInt16LE(6),height:r.readUInt16LE(8)};if(t==="image/webp"&&r.length>=30){let n=r.toString("ascii",12,16);if(n==="VP8X")return{width:1+r.readUIntLE(24,3),height:1+r.readUIntLE(27,3)};if(n==="VP8L"&&r[20]===47){let i=r.readUInt32LE(21);return{width:(i&16383)+1,height:(i>>14&16383)+1}}}if(t==="image/jpeg"){let n=2;for(;n+9<r.length;){if(r[n]!==255){n+=1;continue}let i=r[n+1];if(i===216||i===217){n+=2;continue}let o=r.readUInt16BE(n+2);if(o<2||n+o+2>r.length)break;if([192,193,194,195,197,198,199,201,202,203,205,206,207].includes(i))return{height:r.readUInt16BE(n+5),width:r.readUInt16BE(n+7)};n+=o+2}}if(t==="image/svg+xml"){let i=r.toString("utf8",0,Math.min(r.length,65536)).match(/<svg\b[^>]*>/i)?.[0]??"",o=Number(i.match(/\bwidth=["']?([\d.]+)/i)?.[1]??0),s=Number(i.match(/\bheight=["']?([\d.]+)/i)?.[1]??0);if(o>0&&s>0)return{width:Math.round(o),height:Math.round(s)};let a=i.match(/\bviewBox=["']([^"']+)["']/i)?.[1]?.trim().split(/[\s,]+/).map(Number);if(a?.length===4&&a[2]>0&&a[3]>0)return{width:Math.round(a[2]),height:Math.round(a[3])}}return null}async function QW(e){let t=Number(e.headers.get("content-length")??"");if(Number.isFinite(t)&&t>Nb)throw new Error("media_too_large");if(!e.body){let o=Buffer.from(await e.arrayBuffer());if(o.length>Nb)throw new Error("media_too_large");return o}let r=e.body.getReader(),n=[],i=0;try{for(;;){let{done:o,value:s}=await r.read();if(o)break;if(s){if(i+=s.byteLength,i>Nb)throw new Error("media_too_large");n.push(Buffer.from(s))}}}finally{r.releaseLock()}return Buffer.concat(n)}async function ez(e,t){let r=e;for(let n=0;n<=HP;n+=1){let i=await fe(r,{field:"website media URL"});if(i.error||!i.parsed)throw new Error("media_url_rejected");let o=await fetch(i.parsed.href,{redirect:"manual",headers:{Accept:t==="image"?"image/*":`${t}/*`},signal:AbortSignal.timeout(15e3)});if([301,302,303,307,308].includes(o.status)){let d=o.headers.get("location");if(!d||n===HP)throw new Error("media_redirect_rejected");r=new URL(d,i.parsed).href;continue}if(!o.ok)throw new Error(`media_fetch_${o.status}`);let s=await QW(o),a=o.headers.get("content-type")?.split(";")[0].trim().toLowerCase()??null,c=XW(s,a,t);if(!c||Yh(c.mimeType)!==t)throw new Error("media_content_invalid");let l=t==="image"?ZW(s,c.mimeType):null;return{bytes:s,...c,finalUrl:i.parsed.href,width:l?.width??null,height:l?.height??null,sha256:qW("sha256").update(s).digest("hex")}}throw new Error("media_redirect_rejected")}function tz(e){return new Promise((t,r)=>{let n=new HW,i=[];n.outputStream.on("data",o=>i.push(Buffer.from(o))),n.outputStream.once("error",r),n.outputStream.once("end",()=>t(Buffer.concat(i)));for(let o of e)n.addBuffer(o.content,o.path);n.end()})}function WP(e){return e.toLowerCase().replace(/[^a-z0-9]+/g,"-").replace(/^-+|-+$/g,"").slice(0,70)||"website"}function rz(e){try{return $W(new URL(e).pathname).replace(/[^a-zA-Z0-9._-]+/g,"-").replace(/^-+|-+$/g,"").slice(0,80)}catch{return""}}function nz(e){return to(e,mm)}async function GP(e){return nz(e.artifactId)!==e.ownerId?null:Xn({policy:KP(),artifactId:e.artifactId,maxBytes:60*1024*1024})}async function VP(e){let t=[],r=BW(JW),n=new Map,i=0,o=0,s=0;await Promise.all(e.media.assets.map((f,h)=>r(async()=>{try{let y=await ez(f.url,f.type);f.finalUrl=y.finalUrl,f.mimeType=y.mimeType,f.sizeBytes=y.bytes.length,f.sha256=y.sha256,f.width=y.width??f.width,f.height=y.height??f.height;let b=n.get(y.sha256);if(b){f.filename=b,f.duplicateOf=b,f.downloadStatus="downloaded",f.downloadError=null;return}if(i+y.bytes.length>KW)throw new Error("archive_byte_limit_reached");i+=y.bytes.length;let _=rz(y.finalUrl),S=FW(_),x=_?S?_.slice(0,-S.length):_:f.type,E=`media/${String(h+1).padStart(4,"0")}-${WP(x)}.${y.extension}`;n.set(y.sha256,E),t.push({path:E,content:y.bytes}),f.filename=E,f.duplicateOf=null,f.downloadStatus="downloaded",f.downloadError=null,f.type==="image"&&s<e.maxInlineImages&&y.bytes.length<=GW&&o+y.bytes.length<=VW&&(f.inlinePreview={data:y.bytes.toString("base64"),mimeType:y.mimeType},o+=y.bytes.length,s+=1)}catch(y){f.finalUrl=null,f.mimeType=null,f.sizeBytes=null,f.sha256=null,f.savedPath=null,f.downloadStatus="failed",f.downloadError=y instanceof Error?y.message:"media_download_failed",f.inlinePreview=null}})));let a=e.media.assets.filter(f=>f.downloadStatus==="downloaded").length,c=e.media.assets.length-a;if(c>0&&e.media.warnings.push(`${c} media asset(s) could not be downloaded; media.jsonl contains bounded error codes.`),a===0)return{media:e.media,artifact:null};let l=e.media.assets.map(({inlinePreview:f,...h})=>h),d={pageUrl:e.pageUrl,title:e.title,extractedAt:e.extractedAt,completeness:e.media.completeness,exhausted:e.media.exhausted,stopReason:e.media.stopReason,scrollRounds:e.media.scrollRounds,staticFound:e.media.staticFound,renderedFound:e.media.renderedFound,totalFound:e.media.totalFound,retainedAfterFilteringAndVariantCollapse:e.media.assets.length,downloaded:a,failed:c,notes:["The inventory unions static-source and rendered-page discovery.","complete means rendered scrolling reached three stable document-height checks; it does not claim every interactive control was opened.","Media records preserve discovery evidence but do not assert ownership, rights, identity, or publishability."]};t.push({path:"summary.json",content:Buffer.from(JSON.stringify(d,null,2))}),t.push({path:"media.jsonl",content:Buffer.from(l.map(f=>JSON.stringify(f)).join(`
|
|
1630
|
+
VALUES (?, ?, ?, datetime('now'))`,args:[t,r,JSON.stringify(n)]}).catch(i=>{console.warn("[site-audit-repository] logSiteAuditPhaseComplete failed:",i instanceof Error?i.message:String(i))})}};import*as dm from"child_process";import*as Xl from"path";import*as NP from"util";import{fileURLToPath as _W}from"url";var TP=NP.promisify(dm.execFile),PP=Xl.dirname(_W(import.meta.url)),lm=class{graphMetricsScript;locationClassifierScript;constructor(){try{dm.execFileSync("python3",["-c","import networkx"])}catch{throw new Error("SiteAuditPythonRunner: networkx not available")}this.graphMetricsScript=Xl.join(PP,"scripts","compute_graph_metrics.py"),this.locationClassifierScript=Xl.join(PP,"scripts","location_classifier.py")}async runGraphMetrics(t){let r=_P.parse(t),n,i;try{({stdout:n,stderr:i}=await TP("python3",[this.graphMetricsScript,"--input",JSON.stringify(r)],{timeout:12e4}))}catch(o){let s=o;throw new Error(`python-subprocess-001: ${s.stderr??String(o)}`)}return vP.parse(JSON.parse(n))}async runLocationClassifier(t){let r=xP.parse(t),n,i;try{({stdout:n,stderr:i}=await TP("python3",[this.locationClassifierScript,"--input",JSON.stringify(r)],{timeout:12e4}))}catch(o){let s=o;throw new Error(`python-subprocess-001: ${s.stderr??String(o)}`)}return SP.parse(JSON.parse(n))}};function Xr(){let e=process.env.DEEPINFRA_API_KEY;if(!e)throw new Error("DEEPINFRA_API_KEY is required");let t=new im(e),r=process.env.OPENROUTER_API_KEY?new sm(t,new om(process.env.OPENROUTER_API_KEY)):t,n=new mc,i=new am,o=new lm;return new cm({repo:n,llm:r,http:i,python:o})}import LP from"p-limit";async function Tn(e,t,r){switch(t){case"phase1-ingest":return vW(e,r);case"phase2-build-graph":return xW(e,r);case"phase3-classify":return SW(e,r);case"phase4-compare":return EW(e,r);case"phase5-synthesize":return kW(e,r);default:{let n=t;throw new Error("Unknown phase: "+String(n))}}}async function vW(e,t){let r=pP.parse(t);await e.runIngestPhase(r.jobId,r),await e.runEnrichPhase(r.jobId,r)}async function xW(e,t){let r=mP.parse(t),i=[LP(3)(()=>e.runBuildGraphPhase(r.jobId,r))],o=await Promise.allSettled(i);for(let s of o)s.status==="rejected"&&console.warn("[phases] handleBuildGraphPhase: phase failed for job",r.jobId,s.reason instanceof Error?s.reason.message:String(s.reason))}async function SW(e,t){let r=gP.parse(t),i=[LP(3)(()=>e.runClassifyPhase(r.jobId,r))],o=await Promise.allSettled(i);for(let s of o)s.status==="rejected"&&console.warn("[phases] handleClassifyPhase: phase failed for job",r.jobId,s.reason instanceof Error?s.reason.message:String(s.reason))}async function EW(e,t){let r=fP.parse(t),i=(await e.runComparePhase(r.jobId,r)).linkRecommendations??[],s=iP(i,{resolveFlags:(a,c)=>[]});console.warn("[phases] handleComparePhase: guardrail check complete \u2014 passed:",s.passed.length,"flagged:",s.flagged.length,"(context-dependent flags require site-graph state not yet exposed by runComparePhase)")}async function kW(e,t){let r=hP.parse(t);await e.runSynthesizePhase(r.jobId,r)}var MP=at.createFunction({id:"site-audit",triggers:[{event:"site-architecture-auditor/audit.requested"}]},async({event:e,step:t})=>{let{phase:r,payload:n}=e.data;return await t.run("dispatch-phase",async()=>{let i=Xr();await Tn(i,r,n)}),{phase:r,status:"done"}});var AW=2e3,OP=512*1024,um=180;function IW(e,t=10){let r=Math.max(1,Math.min(10,Math.floor(Number.isFinite(e)?e:1))),n=Math.max(1,Math.min(100,Math.floor(Number.isFinite(t)?t:10)));return Math.min(20,r*Math.min(2,n))}function RW(e){return e==="complete"||e==="partial"||e==="failed"}async function DP(e,t){let r=`extract-site:${e}`,n=await Fs(r).catch(()=>null),i=n?.attempts[0];i?.status==="running"&&await $s({attemptId:String(i.id),status:t,durationMs:Math.max(0,Date.now()-Date.parse(String(i.started_at)))}).catch(()=>{}),n?.run.status==="running"&&await js({runId:r,status:t,failureClass:t==="failed"?"site_extract_failed_or_partial":null}).catch(()=>{})}function CW(e,t){let r=ur(e.url);return r?{...e,archivedUrl:e.url,archiveTimestamp:r.timestamp,archiveRequestedMonth:t??null,originalUrl:r.originalUrl,url:r.originalUrl,finalUrl:r.originalUrl}:e}async function Pb(e){if(e.userId==null||e.billedMc!=null)return;if(e.options.billingClass===ny){if(!await kk(e.id))throw new Error("xray_extract_zero_charge_finalization_rejected");e.publicError&&await ro(e.id,{...e.publicError,charge_status:"not_charged"});return}let t=Number(e.options.heldMc??0),r=await up(e.id),n=Math.min(r*D.page_scrape,t),i="site";try{i=new URL(e.startUrl).hostname}catch{}await rc(e.id,e.userId,t-n,n,i),e.publicError&&await ro(e.id,{...e.publicError,charge_status:n===0?"refunded":"charged"})}function TW(e,t="refund_pending"){let{error:r,...n}=Zn(no(e),{chargeStatus:t});return n}var qP=at.createFunction({id:"site-extract",retries:2,triggers:[{event:"mcp-scraper/extract.requested"}],onFailure:async({event:e})=>{let t=e?.data?.event?.data?.jobId;if(!t)return;let r=String(e?.data?.error?.message??"crawl failed"),n=/FUNCTION_INVOCATION_FAILED|out of memory|heap/i.test(r)?"Result assembly failed on our side \u2014 your crawl data is safe and the job can be re-finalized without re-crawling. Contact support if this persists.":r;await Sk(t,n,TW(new Error(n)));let i=await _i(t);if(i&&i.userId!=null&&i.billedMc==null)try{await Pb(i)}catch(o){console.error("[site-extract/onFailure] settlement pending reconciliation:",o instanceof Error?o.message:String(o))}await DP(t,"failed")}},async({event:e,step:t})=>{let r=e.data.jobId,n=await t.run("load-job",()=>_i(r));if(!n)return{jobId:r,status:"missing"};if(RW(n.status))return n.billedMc==null&&await t.run("settle-terminal-job",()=>Pb(n)),{jobId:r,status:n.status,artifacts:n.artifacts??[]};let i=ae(),o=null,s=null;try{o=(await Ya({id:`extract-site:${r}`,ownerScope:`user:${n.userId}`,userId:n.userId,tool:"extract_site",idempotencyKey:`job:${r}`,requestFingerprint:r,jobId:r,normalizedFlags:{maxPages:n.options.maxPages??null},deploymentCommit:process.env.VERCEL_GIT_COMMIT_SHA?.trim()||process.env.GIT_COMMIT_SHA?.trim()||null})).id;let B=await Fs(o);s=B?.attempts[0]?.id==null?await ip({runId:o,attemptNumber:1,method:"site_crawl_root",normalizedFlags:{maxPages:n.options.maxPages??null}}):String(B.attempts[0].id);let F=typeof n.options.debitKey=="string"?n.options.debitKey:null;F&&Number(n.options.heldMc??0)>0&&await Xi({runId:o,billingTable:"billing_debits",billingKey:F,relation:"debit",amountMc:Number(n.options.heldMc??0)})}catch(B){console.warn(JSON.stringify({event:"site_operation_start_failed",job_id:r,message:B instanceof Error?B.message:String(B)}))}let a={op:"extract_site",probeRunId:n.id,userId:n.userId,subOp:"background_site_extract",operationRunId:o,operationRootAttemptId:s,operationAccountScope:"production-default"},c=Number(n.options.maxPages??1e4),l=Number(n.options.concurrency??1),d=Number(n.options.urlsPerBrowser??n.options.rotateProxyEvery??10),u=n.options.captureRenderedDom===!0,p=n.options.renderJavaScript===!0||n.options.semanticSimilarity===!0||u,m=IW(l,d),g=ur(n.startUrl),f=n.options.waybackTimeline,h=n.options.disableLinkDiscovery===!0||!!g||!!f,y=await t.run("discover",async()=>{if(f){let U=await ve(a,()=>Pt("site_discovery",()=>UT({rootUrl:f.rootUrl,timeline:f.timeline,maxPagesPerSnapshot:f.maxPagesPerSnapshot,maxCaptures:c}))),W=U.captures.map(ee=>ee.rawReplayUrl),K=U.requestedUrls?U.months.length*U.requestedUrls.length:W.length+U.missing.length;return await Ol(r,Math.max(W.length,K)),{urls:W,sitemapUrls:[],waybackMonthsByUrl:Object.fromEntries(U.captures.map(ee=>[ee.rawReplayUrl,ee.requestedMonth??""]))}}if(g){let W=(await ve(a,()=>Pt("site_discovery",()=>Yp(g,c)))).map(K=>K.rawReplayUrl);return await Ol(r,W.length),{urls:W,sitemapUrls:[]}}let B=await ve(a,()=>QT(n.startUrl,c,i,{spiderFallback:!1})),F=B.urls.length?B.urls:[n.startUrl];return await Ol(r,F.length),{urls:F,sitemapUrls:B.sitemapUrls}}),b=Array.isArray(y)?{urls:y,sitemapUrls:[],waybackMonthsByUrl:void 0}:y,_=b.waybackMonthsByUrl??{},S=(()=>{try{return new URL(n.startUrl).hostname.replace(/^www\./,"")}catch{return""}})(),x=B=>{try{let F=new URL(B);return F.hash="",(F.origin+F.pathname).replace(/\/+$/,"")+F.search}catch{return B}},E=B=>{try{return new URL(B).hostname.replace(/^www\./,"")===S}catch{return!1}},v=b.sitemapUrls.length>0?new Set(b.sitemapUrls.map(x)):null,k=[],A=new Set,C=0;for(let B of b.urls){let F=x(B),U=Buffer.byteLength(F);F.length>es||U>es||A.has(F)||C+U>Zp||(A.add(F),C+=U,k.push(B))}let T=0;for(let B=0;T<c&&T<k.length;B++){let F=k.slice(T,Math.min(T+m,c));if(!F.length)break;let U=Math.max(0,c-A.size),W=await t.run(`crawl-batch-${B}`,async()=>{let ee=await ve(a,()=>Eb(F,{kernelApiKey:i,concurrency:l,urlsPerBrowser:d,forceBrowserRender:p,captureRenderedDom:u})),ue=ee.map(ne=>CW({...ne,acquiredHtml:ne.acquiredHtml,mainHtml:ne.mainHtml,inSitemap:v?v.has(x(ne.url)):null},_[ne.url]));if(await wk(r,ue),h)return[];let pe=new Set,he=0,Je=Math.min(U,AW);for(let ne of ee.filter(rs)){let be=ne.discoveryLinks??(ne.outlinks??[]).filter(se=>se.internal).map(se=>se.href);for(let se of be){if(pe.size>=Je)break;if(!se||!E(se)||pe.has(se))continue;let Ne=Buffer.byteLength(se);if(he+Ne>OP)break;pe.add(se),he+=Ne}if(pe.size>=Je||he>=OP)break}return[...pe]});T+=F.length;let K=!1;for(let ee of W){let ue=x(ee),pe=Buffer.byteLength(ue);!A.has(ue)&&A.size<c&&ue.length<=es&&pe<=es&&C+pe<=Zp&&(A.add(ue),C+=pe,k.push(ee),K=!0)}K&&await Ol(r,Math.min(A.size,c))}let I=Array.isArray(n.options.formats)&&n.options.formats.includes("branding")?await t.run("branding",()=>ve(a,()=>Qp(n.startUrl,i).catch(()=>null))):null,L=Array.isArray(n.options.formats)&&n.options.formats.includes("images")?await t.run("image-audit",async()=>{let B=await vk(r,um+1);return DE(B.slice(0,um),{concurrency:12,timeoutMs:12e3,max:um,sampleTruncated:B.length>um})}):null,H=await t.run("finalize",async()=>{let{assembleExtractArtifacts:B}=await import("./extract-bundle-C3N6E6V7.js"),F=await _i(r);if(!F)throw new Error("extract job disappeared before finalization");let U=await ve(a,()=>B(F,{branding:I,imageAudit:L})),W=Ws(F),K=lk(F,W.creditTruncated),ee=K==="failed"?`No pages were extracted successfully (${F.failedUrls} failed, ${F.remainingUrls} remaining).`:K==="partial"?`${F.successfulUrls} pages succeeded; ${F.failedUrls} failed and ${F.remainingUrls} remain.${W.creditTruncated?` Credit availability limited this request from ${W.requestedMaxPages} to ${W.effectiveMaxPages} pages.`:""}`:null,ue=K==="failed"?$r({errorCode:"extraction_failed",retryable:!0,chargeStatus:"refund_pending"}):null;if(!await xk(r,U,K,ee,!1,ue))throw new Error("site extract job state changed before finalization committed");return{artifacts:U,status:K}});return await t.run("settle",async()=>{let B=await _i(r);B&&await Pb(B)}),await DP(r,H.status==="complete"?"succeeded":"failed"),{jobId:r,status:H.status,artifacts:H.artifacts}});import{randomUUID as PW}from"crypto";var UP="image-sources/",NW=10080*60*1e3,LW=900*1e3,jP=20*1024*1024,MW=new Map([["image/jpeg","jpg"],["image/png","png"],["image/webp","webp"],["image/gif","gif"]]);function OW(){return process.env.IMAGE_SOURCE_ARTIFACT_READ_WRITE_TOKEN?.trim()||process.env.PRIVATE_ARTIFACT_READ_WRITE_TOKEN?.trim()||process.env.CONNECTED_DATA_READ_WRITE_TOKEN?.trim()||process.env.CONNECTED_DATA_BLOB_READ_WRITE_TOKEN?.trim()||null}function $P(){return{prefix:UP,artifactTtlMs:NW,downloadTtlMs:LW,token:OW()}}function DW(e){return to(e,UP)}async function FP(e){let t=MW.get(e.contentType);if(!t)throw new Error("image_source_type_unsupported");if(e.content.length===0||e.content.length>jP)throw new Error("image_source_size_invalid");let r=PW().replaceAll("-","");return Vr({policy:$P(),ownerId:e.ownerId,artifactKey:`${r}.${t}`,createdAt:e.createdAt??new Date,filename:e.filename??`mcp-scraper-image-${r.slice(0,12)}.${t}`,contentType:e.contentType,content:e.content})}async function pm(e){return DW(e.artifactId)!==e.ownerId?null:Xn({policy:$P(),artifactId:e.artifactId,maxBytes:jP})}import{createHash as qW,randomBytes as UW}from"crypto";import{homedir as jW}from"os";import{basename as $W,extname as FW,join as BP}from"path";import BW from"p-limit";import{ZipFile as HW}from"yazl";var mm="page-media/",WW=10080*60*1e3,zW=900*1e3,Nb=10*1024*1024,KW=45*1024*1024,GW=15e5,VW=25e5,JW=6,HP=5;function zP(){return process.env.PAGE_MEDIA_ARTIFACT_READ_WRITE_TOKEN?.trim()||process.env.PRIVATE_ARTIFACT_READ_WRITE_TOKEN?.trim()||process.env.CONNECTED_DATA_READ_WRITE_TOKEN?.trim()||process.env.CONNECTED_DATA_BLOB_READ_WRITE_TOKEN?.trim()||null}function KP(){return{prefix:mm,artifactTtlMs:WW,downloadTtlMs:zW,token:zP()}}function YW(e,t){if(e.length>=3&&e[0]===255&&e[1]===216&&e[2]===255)return{mimeType:"image/jpeg",extension:"jpg"};if(e.length>=8&&e[0]===137&&e[1]===80&&e[2]===78&&e[3]===71)return{mimeType:"image/png",extension:"png"};if(e.length>=12&&String.fromCharCode(...e.slice(0,4))==="RIFF"&&String.fromCharCode(...e.slice(8,12))==="WEBP")return{mimeType:"image/webp",extension:"webp"};if(e.length>=6&&["GIF87a","GIF89a"].includes(String.fromCharCode(...e.slice(0,6))))return{mimeType:"image/gif",extension:"gif"};if(e.length>=12&&String.fromCharCode(...e.slice(4,12)).match(/^ftyp(?:avif|avis)$/))return{mimeType:"image/avif",extension:"avif"};let r=Buffer.from(e.slice(0,Math.min(e.length,1024))).toString("utf8").replace(/^\uFEFF/,"").trimStart();return(t==="image/svg+xml"||r.startsWith("<svg")||/^<\?xml[\s\S]*?<svg/i.test(r))&&/<svg[\s>]/i.test(r)?{mimeType:"image/svg+xml",extension:"svg"}:null}function XW(e,t,r){if(r==="image")return YW(e,t);if(e.length>=12&&String.fromCharCode(...e.slice(4,8))==="ftyp")return{mimeType:"video/mp4",extension:"mp4"};if(e.length>=4&&e[0]===26&&e[1]===69&&e[2]===223&&e[3]===163)return{mimeType:"video/webm",extension:"webm"};if(e.length>=4&&String.fromCharCode(...e.slice(0,4))==="OggS")return r==="audio"?{mimeType:"audio/ogg",extension:"ogg"}:{mimeType:"video/ogg",extension:"ogv"};if(r==="audio"&&e.length>=3&&String.fromCharCode(...e.slice(0,3))==="ID3")return{mimeType:"audio/mpeg",extension:"mp3"};if(t&&Yh(t)===r){let n=t.split("/")[1]?.replace("jpeg","jpg").replace("mpeg","mp3").replace("svg+xml","svg");if(n&&/^[a-z0-9]+$/i.test(n))return{mimeType:t,extension:n}}return null}function ZW(e,t){let r=Buffer.from(e);if(t==="image/png"&&r.length>=24)return{width:r.readUInt32BE(16),height:r.readUInt32BE(20)};if(t==="image/gif"&&r.length>=10)return{width:r.readUInt16LE(6),height:r.readUInt16LE(8)};if(t==="image/webp"&&r.length>=30){let n=r.toString("ascii",12,16);if(n==="VP8X")return{width:1+r.readUIntLE(24,3),height:1+r.readUIntLE(27,3)};if(n==="VP8L"&&r[20]===47){let i=r.readUInt32LE(21);return{width:(i&16383)+1,height:(i>>14&16383)+1}}}if(t==="image/jpeg"){let n=2;for(;n+9<r.length;){if(r[n]!==255){n+=1;continue}let i=r[n+1];if(i===216||i===217){n+=2;continue}let o=r.readUInt16BE(n+2);if(o<2||n+o+2>r.length)break;if([192,193,194,195,197,198,199,201,202,203,205,206,207].includes(i))return{height:r.readUInt16BE(n+5),width:r.readUInt16BE(n+7)};n+=o+2}}if(t==="image/svg+xml"){let i=r.toString("utf8",0,Math.min(r.length,65536)).match(/<svg\b[^>]*>/i)?.[0]??"",o=Number(i.match(/\bwidth=["']?([\d.]+)/i)?.[1]??0),s=Number(i.match(/\bheight=["']?([\d.]+)/i)?.[1]??0);if(o>0&&s>0)return{width:Math.round(o),height:Math.round(s)};let a=i.match(/\bviewBox=["']([^"']+)["']/i)?.[1]?.trim().split(/[\s,]+/).map(Number);if(a?.length===4&&a[2]>0&&a[3]>0)return{width:Math.round(a[2]),height:Math.round(a[3])}}return null}async function QW(e){let t=Number(e.headers.get("content-length")??"");if(Number.isFinite(t)&&t>Nb)throw new Error("media_too_large");if(!e.body){let o=Buffer.from(await e.arrayBuffer());if(o.length>Nb)throw new Error("media_too_large");return o}let r=e.body.getReader(),n=[],i=0;try{for(;;){let{done:o,value:s}=await r.read();if(o)break;if(s){if(i+=s.byteLength,i>Nb)throw new Error("media_too_large");n.push(Buffer.from(s))}}}finally{r.releaseLock()}return Buffer.concat(n)}async function ez(e,t){let r=e;for(let n=0;n<=HP;n+=1){let i=await fe(r,{field:"website media URL"});if(i.error||!i.parsed)throw new Error("media_url_rejected");let o=await fetch(i.parsed.href,{redirect:"manual",headers:{Accept:t==="image"?"image/*":`${t}/*`},signal:AbortSignal.timeout(15e3)});if([301,302,303,307,308].includes(o.status)){let d=o.headers.get("location");if(!d||n===HP)throw new Error("media_redirect_rejected");r=new URL(d,i.parsed).href;continue}if(!o.ok)throw new Error(`media_fetch_${o.status}`);let s=await QW(o),a=o.headers.get("content-type")?.split(";")[0].trim().toLowerCase()??null,c=XW(s,a,t);if(!c||Yh(c.mimeType)!==t)throw new Error("media_content_invalid");let l=t==="image"?ZW(s,c.mimeType):null;return{bytes:s,...c,finalUrl:i.parsed.href,width:l?.width??null,height:l?.height??null,sha256:qW("sha256").update(s).digest("hex")}}throw new Error("media_redirect_rejected")}function tz(e){return new Promise((t,r)=>{let n=new HW,i=[];n.outputStream.on("data",o=>i.push(Buffer.from(o))),n.outputStream.once("error",r),n.outputStream.once("end",()=>t(Buffer.concat(i)));for(let o of e)n.addBuffer(o.content,o.path);n.end()})}function WP(e){return e.toLowerCase().replace(/[^a-z0-9]+/g,"-").replace(/^-+|-+$/g,"").slice(0,70)||"website"}function rz(e){try{return $W(new URL(e).pathname).replace(/[^a-zA-Z0-9._-]+/g,"-").replace(/^-+|-+$/g,"").slice(0,80)}catch{return""}}function nz(e){return to(e,mm)}async function GP(e){return nz(e.artifactId)!==e.ownerId?null:Xn({policy:KP(),artifactId:e.artifactId,maxBytes:60*1024*1024})}async function VP(e){let t=[],r=BW(JW),n=new Map,i=0,o=0,s=0;await Promise.all(e.media.assets.map((f,h)=>r(async()=>{try{let y=await ez(f.url,f.type);f.finalUrl=y.finalUrl,f.mimeType=y.mimeType,f.sizeBytes=y.bytes.length,f.sha256=y.sha256,f.width=y.width??f.width,f.height=y.height??f.height;let b=n.get(y.sha256);if(b){f.filename=b,f.duplicateOf=b,f.downloadStatus="downloaded",f.downloadError=null;return}if(i+y.bytes.length>KW)throw new Error("archive_byte_limit_reached");i+=y.bytes.length;let _=rz(y.finalUrl),S=FW(_),x=_?S?_.slice(0,-S.length):_:f.type,E=`media/${String(h+1).padStart(4,"0")}-${WP(x)}.${y.extension}`;n.set(y.sha256,E),t.push({path:E,content:y.bytes}),f.filename=E,f.duplicateOf=null,f.downloadStatus="downloaded",f.downloadError=null,f.type==="image"&&s<e.maxInlineImages&&y.bytes.length<=GW&&o+y.bytes.length<=VW&&(f.inlinePreview={data:y.bytes.toString("base64"),mimeType:y.mimeType},o+=y.bytes.length,s+=1)}catch(y){f.finalUrl=null,f.mimeType=null,f.sizeBytes=null,f.sha256=null,f.savedPath=null,f.downloadStatus="failed",f.downloadError=y instanceof Error?y.message:"media_download_failed",f.inlinePreview=null}})));let a=e.media.assets.filter(f=>f.downloadStatus==="downloaded").length,c=e.media.assets.length-a;if(c>0&&e.media.warnings.push(`${c} media asset(s) could not be downloaded; media.jsonl contains bounded error codes.`),a===0)return{media:e.media,artifact:null};let l=e.media.assets.map(({inlinePreview:f,...h})=>h),d={pageUrl:e.pageUrl,title:e.title,extractedAt:e.extractedAt,completeness:e.media.completeness,exhausted:e.media.exhausted,stopReason:e.media.stopReason,scrollRounds:e.media.scrollRounds,staticFound:e.media.staticFound,renderedFound:e.media.renderedFound,totalFound:e.media.totalFound,retainedAfterFilteringAndVariantCollapse:e.media.assets.length,downloaded:a,failed:c,notes:["The inventory unions static-source and rendered-page discovery.","complete means rendered scrolling reached three stable document-height checks; it does not claim every interactive control was opened.","Media records preserve discovery evidence but do not assert ownership, rights, identity, or publishability."]};t.push({path:"summary.json",content:Buffer.from(JSON.stringify(d,null,2))}),t.push({path:"media.jsonl",content:Buffer.from(l.map(f=>JSON.stringify(f)).join(`
|
|
1631
1631
|
`)+`
|
|
1632
1632
|
`)});let u=await tz(t),p=UW(6).toString("hex"),m=await Vr({policy:KP(),ownerId:e.ownerId,artifactKey:`${p}.zip`,createdAt:new Date(e.extractedAt),filename:`${WP(e.title||new URL(e.pageUrl).hostname)}-website-media.zip`,contentType:"application/zip",content:u}),g=zP()?null:BP(process.env.MCP_SCRAPER_OUTPUT_DIR?.trim()||BP(jW(),"Downloads","mcp-scraper"),"blobs",m.artifactId);return{media:e.media,artifact:{...m,localPath:g}}}import{createHash as iz}from"crypto";var oz=25;function sz(e,t,r){return`scrape-image-${iz("sha256").update(`${e}\0${t}\0${r.sourceKind}\0${r.sourceUrl}\0${r.imageUrl??""}\0${r.imageBase64??""}`).digest("hex")}`}async function JP(e,t,r){let n=t.slice(0,oz);if(n.length===0)return{requested:0,saved:0,skipped:0,assets:[]};let{key:i,error:o}=await Re(e);if(!i)return{requested:n.length,saved:0,skipped:t.length-n.length,assets:[],error:o??"memory unavailable"};let s=r?.trim()||"Library",a=[];for(let c of n){let l=await Y("image_asset_save",{vault:s,...c.imageBase64?{imageBase64:c.imageBase64}:{sourceUrl:c.imageUrl},title:c.title.slice(0,240),sourceRef:{sourceUrl:c.sourceUrl,sourceKind:c.sourceKind,...c.sourceRef??{}},idempotencyKey:sz(e.id,s,c)},i);a.push({sourceUrl:c.imageUrl??c.sourceUrl,sourceKind:c.sourceKind,saved:l.ok,...l.asset?.assetId?{assetId:l.asset.assetId}:{},...l.ok?{}:{error:l.error??"image save failed"}})}return{requested:n.length,saved:a.filter(c=>c.saved).length,skipped:t.length-n.length,assets:a}}import{mkdirSync as az,writeFileSync as cz}from"fs";import{homedir as lz}from"os";import{join as Lb,dirname as dz}from"path";function YP(e){return Buffer.isBuffer(e)?e.length:Buffer.byteLength(e)}var Mb=class{constructor(t){this.baseDir=t}baseDir;kind="local";async put(t,r,n="application/octet-stream"){let i=Lb(this.baseDir,"blobs",t);return az(dz(i),{recursive:!0}),cz(i,r),{key:t,url:`file://${i}`,bytes:YP(r),contentType:n}}async get(t){let r=Lb(this.baseDir,"blobs",t);try{let{readFileSync:n}=await import("fs");return n(r)}catch{return null}}};function uz(){return process.env.MCP_SCRAPER_OUTPUT_DIR?.trim()||Lb(lz(),"Downloads","mcp-scraper")}var gm=null,Ob=class{kind="unconfigured";async put(){throw new Error("Blob storage is not configured on this deployment: set BLOB_READ_WRITE_TOKEN. Refusing to write to a local path that no request can read back.")}async get(){return null}};function pz(e=process.env){return!!(e.VERCEL||e.AWS_LAMBDA_FUNCTION_NAME)}function mz(e=process.env){let t=e.BLOB_READ_WRITE_TOKEN?.trim();return t?new Zl(t):pz(e)?new Ob:new Mb(uz())}function gc(){return gm||(gm=mz(),gm)}var Zl=class{constructor(t){this.token=t}token;kind="vercel-blob";async put(t,r,n="application/octet-stream"){let{put:i}=await import("@vercel/blob"),o=Buffer.isBuffer(r)?r:Buffer.from(r),s=await i(t,o,{access:"public",token:this.token,contentType:n,addRandomSuffix:!0});return{key:t,url:s.url,bytes:YP(r),contentType:n}}async get(t){try{let{list:r}=await import("@vercel/blob"),i=(await r({prefix:t,token:this.token,limit:1})).blobs.find(s=>s.pathname===t||s.pathname.startsWith(t));if(!i)return null;let o=await fetch(i.url);return o.ok?Buffer.from(await o.arrayBuffer()):null}catch{return null}}};var XP=5e6,fm="scrape-fallback/",Db=1440*60*1e3;async function qb(e,t){try{let r=(t.content??"").trim();if(!r)return{deposited:!1,error:"empty content"};let n=r.length>XP?r.slice(0,XP):r,{key:i,error:o}=await Re(e);if(!i)return{deposited:!1,error:o??"memory unavailable"};let s=t.vault?.trim()||"Library",a=(t.title?.trim()||t.source).slice(0,200),c=await Y("libraryIngestTool",{title:a,content:n,source:t.source,vault:s,...t.capturedAt?{capturedAt:t.capturedAt}:{},...t.summary?{summary:t.summary}:{}},i);return c.ok?{deposited:!0,vault:s,noteId:c.noteId,path:c.path,chunks:c.indexed}:{deposited:!1,vault:s,error:c.error??"ingest failed"}}catch(r){return{deposited:!1,error:r?.message??"deposit failed"}}}async function gz(e){try{let t=(e.content??"").trim();if(!t)return null;let r=(Ul(e.title||e.source)||"scrape").slice(0,60),n=`${fm}${r}.md`,i=`# ${e.title}
|
|
1633
1633
|
Source: ${e.source}
|
|
@@ -4881,7 +4881,7 @@ ${d.path}`,d.title.trim()||d.path.split("/").pop()||d.path,WB(d.vault),!1):i(`gh
|
|
|
4881
4881
|
WHERE vault = ANY($1)
|
|
4882
4882
|
ORDER BY updated_at DESC`,[o])).flatMap(c=>{let l=i.get(c.vault);return l?[{vault:l,path:c.path,title:c.title,content:c.content,updatedAt:c.updated_at,props:c.props&&typeof c.props=="object"?c.props:{}}]:[]});return Ux(a)}var zB=1;async function Wx(e,t){let r=new Date,n=`${r.getUTCFullYear()}-${String(r.getUTCMonth()+1).padStart(2,"0")}`,i=await Ps().query(`SELECT source, COALESCE(SUM(cost_usd), 0)::float8 AS total
|
|
4883
4883
|
FROM mem_usage_ledger WHERE identity = $1 AND period = $2 GROUP BY source`,[e,n]),o={},s=0;for(let a of i){let c=Number(a.total);o[a.source]=c,s+=c}return{ok:!0,period:n,plan:t,costUsd:s,bySource:o,freeCapUsd:zB,freeCapReached:t==="free"&&s>=zB,unlimited:GB(t)}}import{createHash as Iae}from"crypto";var Rae="https://mcp-scraper-scheduler.vercel.app",Uo="resend",Cae="https://cdn.resend.com/brand/resend-icon-black.svg",Tae="https://resend.com/docs/mcp-server",XB=["connect-to-editor","create-webhook","create-api-key","list-api-keys","remove-api-key","list-oauth-grants","revoke-oauth-grant"],Pae=["list-emails","get-email","list-received-emails","get-received-email","list-logs","get-log","list-contacts","get-contact","list-broadcasts","get-broadcast","list-templates","get-template"],zr=class extends Error{status;constructor(t,r=502){super(t),this.name="ResendControlError",this.status=r}};function Nae(){return(process.env.SCHEDULE_INTEGRATIONS_CONTROL_URL?.trim()||process.env.NANGO_CONTROL_URL?.trim()||Rae).replace(/\/$/,"")}function Lae(){let e=process.env.SCHEDULE_INTEGRATIONS_SECRET?.trim();if(!e)throw new zr("Resend connections are not configured.",503);return e}function yr(e){return!!e&&typeof e=="object"&&!Array.isArray(e)}function Ha(e){return yr(e)&&yr(e.data)?e.data:e}function wl(e,t=300){if(typeof e!="string")return null;let r=e.trim();return r?r.slice(0,t):null}function Tt(e,t,r=300){for(let n of t){let i=wl(e[n],r);if(i)return i}return null}function Ba(e){return Array.isArray(e)?[...new Set(e.map(t=>wl(t,200)).filter(t=>!!t))].slice(0,200):[]}function Mae(e,t=50,r=500){return Array.isArray(e)?[...new Set(e.map(n=>wl(n,r)).filter(n=>!!n))].slice(0,t):[]}function zx(e){let t=wl(e,2e3);if(!t)return null;try{let r=new URL(t);return r.protocol==="https:"?r.toString():null}catch{return null}}function Oae(e){let t=zx(e);if(!t)return null;let r=new URL(t);return r.hostname.toLowerCase()==="api.resend.com"&&r.pathname==="/oauth/authorize"?t:null}function ZB(e,t){let r=Ha(e);if(Array.isArray(r))return r;if(!yr(r))return[];for(let n of t)if(Array.isArray(r[n]))return r[n];return[]}async function yi(e,t,r=2e4){let n=await fetch(`${Nae()}${e}`,{...t,headers:{accept:"application/json",authorization:`Bearer ${Lae()}`,...t?.body?{"content-type":"application/json"}:{},...t?.headers??{}},signal:AbortSignal.timeout(r)}).catch(s=>{throw new zr(`Resend connection control is unavailable: ${s instanceof Error?s.message:"network error"}`)}),i=await n.text(),o={};try{o=i?JSON.parse(i):{}}catch{o={}}if(!n.ok){let s=yr(o)?wl(o.error,300):null;throw new zr(s||`Resend connection control failed (${n.status}).`)}return o}async function QB(){let e=await yi("/api/internal/resend/catalog"),t=ZB(e,["providers","catalog","integrations","services"]),r=[];for(let n of t){if(!yr(n)||Tt(n,["providerConfigKey","provider_config_key","id"])!==Uo)continue;let o=Ba(n.safeDefaultAllowedTools??n.safe_default_allowed_tools??n.readTools??n.read_tools),s=Ba(n.actionTools??n.action_tools??n.writeTools??n.write_tools),a=Ba(n.adminBlockedTools??n.admin_blocked_tools);r.push({providerConfigKey:Uo,provider:Uo,label:Tt(n,["label","displayName","display_name","name"])||"Resend",description:Tt(n,["description"],500)||"Connect Resend through its official remote MCP to read email operations and run explicitly enabled delivery workflows.",logoUrl:zx(n.logoUrl??n.logo_url??n.logo)||Cae,docsUrl:zx(n.docsUrl??n.docs_url??n.docs)||Tae,authMode:"REMOTE_MCP_OAUTH2",categories:["Email","Developer tools"],safeDefaultAllowedTools:o,actionTools:s,adminBlockedTools:a.length>0?a:[...XB],connectionSyncSupported:!0,connectionSyncRequiredTools:[...Pae],platformSetupStatus:Tt(n,["platformSetupStatus","platform_setup_status","setupStatus","setup_status"],100)||"ready",appReviewStatus:null,appReviewNote:"Secret-returning webhook creation, API-key, OAuth-grant, and raw editor-session access is permanently blocked from agents and schedules.",transport:"remote_mcp"})}return r}async function bh(e){let t=new URLSearchParams({identity:e}).toString(),r=await yi(`/api/internal/resend/connections?${t}`),n=ZB(r,["connections"]),i=[];for(let o of n){if(!yr(o))continue;let s=Tt(o,["connectionId","connection_id","id"]),a=Tt(o,["providerConfigKey","provider_config_key","provider"])||Uo;if(!s||a!==Uo)continue;let c=(Tt(o,["status"],64)||"needs_reauth").toLowerCase(),l=["pending","pending_oauth","authorizing","authorization_pending"].includes(c),d=["active","connected","ready"].includes(c),u=o.reconnectRequired===!0||o.reconnect_required===!0||!l&&!d,p=Tt(o,["providerAccountId","provider_account_id"],512),m=Tt(o,["providerAccountEmail","provider_account_email"],320)?.toLowerCase()??null,g=Tt(o,["providerAccountName","provider_account_name"],300),f=Tt(o,["label","accountLabel","account_label","displayName","display_name"]),h=f?.toLowerCase()===e.trim().toLowerCase()?null:f,y=Tt(o,["providerIdentityStatus","provider_identity_status"],32),b=y==="verified"||y==="unavailable"?y:p||m||g?"verified":l?"pending":"unavailable";i.push({connectionId:s,providerConfigKey:Uo,provider:Uo,label:m||g||h||"Resend account",providerAccountId:p,providerAccountEmail:m,providerAccountName:g,providerIdentityStatus:b,status:l?"pending_oauth":u?"needs_reauth":"connected",reconnectRequired:u,actionsEnabled:o.actionsEnabled===!0||o.actions_enabled===!0,readTools:Ba(o.readTools??o.read_tools),actionTools:Ba(o.actionTools??o.action_tools??o.writeTools??o.write_tools),adminBlockedTools:Ba(o.adminBlockedTools??o.admin_blocked_tools).length>0?Ba(o.adminBlockedTools??o.admin_blocked_tools):[...XB],mcpEndpoint:null,schemaDiscovery:"compatibility_describe",toolRevision:Tt(o,["toolRevision","tool_revision"],200),vaultName:Tt(o,["vaultName","vault_name"],100),tableName:Tt(o,["tableName","table_name"],100),createdAt:Tt(o,["createdAt","created_at"],100),updatedAt:Tt(o,["updatedAt","updated_at"],100),transport:"remote_mcp"})}return i}function eH(e){let t=Ha(e);if(!yr(t))throw new zr("The Resend connection service returned an invalid connect session.");let r=Oae(t.authorizationUrl??t.authorization_url??t.connectLink??t.connect_link);if(!r)throw new zr("The Resend connection service did not return a trusted authorization link.");return{connectLink:r,sessionToken:null,expiresAt:Tt(t,["expiresAt","expires_at"],100)}}async function tH(e,t){let r=await yi("/api/internal/resend/connect-session",{method:"POST",body:JSON.stringify({identity:e,email:e,redirectUri:t})});return eH(r)}async function rH(e,t,r){let n=await yi("/api/internal/resend/reconnect-session",{method:"POST",body:JSON.stringify({identity:e,email:e,connectionId:t,redirectUri:r})});return eH(n)}async function nH(e){await yi("/api/internal/resend/oauth/callback",{method:"POST",body:JSON.stringify(e)})}async function Kx(e,t){await yi("/api/internal/resend/connections",{method:"DELETE",body:JSON.stringify({identity:e,connectionId:t})})}async function iH(e,t,r){let n=await yi("/api/internal/resend/connections/actions",{method:"PUT",body:JSON.stringify({identity:e,connectionId:t,enabled:r})}),i=Ha(n);return!yr(i)||!yr(i.connection)?r:i.connection.actionsEnabled===!0||i.connection.actions_enabled===!0}async function Gx(e,t,r,n){let i=await yi("/api/internal/resend/read",{method:"POST",body:JSON.stringify({identity:e,connectionId:t,tool:r,input:n??{}})}),o=Ha(i);return yr(o)?o.result??o:o}async function oH(e,t,r,n,i){let o=`main-resend-action:${Iae("sha256").update(e).update("\0").update(i.trim()).digest("hex")}`,s=await yi("/api/internal/resend/actions/call",{method:"POST",headers:{"x-request-id":o},body:JSON.stringify({identity:e,connectionId:t,tool:r,input:n})}),a=Ha(s);return yr(a)?a.result??a:a}async function sH(e,t,r){let n=await yi("/api/internal/resend/describe",{method:"POST",body:JSON.stringify({identity:e,connectionId:t,tool:r})}),i=Ha(n),o=yr(i)&&yr(i.tool)?i.tool:i;if(!yr(o))throw new zr("Resend returned an invalid tool description.");let s=Tt(o,["name"],200),a=Tt(o,["classification","kind"],20),c=o.inputSchema??o.input_schema;if(!s||s!==r||a!=="read"&&a!=="action"||!yr(c))throw new zr("Resend returned an invalid tool description.");return{name:s,title:Tt(o,["title"],300),description:Tt(o,["description"],2e3),inputSchema:c,classification:a}}async function aH(e,t){let r=await yi("/api/internal/resend/export-page",{method:"POST",body:JSON.stringify({identity:e,...t})},9e4),n=Ha(r);if(!yr(n)||n.ok!==!0||n.providerConfigKey!==Uo)throw new zr("Resend export returned an invalid page response.");let i=wl(n.dataset,64);if(!i||i==="auto"||!Na.includes(i))throw new zr("Resend export returned invalid dataset metadata.");if(!Array.isArray(n.records))throw new zr("Resend export returned invalid records.");if(n.nextCursor!==void 0&&n.nextCursor!==null&&typeof n.nextCursor!="string")throw new zr("Resend export returned an invalid continuation cursor.");let o=yr(n.counts)?n.counts:{},s=a=>typeof a=="number"&&Number.isFinite(a)&&a>=0?a:0;return{providerConfigKey:Uo,dataset:i,records:n.records,nextCursor:typeof n.nextCursor=="string"?n.nextCursor:null,complete:n.complete===!0,counts:{listed:s(o.listed),exported:s(o.exported??o.returned),failed:s(o.failed)},warnings:Mae(n.warnings),untrustedContent:!0}}var vl=process.env.NODE_ENV==="production"||process.env.VERCEL==="1",$ae=vl,hH={httpOnly:!0,secure:vl,path:"/",maxAge:3600*24*30,sameSite:"Strict"};function Fae(){let e=new Set(["https://mcpscraper.dev","https://www.mcpscraper.dev"]);for(let t of(process.env.ALLOWED_ORIGINS??process.env.APP_ORIGIN??"").split(",")){let r=t.trim();r&&e.add(r)}return $ae||(e.add("http://localhost:5173"),e.add("http://localhost:3000"),e.add("http://localhost:63316")),e}function _n(){return process.env.APP_ORIGIN?.trim()||"https://mcpscraper.dev"}function cH(e){return e.req.header("cf-connecting-ip")||e.req.header("x-real-ip")||e.req.header("x-forwarded-for")?.split(",")[0]?.trim()||"unknown"}function xh(e,t){let r=t?.trim().toLowerCase();return r?`${cH(e)}:${r}`:cH(e)}async function Sh(e,t,r,n,i){let o=await bi(t,r,n,i);return o.allowed?null:e.json({error:"Too many attempts. Try again shortly."},429,{"Retry-After":String(o.resetSeconds)})}var cn=ju(async(e,t)=>{let r=e.req.header("origin");return r?Fae().has(r)?t():e.json({error:"Origin not allowed"},403):t()}),re=ju(async(e,t)=>{let r=e.req.header("x-api-key");if(!r)return e.json({error:"Missing API key"},401);let n=await Bo(r);if(!n)return e.json({error:"Invalid or inactive API key"},401);let i=await fh(n),o=await Fr(i.id,i.balance_mc);e.set("user",{...i,balance_mc:o});let s=we();return s&&s.userId==null&&(s.userId=Number(i.id)),t()}),jo=ju(async(e,t)=>{let r=gH(e,"session");if(!r)return e.json({error:"Not authenticated"},401);let n=ql(r);if(!n)return e.json({error:"Session expired"},401);let i=await tt(n);if(!i)return e.json({error:"User not found"},401);let o=await fh(i),s=await Fr(o.id,o.balance_mc);return e.set("sessionUser",{...o,balance_mc:s}),t()}),cr=ju(async(e,t)=>e.get("user").subscription_tier?t():e.json({ok:!1,error:"Integrations require an active Starter plan or higher.",code:"subscription_required"},403)),$u=ju(async(e,t)=>{let r=e.get("user");return la(r)?t():e.json(sw(),403)}),M=new Uae,lH=RI(),an="2026-02-25.clover";function Jx(e){return(lH==="development"||lH==="test"||process.env.MCP_SCRAPER_ALLOW_PRIVATE_NETWORK==="1")&&e.headers.get("x-mcp-scraper-local-network")==="1"&&CI(e.url)}function Yx(){let e=process.env.STRIPE_SECRET_KEY?.trim();return e||(process.env.NODE_ENV==="production"||process.env.VERCEL==="1",null)}async function Bae(e){let t=Yx();return t?(await new sn(t,{apiVersion:an}).customers.create({email:e,description:"MCP Scraper free account",metadata:{app:"mcp-scraper",account_status:"free"}})).id:null}M.use("*",async(e,t)=>{if(await t(),e.res.status>=500&&e.res.headers.get("content-type")?.includes("application/json"))try{let r=await e.res.clone().json(),n=new Headers(e.res.headers);e.res=new Response(JSON.stringify(NE(r)),{status:e.res.status,statusText:e.res.statusText,headers:n})}catch{}e.header("Cache-Control","no-store"),e.header("X-Content-Type-Options","nosniff"),e.header("Referrer-Policy","same-origin")});function Hae(e){return e==="/harvest"||e==="/harvest/sync"?"harvest":e==="/diff-page"?"diff_page":e==="/extract-url"?"page_scrape":e==="/extract-site"?"extract_site":e==="/map-urls"||e==="/wayback/snapshots"?"url_map":e==="/youtube/harvest"?"yt_channel":e==="/youtube/transcribe"?"yt_transcription":e==="/tiktok/video-transcribe"?"tiktok_transcribe":e==="/maps/search"?"maps_search":e==="/maps/place"?"maps_place":e.startsWith("/facebook/")?e.endsWith("transcribe")?"fb_transcribe":e.endsWith("/search")?"fb_search":e.endsWith("/reels-inventory")?"fb_reels_inventory":e.endsWith("/page-intel")?"fb_page_intel":"fb_ad":e.startsWith("/instagram/")?"instagram":e.startsWith("/google-ads/")?e.endsWith("/transcribe")?"google_ads_transcribe":e.endsWith("/search")?"google_ads_search":"google_ads_intel":e==="/reddit/thread"||e==="/reddit/trending"||e==="/kernel-reddit"||e.startsWith("/kernel-reddit/")?"reddit_thread":e.startsWith("/trustpilot/")?"trustpilot_reviews":e.startsWith("/g2/")?"g2_reviews":e.startsWith("/api/internal/site-architecture-auditor")?"audit_site":e.startsWith("/api/internal/memory")?"memory_ai":e.startsWith("/agent/")?"browser_session":e.startsWith("/video/")?"video_analysis":null}M.use("*",(e,t)=>{let r=Hae(new URL(e.req.url).pathname);if(!r)return t();let n=e.req.header("x-cost-probe-id")??null;return ve({op:r,probeRunId:n},()=>t())});M.post("/auth/register",cn,async e=>{let{email:t,password:r}=await e.req.json(),n=t?.trim().toLowerCase();if(!n||!r)return e.json({error:"Email and password required"},400);if(r.length<8)return e.json({error:"Password must be at least 8 characters"},400);try{if(await gt(n))return e.json({error:"Email already registered"},409);let o=null;try{o=await Bae(n)}catch(l){console.warn("[auth/register] Stripe customer creation failed; continuing without it (created lazily at checkout):",l instanceof Error?l.message:String(l)),o=null}let s=await Ih(n,void 0,r,o??void 0);if(o)try{await new sn(process.env.STRIPE_SECRET_KEY,{apiVersion:an}).customers.update(o,{metadata:{app:"mcp-scraper",app_user_id:String(s.id),account_status:"free"}})}catch{}let a=await Rh(s.id);(async()=>{try{let l=await tt(s.id);l&&await ay(l)}catch(l){console.warn("[auth/register] memory provision failed (will retry on first write):",l instanceof Error?l.message:String(l))}})();let c=Dl(s.id);return fH(e,"session",c,hH),e.json({email:s.email,api_key:a},201)}catch(i){if((i instanceof Error?i.message:"").includes("UNIQUE"))return e.json({error:"Email already registered"},409);throw i}});M.post("/auth/login",cn,async e=>{let{email:t,password:r}=await e.req.json();if(!t||!r)return e.json({error:"Email and password required"},400);let n=t.trim().toLowerCase(),i=await Sh(e,"auth_login",xh(e,n),10,900);if(i)return i;let o=await gt(n);if(!o||!o.password_hash||!Ju(r,o.password_hash))return e.json({error:"Invalid email or password"},401);let s=Dl(o.id);return fH(e,"session",s,hH),e.json({email:o.email,api_key:o.key_active?o.api_key:null})});M.post("/auth/logout",cn,e=>(vh(e,"session",{path:"/",secure:vl,sameSite:"Strict"}),e.json({ok:!0})));M.post("/account/delete",cn,jo,async e=>{let t=e.get("sessionUser");if(process.env.SCHEDULE_INTEGRATIONS_SECRET?.trim())try{let[n,i]=await Promise.all([bs(t.email,{summaryOnly:!0}),bh(t.email)]);for(let o of n)await _0(t.email,o.connectionId);for(let o of i)await Kx(t.email,o.connectionId)}catch(n){return console.error("[account/delete] failed to remove provider connections",n instanceof Error?n.message:String(n)),e.json({error:"Could not remove every connected account \u2014 please retry before deleting the account."},502)}let r=Yx();if(r&&t.stripe_customer_id){let n=new sn(r,{apiVersion:an}),i=await n.subscriptions.list({customer:t.stripe_customer_id,status:"all",limit:100});for(let o of i.data)if(o.status==="active"||o.status==="trialing"||o.status==="past_due")try{await n.subscriptions.cancel(o.id)}catch(s){return console.error("[account/delete] failed to cancel subscription",o.id,s instanceof Error?s.message:String(s)),e.json({error:"Could not cancel an active subscription \u2014 please try again or contact support."},500)}}return await Bh(t.id,null,0,null),await El(t.id,0),await Fh(t.id,null),await tE(t.id,"free",null),await Th(t.id),vh(e,"session",{path:"/",secure:vl,sameSite:"Strict"}),e.json({ok:!0})});M.post("/auth/forgot-password",cn,async e=>{let{email:t}=await e.req.json(),r=t?.trim().toLowerCase();if(!r)return e.json({ok:!0});let n=await Sh(e,"auth_forgot_ip",xh(e),20,3600);if(n)return n;let i=await Sh(e,"auth_forgot_email",xh(e,r),5,3600);if(i)return i;let o=await gt(r);if(!o)return e.json({ok:!0});let s=await wS(o.id),c=`${process.env.APP_URL??"https://mcpscraper.dev"}/reset-password?token=${s}`;if(process.env.RESEND_API_KEY)try{let d=await new Dae(process.env.RESEND_API_KEY).emails.send({from:"MCP Scraper <noreply@updates.mcpscraper.dev>",to:r,subject:"Reset your MCP Scraper password",html:`<p>Hi,</p><p>Click the link below to reset your password. This link expires in 1 hour.</p><p><a href="${c}">${c}</a></p><p>If you didn't request this, you can ignore this email.</p>`});d.error&&console.error("[auth/forgot-password] Resend rejected the email:",JSON.stringify(d.error))}catch(l){console.error("[auth/forgot-password] Resend send threw:",l instanceof Error?l.message:String(l))}else console.warn("[auth/forgot-password] RESEND_API_KEY not set \u2014 no reset email sent for",r);return e.json({ok:!0})});M.post("/auth/reset-password",cn,async e=>{let{token:t,password:r}=await e.req.json();if(!t||!r)return e.json({error:"Token and password required"},400);if(r.length<8)return e.json({error:"Password must be at least 8 characters"},400);let n=await Sh(e,"auth_reset",xh(e),10,3600);if(n)return n;let i=await _S(t);return i?(await yS(i,r),e.json({ok:!0})):e.json({error:"Invalid or expired reset link"},400)});M.get("/stats/spots",async e=>{let t=await HS();return e.json({taken:t,open:!0})});M.get("/catalog",e=>(e.header("Cache-Control","public, max-age=60"),e.json(DL)));M.get("/rates",e=>e.json(VE()));function dH(e){let t=Ns(e);return{authenticated:!0,id:e.id,email:e.email,api_key:e.key_active?e.api_key:null,created_at:e.created_at,balance_mc:e.balance_mc,balance_credits:e.balance_mc/rt,concurrency_pack_quantity:t.pack_quantity,slots_per_pack:t.slots_per_pack,extra_concurrency_slots:t.extra_concurrency_slots,concurrency_limit:t.effective_limit,concurrency_addon_monthly_usd:t.monthly_amount_usd,has_concurrency_sub:!!e.concurrency_stripe_sub_id,subscription_tier:e.subscription_tier,subscription_plan:eo(e.subscription_tier),subscription_concurrency:e.subscription_concurrency,memory_plan:e.memory_plan??"free",has_memory_sub:!!e.memory_subscription_id}}function yH(e){let t=e==null?1:Number(e);if(!Number.isSafeInteger(t)||t<1)throw new Error("quantity must be a positive whole number");return t}function Eh(e){return typeof e.customer=="string"?e.customer:e.customer.id}async function bH(e,t){let r=await e.subscriptions.list({customer:t,status:"all",limit:100});return PA(r.data)?.id}function Ns(e,t){let r=t??Kb(e),n=r*Qa;return{pack_quantity:r,slots_per_pack:Qa,extra_concurrency_slots:n,effective_limit:ia({concurrency_pack_quantity:r,extra_concurrency_slots:n,subscription_concurrency:e.subscription_concurrency}),monthly_amount_usd:r*JE,billing_cadence:"monthly",receipt_delivery:"email_after_each_successful_payment"}}function wH(){return{custom_text:{submit:{message:`${YE} ${XE}`}},subscription_data:{metadata:{mcp_product:"browser_concurrency_pack",pack_benefit:"2 browsers per $5 monthly pack",billing_cadence:"monthly"}}}}M.get("/me",async e=>{e.header("Cache-Control","no-store");let t=gH(e,"session");if(!t)return e.json({authenticated:!1});let r=ql(t);if(!r)return vh(e,"session",{path:"/",secure:vl,sameSite:"Strict"}),e.json({authenticated:!1});let n=await tt(r);if(!n)return vh(e,"session",{path:"/",secure:vl,sameSite:"Strict"}),e.json({authenticated:!1});let i=dH(n);if(e.req.query("bootstrap")==="1")return e.json(i);let o=await fh(n),[s,a]=await Promise.all([Fr(o.id,o.balance_mc),bS(o.id)]),c={...o,balance_mc:s};return e.json({...dH(c),...a})});M.post("/api-key/rotate",cn,jo,async e=>{let t=e.get("sessionUser"),r=await Rh(t.id);return e.json({api_key:r})});M.delete("/api-key",cn,jo,async e=>{let t=e.get("sessionUser");return await vS(t.id),e.json({ok:!0})});function _H(e,t){t>0||e.memory_provisioned===1||(async()=>{try{if(await rE(e.id))return;await ay(e)}catch(r){console.warn("[memory] background provision failed:",r instanceof Error?r.message:String(r))}})()}M.get("/memory/overview",re,async e=>{let t=e.get("user"),r=io(t),n=hp(t),i=e.req.query("vault")?.trim()||null;try{let[o,s,a]=await Promise.all([yh(r),Fx(r,n),Wx(r,n)]);_H(t,o.length);let c=i&&o.some(u=>u.vault===i)?i:o[0]?.vault??null,l=c?await Bx(r,c):[],d={ok:!0,memory_plan:t.memory_plan??"free",has_memory_sub:!!t.memory_subscription_id,free_quota_gb:ec.free/1e9,plans:Object.values(lp).map(u=>({plan:u.plan,label:u.label,price_id:u.price_id,interval:u.interval,monthly_usd:u.monthly_usd,quota_gb:u.quota_bytes/1e9}))};return e.json({ok:!0,vaults:o,storage:s,activeVault:c,notes:l,usage:a,billing:d})}catch(o){let s=o instanceof Error?o.message:"memory overview failed";return console.error("[memory/overview]",s),e.json({ok:!1,error:s},502)}});M.get("/memory/vaults",re,async e=>{let t=e.get("user");try{let r=await yh(io(t));return _H(t,r.length),e.json({ok:!0,vaults:r})}catch(r){return e.json({ok:!1,error:r instanceof Error?r.message:"could not load vaults"},502)}});M.post("/memory/mcp-call",re,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({}));if(!r.toolName)return e.json({ok:!1,error:"toolName is required"},400);let n=Dm(r.toolName,t);if(n)return e.json(n);if(qm(r.toolName)){let a=await zo(t);if(!a.ok)return e.json({ok:!1,code:"schedule_credentials_unavailable",error:a.error??"Scheduled Actions credentials are unavailable."},503)}let{key:i,error:o}=await Re(t);if(!i)return e.json({ok:!1,error:o??"memory unavailable"},502);let s=await Y(r.toolName,r.args??{},i);return e.json(s)});M.get("/memory/approved-senders",re,async e=>{let{key:t,error:r}=await Re(e.get("user"));if(!t)return e.json({ok:!1,error:r??"memory unavailable"},502);let n=await Y("listApprovedSendersTool",{},t);return e.json(n)});M.post("/memory/approved-senders",re,async e=>{let{key:t,error:r}=await Re(e.get("user"));if(!t)return e.json({ok:!1,error:r??"memory unavailable"},502);let i=(await e.req.json().catch(()=>({}))).senderIdentity?.trim();if(!i)return e.json({ok:!1,error:"senderIdentity is required"},400);let o=await Y("approveSenderTool",{senderIdentity:i},t);return e.json(o)});M.delete("/memory/approved-senders/:identity",re,async e=>{let{key:t,error:r}=await Re(e.get("user"));if(!t)return e.json({ok:!1,error:r??"memory unavailable"},502);let n=e.req.param("identity"),i=await Y("removeApprovedSenderTool",{senderIdentity:n},t);return e.json(i)});M.get("/memory/connect",re,async e=>{let t="https://mcpscraper.dev/mcp",r={mcpServers:{"MCP Scraper":{url:t,headers:{"x-api-key":"${MCP_SCRAPER_API_KEY}"}}}};return e.json({ok:!0,url:t,apiKeyEnvironmentVariable:"MCP_SCRAPER_API_KEY",authModes:["x-api-key","oauth"],config:r,codeCommand:'claude mcp add --transport http mcp-scraper https://mcpscraper.dev/mcp --header "x-api-key: $MCP_SCRAPER_API_KEY"'})});M.post("/memory/vault",re,async e=>{let{key:t,error:r}=await Re(e.get("user"));if(!t)return e.json({error:r??"memory unavailable"},502);let i=(await e.req.json().catch(()=>({}))).vault?.trim();if(!i)return e.json({error:"vault name required"},400);let o=await Y("addVaultTool",{vault:i},t);return e.json(o,o.ok?200:502)});M.get("/memory/notes",re,async e=>{let t=e.get("user"),r=io(t);try{let n=e.req.query("vault")?.trim()||null;if(n||(n=(await yh(r))[0]?.vault??null),!n)return e.json({ok:!0,notes:[],vault:null});let i=await Bx(r,n);return e.json({ok:!0,vault:n,notes:i})}catch(n){return e.json({ok:!1,error:n instanceof Error?n.message:"could not load notes"},502)}});M.get("/memory/note",re,async e=>{let t=e.get("user"),r=e.req.query("vault")?.trim(),n=e.req.query("path");if(!n)return e.json({error:"path required"},400);if(!r)return e.json({ok:!1,error:"vault required"},400);try{let i=await Hx(io(t),r,n);return i?e.json({ok:!0,note:i}):e.json({ok:!1,error:"note not found"},404)}catch(i){return e.json({ok:!1,error:i instanceof Error?i.message:"could not load note"},502)}});M.get("/memory/universe",re,async e=>{let t=e.get("user"),r=e.req.query("includeConnections")==="1";try{let n=await YB(io(t),r);return e.json({ok:!0,...n})}catch(n){return e.json({ok:!1,error:n instanceof Error?n.message:"could not build universe"},502)}});M.put("/memory/note",re,async e=>{let{key:t,error:r}=await Re(e.get("user"));if(!t)return e.json({error:r??"memory unavailable"},502);let n=await e.req.json().catch(()=>({}));if(!n.path)return e.json({error:"path required"},400);let i=await Y("putTool",{vault:n.vault,path:n.path,title:n.title,content:n.content},t);return e.json(i,i.ok?200:502)});M.delete("/memory/note",re,async e=>{let{key:t,error:r}=await Re(e.get("user"));if(!t)return e.json({error:r??"memory unavailable"},502);let n=e.req.query("vault"),i=e.req.query("path");if(!i)return e.json({error:"path required"},400);let o=await Y("deleteNoteTool",{vault:n,path:i},t);return e.json(o,o.ok?200:502)});M.get("/memory/note/download",re,async e=>{let t=e.get("user"),r=e.req.query("vault")?.trim(),n=e.req.query("path");if(!n)return e.json({error:"path required"},400);if(!r)return e.json({error:"vault required"},400);let i=await Hx(io(t),r,n);if(!i)return e.json({error:"note not found"},404);let o=(n.split("/").pop()||"note").replace(/"/g,"");return new Response(i.content??"",{status:200,headers:{"Content-Type":"text/markdown; charset=utf-8","Content-Disposition":`attachment; filename="${o.endsWith(".md")?o:o+".md"}"`}})});M.get("/memory/storage",re,async e=>{let t=e.get("user");try{let r=await Fx(io(t),hp(t));return e.json(r)}catch(r){return e.json({ok:!1,error:r instanceof Error?r.message:"could not load storage"},502)}});M.get("/memory/billing",re,async e=>{let t=e.get("user"),r=Object.values(lp).map(n=>({plan:n.plan,label:n.label,price_id:n.price_id,interval:n.interval,monthly_usd:n.monthly_usd,quota_gb:n.quota_bytes/1e9}));return e.json({ok:!0,memory_plan:t.memory_plan??"free",has_memory_sub:!!t.memory_subscription_id,free_quota_gb:ec.free/1e9,plans:r})});M.post("/memory/checkout",re,async e=>{try{let t=e.get("user"),n=(await e.req.json().catch(()=>({}))).priceId?.trim();if(!n||!lp[n])return e.json({error:"Invalid priceId \u2014 must be a memory plan price."},400);if(Lk(t.subscription_tier))return e.json({error:"Your current plan already includes Memory Pro \u2014 there is nothing to buy here.",error_code:"memory_already_included"},409);let i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);let o=new sn(i,{apiVersion:an}),s=t.stripe_customer_id;s||(s=(await o.customers.create({email:t.email})).id,await qs(t.id,s));let a=await o.checkout.sessions.create({mode:"subscription",customer:s,line_items:[{price:n,quantity:1}],client_reference_id:String(t.id),metadata:{product:"mcp-memory",userId:String(t.id)},subscription_data:{metadata:{product:"mcp-memory",userId:String(t.id)}},automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},success_url:`${_n()}/memory?billing=success`,cancel_url:`${_n()}/memory?billing=cancel`});return a.url?e.json({url:a.url}):e.json({error:"Stripe did not return a checkout URL."},502)}catch(t){let r=t instanceof Error?t.message:"Unable to start memory checkout.";return console.error("[memory/checkout]",r),e.json({error:r},500)}});M.post("/memory/portal",re,async e=>{try{let t=e.get("user");if(!t.stripe_customer_id)return e.json({error:"No billing account yet \u2014 subscribe first."},409);let r=process.env.STRIPE_SECRET_KEY?.trim();if(!r)return e.json({error:"Stripe is not configured."},503);let i=await new sn(r,{apiVersion:an}).billingPortal.sessions.create({customer:t.stripe_customer_id,return_url:`${_n()}/memory`});return e.json({url:i.url})}catch(t){let r=t instanceof Error?t.message:"Unable to open billing portal.";return console.error("[memory/portal]",r),e.json({error:r},500)}});M.post("/schedule-checkout",re,async e=>e.json({error_code:"scheduled_actions_now_credit_metered",error:"Scheduled Actions is included and no longer requires a separate subscription.",billingMode:"credits",runBaseCredits:Cl,llmCostMultiplier:Pl},410));function mt(e,t,r){return t instanceof Fi?e.json({ok:!1,error:t.message},400):t instanceof TA?e.json({ok:!1,error:t.message,code:t.code,errorCode:t.code},t.status):t instanceof Ge?e.json({ok:!1,error:t.message,code:t.code,errorCode:t.code,retryable:t.retryable},t.status):t instanceof xp?e.json({ok:!1,error:t.message,code:t.code,errorCode:t.code,retryable:t.retryable},t.status):t instanceof zr?e.json({ok:!1,error:t.message},t.status):(console.error("[schedule-connections]",t instanceof Error?t.message:String(t)),e.json({ok:!1,error:r},502))}function Fu(e){let t=(e.req.header("idempotency-key")??e.req.header("x-idempotency-key"))?.trim();if(!t||t.length<8||t.length>200)throw new Fi("Idempotency-Key is required and must contain 8-200 characters for external actions.");return t}function Wae(e,t){return e.json({ok:!1,stored:!1,errorCode:t.code,retryable:t.retryable,error:t.message,...t.details},t.status)}function pt(e){if(typeof e!="string")return null;let t=e.trim();return t&&t.length<=200?t:null}function zae(e){if(!Array.isArray(e)||e.length>100)return null;let t=[];for(let r of e){if(!r||typeof r!="object"||Array.isArray(r))return null;let n=r;if(typeof n.email!="string"||!n.email.trim()||n.displayName!==void 0&&typeof n.displayName!="string")return null;t.push({email:n.email.trim(),...typeof n.displayName=="string"&&n.displayName.trim()?{displayName:n.displayName.trim()}:{}})}return t}function vH(e,t){return e.filter(r=>r.scheduleActionId===t)}function Xx(){return new URL("/oauth/resend/callback",_n()).toString()}async function Kae(){let[e,t]=await Promise.allSettled([o$(),QB()]);if(e.status==="rejected"&&console.warn("[schedule-connections] Nango catalog unavailable:",e.reason instanceof Error?e.reason.message:"unknown error"),t.status==="rejected"&&console.warn("[schedule-connections] Resend catalog unavailable:",t.reason instanceof Error?t.reason.message:"unknown error"),e.status==="rejected"&&t.status==="rejected")throw e.reason;return[...e.status==="fulfilled"?e.value.filter(r=>!yy(r.providerConfigKey)):[],...t.status==="fulfilled"?t.value:[]]}async function xH(e){let[t,r]=await Promise.allSettled([bs(e,{summaryOnly:!0}),bh(e)]);if(t.status==="rejected"&&console.warn("[schedule-connections] Nango connection listing unavailable:",t.reason instanceof Error?t.reason.message:"unknown error"),r.status==="rejected"&&console.warn("[schedule-connections] Resend connection listing unavailable:",r.reason instanceof Error?r.reason.message:"unknown error"),t.status==="rejected"&&r.status==="rejected")throw t.reason;return[...t.status==="fulfilled"?t.value.filter(n=>!yy(n.providerConfigKey)).map(n=>({...n,transport:"nango",adminBlockedTools:[]})):[],...r.status==="fulfilled"?r.value.map(n=>({...n,toolCapabilities:[...n.readTools.map(i=>({name:i,classification:"read",requiredPermissions:[],requiredFeatures:[],available:!0,blockedReason:null,missingPermissions:[],missingFeatures:[]})),...n.actionTools.map(i=>({name:i,classification:"action",requiredPermissions:[],requiredFeatures:[],available:!0,blockedReason:null,missingPermissions:[],missingFeatures:[]}))],grantedPermissions:[],enabledFeatures:[],permissionVerification:null})):[]]}async function Ls(e,t,r){if(r==="resend")return!0;if(r)return!1;try{return(await bh(e)).some(n=>n.connectionId===t)}catch(n){return console.warn("[schedule-connections] Resend connection resolution unavailable:",n instanceof Error?n.message:"unknown error"),!1}}function Vx(e){let t=e?"Resend connected":"Resend connection not completed";return`<!doctype html><html lang="en"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1"><meta name="referrer" content="no-referrer"><title>${t}</title><style>:root{color-scheme:light dark}*{box-sizing:border-box}body{margin:0;min-height:100vh;display:grid;place-items:center;padding:24px;background:#f5f2ee;color:#26211d;font-family:ui-sans-serif,system-ui,-apple-system,sans-serif}.card{width:min(100%,440px);padding:32px;border:1px solid #d8d0c8;border-radius:16px;background:#fff;box-shadow:0 18px 55px rgba(31,24,18,.1)}.mark{width:38px;height:38px;display:grid;place-items:center;border-radius:10px;background:#26211d;color:#fff;font-size:20px;font-weight:700}h1{margin:18px 0 8px;font-size:24px;line-height:1.15}p{margin:0;color:#71675f;line-height:1.55}a{display:inline-flex;margin-top:20px;color:#b44d36;font-weight:650;text-underline-offset:3px}@media(prefers-color-scheme:dark){body{background:#171411;color:#f4eee8}.card{background:#211d19;border-color:#3b342d}.mark{background:#f4eee8;color:#211d19}p{color:#b9aea5}}</style></head><body><main class="card"><div class="mark" aria-hidden="true">R</div><h1>${t}</h1><p>${e?"Your Resend account is connected privately to MCP Scraper. You can close this tab.":"The connection could not be completed. Close this tab and try again from Integrations."}</p><a href="/dashboard/integrations">Return to Integrations</a></main><script>if(window.opener){try{window.opener.postMessage({type:'mcp-scraper:resend-oauth',ok:${e}},window.location.origin)}catch(e){}}setTimeout(function(){window.close()},1200)</script></body></html>`}M.get("/schedule-status",re,$u,async e=>{try{let t=e.get("user"),{key:r,error:n}=await Re(t);if(!r)return e.json({error:n??"memory unavailable"},503);let i=await Y("getScheduleStatusTool",{},r);return e.json({...i,ok:!0,enabled:!0,billingMode:"credits",runBaseCredits:Cl,llmCostMultiplier:Pl})}catch(t){let r=t instanceof Error?t.message:"Unable to get schedule status.";return console.error("[schedule-status]",r),e.json({error:r},500)}});M.get("/schedule-connection-catalog",re,async e=>{try{return e.json({ok:!0,catalog:await Kae()})}catch(t){return mt(e,t,"Unable to load the service connection catalog.")}});M.get("/schedule-connections",re,async e=>{try{let t=e.get("user"),r=await xH(t.email),n=r.filter(i=>i.transport==="nango"&&i.billingActive).length;return e.json({ok:!0,connections:r,billing:await hy(t.id,n)})}catch(t){return mt(e,t,"Unable to load service connections.")}});M.patch("/schedule-connections/:id/label",re,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({}));if(r.label!==null&&typeof r.label!="string")return e.json({ok:!1,error:"label must be a string or null."},400);let n=typeof r.label=="string"?r.label.normalize("NFKC").trim().replace(/\s+/g," "):null;if(n&&n.length>80)return e.json({ok:!1,error:"label must contain at most 80 characters."},400);try{let i=await kA(t.email,e.req.param("id"),n||null);return e.json({ok:!0,userLabel:i})}catch(i){return i instanceof Error&&i.message==="service_connection_not_found"?e.json({ok:!1,error:"Service connection not found."},404):mt(e,i,"Unable to rename this service connection.")}});M.post("/schedule-connections/billing/reconcile",re,cr,async e=>{let t=e.get("user");try{return e.json({ok:!0,billing:await nc(t,await bs(t.email,{summaryOnly:!0}))})}catch(r){return mt(e,r,"Unable to synchronize connected-account billing.")}});M.post("/schedule-connections/session",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.providerConfigKey);if(!n)return e.json({ok:!1,error:"providerConfigKey is required."},400);try{MA(n),n!=="resend"&&await nc(t,await bs(t.email,{summaryOnly:!0}));let i=n==="resend"?await tH(t.email,Xx()):await a$(t.email,n);return e.json({ok:!0,...i})}catch(i){return mt(e,i,"Unable to start the service connection flow.")}});M.post("/schedule-connections/:id/reconnect-session",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.providerConfigKey);try{let i=await Ls(t.email,e.req.param("id"),n)?await rH(t.email,e.req.param("id"),Xx()):await c$(t.email,e.req.param("id"));return e.json({ok:!0,...i})}catch(i){return mt(e,i,"Unable to start the service reconnection flow.")}});M.delete("/schedule-connections/:id",re,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.providerConfigKey);try{if(await Ls(t.email,e.req.param("id"),n))return await Kx(t.email,e.req.param("id")),e.json({ok:!0,transport:"remote_mcp"});let i=await _0(t.email,e.req.param("id")),o;try{o=await nc(t,await bs(t.email,{summaryOnly:!0}))}catch(s){console.error("[schedule-connections/delete/billing]",s instanceof Error?s.message:String(s))}return e.json({ok:!0,transport:"nango",remainingConnections:i,billing:o??await hy(t.id,i)})}catch(i){return mt(e,i,"Unable to disconnect this service connection.")}});M.post("/schedule-connections/:id/enable-actions",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({}));if(typeof r.enabled!="boolean")return e.json({ok:!1,error:"enabled must be a boolean."},400);let n=pt(r.providerConfigKey);try{let i=await Ls(t.email,e.req.param("id"),n)?await iH(t.email,e.req.param("id"),r.enabled):await u$(t.email,e.req.param("id"),r.enabled);return e.json({ok:!0,actionsEnabled:i})}catch(i){return mt(e,i,"Unable to update this connection's action setting.")}});M.post("/schedule-connections/:id/test",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.providerConfigKey);try{return await Ls(t.email,e.req.param("id"),n)?e.json({ok:!1,code:"connection_test_not_supported",error:"Resend connection testing will be enabled after its credential migration to the main MCP."},409):e.json({ok:!0,...await S0(t.email,e.req.param("id"))})}catch(i){return mt(e,i,"Unable to test this service connection.")}});function Zx(e,t){if(t instanceof hn){let r=t.code==="not_configured"?503:t.code==="replayed_request"?409:401;return e.json({ok:!1,code:t.code,error:"Scheduler integration authorization failed."},r)}return mt(e,t,"The scheduler integration request failed.")}M.post("/api/internal/integrations/:id/describe",async e=>{let t=await e.req.text();try{await vo(e.req.raw,t);let r=JSON.parse(t),n=typeof r.identity=="string"?r.identity.trim():"",i=pt(r.tool);return!n||!i?e.json({ok:!1,code:"invalid_request",error:"identity and tool are required."},400):e.json({ok:!0,tool:await E0(n,e.req.param("id"),i,!0)})}catch(r){return Zx(e,r)}});M.post("/api/internal/integrations/:id/execute",async e=>{let t=await e.req.text();try{let r=await vo(e.req.raw,t),n=JSON.parse(t),i=typeof n.identity=="string"?n.identity.trim():"",o=pt(n.tool),s=n.args&&typeof n.args=="object"&&!Array.isArray(n.args)?n.args:null;if(!i||!o||!s||n.classification!=="read"&&n.classification!=="action")return e.json({ok:!1,code:"invalid_request",error:"identity, tool, classification, and args are required."},400);let a=n.classification==="action"?await Fn(i,e.req.param("id"),s,o,r.requestId):await ws(i,e.req.param("id"),o,s,r.requestId);return e.json({ok:!0,result:a})}catch(r){return Zx(e,r)}});M.post("/api/internal/integrations/:id/test",async e=>{let t=await e.req.text();try{await vo(e.req.raw,t);let r=JSON.parse(t),n=typeof r.identity=="string"?r.identity.trim():"";return n?e.json({ok:!0,...await S0(n,e.req.param("id"))}):e.json({ok:!1,code:"invalid_request",error:"identity is required."},400)}catch(r){return Zx(e,r)}});M.get("/oauth/resend/callback",async e=>{let t=e.req.query("code")?.trim(),r=e.req.query("state")?.trim(),n=e.req.query("error")?.trim();if(e.header("cache-control","no-store"),e.header("content-security-policy","default-src 'none'; style-src 'unsafe-inline'; script-src 'unsafe-inline'; base-uri 'none'; frame-ancestors 'none'"),e.header("referrer-policy","no-referrer"),e.header("x-content-type-options","nosniff"),!r||r.length>4e3||!t&&!n||(t?.length??0)>4e3||(n?.length??0)>200)return e.html(Vx(!1),400);try{return await nH({...t?{code:t}:{},state:r,...n?{error:n}:{},redirectUri:Xx()}),e.html(Vx(!n))}catch(i){return console.warn("[resend-oauth-callback]",i instanceof Error?i.name:"unknown_error"),e.html(Vx(!1),502)}});M.post("/schedule-connections/actions/slack/send-message",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId);if(!n||typeof r.channel!="string"||typeof r.text!="string")return e.json({ok:!1,error:"connectionId, channel, and text are required."},400);try{let i=await Fn(t.email,n,{channel:r.channel,text:r.text},"send-message",Fu(e));return e.json({ok:!0,result:i})}catch(i){return mt(e,i,"Unable to send the Slack message.")}});var Gae=new Sp;M.post("/schedule-connections/actions/gmail/send-message",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId);if(!n||typeof r.to!="string"||typeof r.subject!="string"||typeof r.body!="string")return e.json({ok:!1,error:"connectionId, to, subject, and body are required."},400);try{let i=await Gae.sendMessage({ownerId:qae("sha256").update(t.api_key).digest("hex").slice(0,24),providerIdentity:t.email,connectionId:n,to:r.to,subject:r.subject,body:r.body,idempotencyKey:Fu(e)});return e.json({ok:!0,result:i})}catch(i){return mt(e,i,"Unable to send the email.")}});function uH(e){if(!e||typeof e!="object")return null;let t=e.content;if(!Array.isArray(t))return null;let n=t.find(i=>i&&typeof i=="object"&&i.type==="text")?.text;return typeof n=="string"?n:null}function Vae(e){let t=e?.trim();if(!t)return null;let r=t.match(/^"?([^"<]*?)"?\s*<([^>]+)>$/);if(r)return{email:r[2].trim().toLowerCase(),name:r[1].trim()||null};let n=t.match(/^([^\s<>]+@[^\s<>]+)$/);return n?{email:n[1].trim().toLowerCase(),name:null}:null}M.post("/schedule-connections/actions/gmail/search-contacts",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId),i=typeof r.query=="string"?r.query.trim():"",o=Math.min(150,Math.max(1,Number.isFinite(r.maxMessages)?Number(r.maxMessages):50)),s=typeof r.pageToken=="string"?r.pageToken:void 0;if(!n||!i)return e.json({ok:!1,error:"connectionId and query are required."},400);try{let a=await ws(t.email,n,"list-messages",{q:i,maxResults:o,pageToken:s}),c=uH(a),l=c?JSON.parse(c):{},d=(l.messages??[]).map(g=>g.id),u=[],p=new Map;for(let g of d){let f=await ws(t.email,n,"get-message",{id:g,format:"metadata"}),h=uH(f);if(!h)continue;let y=JSON.parse(h),b=new Map((y.payload?.headers??[]).map(v=>[v.name.toLowerCase(),v.value])),_=b.get("date")??null,S=b.get("from"),x=Vae(S),E=b.get("subject")??null;if(u.push({id:g,date:_,from:S??null,to:b.get("to")??null,subject:E,snippet:y.snippet??null}),x){let v=x.email.split("@")[1]??null,k=p.get(x.email);k?(k.messageCount+=1,k.name=k.name??x.name,E&&k.sampleSubjects.length<5&&!k.sampleSubjects.includes(E)&&k.sampleSubjects.push(E),_&&(!k.lastSeen||_>k.lastSeen)&&(k.lastSeen=_),_&&(!k.firstSeen||_<k.firstSeen)&&(k.firstSeen=_)):p.set(x.email,{email:x.email,name:x.name,domain:v,messageCount:1,firstSeen:_,lastSeen:_,sampleSubjects:E?[E]:[]})}}let m=l.resultSizeEstimate;return e.json({ok:!0,totalMatches:m,messagesFetched:u.length,truncated:typeof m=="number"?m>u.length:!!l.nextPageToken,nextPageToken:l.nextPageToken??null,messages:u,contacts:Array.from(p.values())})}catch(a){return mt(e,a,"Unable to search Gmail contacts.")}});M.post("/schedule-connections/actions/google-calendar/create-event",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId),i=zae(r.attendees);if(!n||typeof r.summary!="string"||typeof r.description!="string"||typeof r.startDateTime!="string"||typeof r.endDateTime!="string"||!i)return e.json({ok:!1,error:"connectionId, summary, description, startDateTime, endDateTime, and attendees are required."},400);let o=typeof r.timeZone=="string"?r.timeZone:void 0;try{let s=await Fn(t.email,n,{calendarId:typeof r.calendarId=="string"?r.calendarId:"primary",summary:r.summary,description:r.description,location:typeof r.location=="string"?r.location:void 0,start:{dateTime:r.startDateTime,timeZone:o},end:{dateTime:r.endDateTime,timeZone:o},attendees:i},"create-event",Fu(e));return e.json({ok:!0,result:s})}catch(s){return mt(e,s,"Unable to create the calendar event.")}});M.post("/schedule-connections/actions/zoom/create-meeting",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId);if(!n||typeof r.topic!="string"||typeof r.startDateTime!="string"||typeof r.agenda!="string")return e.json({ok:!1,error:"connectionId, topic, startDateTime, and agenda are required."},400);try{let i=await Fn(t.email,n,{topic:r.topic,startDateTime:r.startDateTime,durationMinutes:typeof r.durationMinutes=="number"?r.durationMinutes:30,timezone:typeof r.timezone=="string"?r.timezone:void 0,agenda:r.agenda},"create-meeting",Fu(e));return e.json({ok:!0,result:i})}catch(i){return mt(e,i,"Unable to create the Zoom meeting.")}});M.post("/schedule-connections/actions/read",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId),i=pt(r.providerConfigKey),o=pt(r.tool);if(!n||!o)return e.json({ok:!1,error:"connectionId and tool are required."},400);let s=r.args&&typeof r.args=="object"&&!Array.isArray(r.args)?r.args:{};try{let a=await Ls(t.email,n,i)?await Gx(t.email,n,o,s):await ws(t.email,n,o,s);return e.json({ok:!0,result:a})}catch(a){return mt(e,a,"Unable to read from this connection.")}});M.post("/schedule-connections/actions/import-memory",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId),i=pt(r.providerConfigKey),o=pt(r.tool),s=typeof r.vault=="string"?r.vault.trim():"";if(!n||!i||!o||!s||s.length>100)return e.json({ok:!1,stored:!1,errorCode:"invalid_import_request",retryable:!1,error:"connectionId, providerConfigKey, tool, and an existing Memory vault are required."},400);if(r.args!==void 0&&(!r.args||typeof r.args!="object"||Array.isArray(r.args)))return e.json({ok:!1,stored:!1,errorCode:"invalid_import_args",retryable:!1,error:"args must be a JSON object."},400);if(r.title!==void 0&&(typeof r.title!="string"||!r.title.trim()||r.title.length>200))return e.json({ok:!1,stored:!1,errorCode:"invalid_import_title",retryable:!1,error:"title must be a non-empty string of at most 200 characters."},400);try{let a=await DB(t.email,{connectionId:n,providerConfigKey:i,tool:o,args:r.args??{},vault:s,...typeof r.title=="string"?{title:r.title}:{}},{listConnections:xH,prepareMemory:async(c,l)=>{let{key:d,error:u}=await Re(t);if(!d)throw new Ct("memory_unavailable",u||"Memory is unavailable.",503,!0);let p=await Y("listVaultsTool",{},d);if(!p.ok)throw new Ct("memory_unavailable",p.error||"Memory vaults are unavailable.",503,!0);let m=p.vaults?.find(g=>g.handle===l||g.vault===l);if(!m)throw new Ct("memory_vault_not_found","No matching Memory vault belongs to this caller.",404);if(m.kind!=="notes")throw new Ct("memory_vault_not_indexable","Connected-service snapshots can only be imported into an ordinary searchable notes vault.",400);return{key:d,vault:m.handle}},readConnection:async(c,l,d,u)=>l.providerConfigKey==="resend"?Gx(c,l.connectionId,d,u):ws(c,l.connectionId,d,u),uploadMemory:(c,l)=>Y("uploadTool",l,c)});return e.json(a)}catch(a){return a instanceof Ct?Wae(e,a):mt(e,a,"Unable to import this connected-service snapshot into Memory.")}});M.post("/schedule-connections/actions/describe",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId),i=pt(r.providerConfigKey),o=pt(r.tool);if(!n||!o)return e.json({ok:!1,error:"connectionId and tool are required."},400);if(r.fresh!==void 0&&typeof r.fresh!="boolean")return e.json({ok:!1,error:"fresh must be a boolean when provided."},400);try{let s=await Ls(t.email,n,i)?await sH(t.email,n,o):await E0(t.email,n,o,r.fresh);return e.json({ok:!0,tool:s})}catch(s){return mt(e,s,"Unable to describe this connection tool.")}});M.post("/schedule-connections/actions/export",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId);if(!n)return e.json({ok:!1,error:"connectionId is required."},400);let i=typeof r.channelId=="string"?r.channelId.trim():"";if(r.channelId!==void 0&&!i)return e.json({ok:!1,error:"channelId must be a non-empty Slack conversation ID."},400);if(r.includeThreads!==void 0&&typeof r.includeThreads!="boolean")return e.json({ok:!1,error:"includeThreads must be a boolean when provided."},400);if(r.allTime!==void 0&&typeof r.allTime!="boolean")return e.json({ok:!1,error:"allTime must be a boolean when provided."},400);let o=typeof r.dataset=="string"?r.dataset.trim():"auto",s=o==="auto"&&i?"slack_channel_messages":o,a=[...Na];if(!a.includes(s))return e.json({ok:!1,error:`dataset must be one of: ${a.join(", ")}.`},400);let c=r.lastDays===void 0?void 0:Number(r.lastDays);if(c!==void 0&&(!Number.isInteger(c)||c<1||c>90))return e.json({ok:!1,error:"lastDays must be an integer from 1 to 90."},400);let l=r.maxItems===void 0?2e3:Number(r.maxItems);if(!Number.isInteger(l)||l<1||l>5e3)return e.json({ok:!1,error:"maxItems must be an integer from 1 to 5000."},400);if(r.delivery!==void 0&&r.delivery!=="auto"&&r.delivery!=="artifact")return e.json({ok:!1,error:"delivery must be auto or artifact."},400);if(r.cursor!==void 0&&typeof r.cursor!="string")return e.json({ok:!1,error:"cursor must be the opaque continuation returned by a prior export."},400);let d=r.continuation&&typeof r.continuation=="object"&&!Array.isArray(r.continuation)?r.continuation:null;if(r.continuation!==void 0&&!d)return e.json({ok:!1,error:"continuation must be the complete object returned by a prior export."},400);if(r.allTime===!0&&s!=="slack_channel_messages")return e.json({ok:!1,error:"allTime is supported only for Slack channel exports."},400);if(r.allTime===!0&&(r.from!==void 0||r.lastDays!==void 0||d))return e.json({ok:!1,error:"Do not pass allTime with from, lastDays, or continuation."},400);if(s==="slack_channel_messages"&&!i&&!d)return e.json({ok:!1,error:"channelId is required for a Slack channel export."},400);try{let u=await Ls(t.email,n),p=B2({requestedDataset:s,from:r.allTime===!0?"1970-01-01T00:00:00.000Z":typeof r.from=="string"?r.from:void 0,to:typeof r.to=="string"?r.to:void 0,lastDays:c,cursor:typeof r.cursor=="string"?r.cursor:void 0,continuation:d,scope:i?{slack:{channelId:i,includeThreads:r.includeThreads!==!1}}:void 0}),m=await H2({ownerId:Hn(t.api_key).slice(0,24),connectionId:n,dataset:p.dataset,range:p.range,...p.scope?{scope:p.scope}:{},maxItems:l,forceArtifact:r.delivery==="artifact",...p.cursor?{cursor:p.cursor}:{},fetchPage:g=>u?aH(t.email,g):p$(t.email,g),writeArtifact:cy});return e.json(m)}catch(u){return u instanceof Xt?e.json({ok:!1,error:u.message},400):u instanceof Error&&u.message==="connected_data_private_blob_not_configured"?e.json({ok:!1,error:"Private connected-data artifact storage is not configured."},503):mt(e,u,"Unable to export this connected service.")}});M.post("/schedule-connections/actions/export-search-console-table",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({}));try{let n=FB(r),{key:i,error:o}=await Re(t);if(!i)return e.json({ok:!1,error:o??"Memory table access is unavailable."},503);let s=await BB({ownerId:Hn(t.api_key).slice(0,24),...n,queryPage:a=>Y("queryTableTool",a,i),writeArtifact:cy});return e.json(s)}catch(n){return n instanceof Ki?e.json({ok:!1,error:n.message},400):n instanceof Error&&n.message==="connected_data_private_blob_not_configured"?e.json({ok:!1,error:"Private connected-data artifact storage is not configured."},503):(console.error("[search-console-table-export]",n instanceof Error?n.name:"unknown_error"),e.json({ok:!1,error:"Unable to export the persisted Search Console table."},502))}});M.post("/schedule-connections/actions/export-download",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=typeof r.artifactId=="string"?r.artifactId.trim():"";if(!n)return e.json({ok:!1,error:"artifactId is required."},400);try{let i=await Mk({artifactId:n,ownerId:Hn(t.api_key).slice(0,24)});return i?e.json({ok:!0,artifactId:n,...i}):e.json({ok:!1,error:"Artifact not found, expired, or not owned by this caller."},404)}catch(i){return console.error("[connected-data-export-download]",i instanceof Error?i.name:"unknown_error"),e.json({ok:!1,error:"Unable to renew this artifact download."},502)}});M.post("/schedule-connections/actions/call",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=pt(r.connectionId),i=pt(r.providerConfigKey),o=pt(r.tool);if(!n||!o||!r.args||typeof r.args!="object"||Array.isArray(r.args))return e.json({ok:!1,error:"connectionId, tool, and an args object are required."},400);try{let s=Fu(e),a=await Ls(t.email,n,i)?await oH(t.email,n,o,r.args,s):await Fn(t.email,n,r.args,o,s);return e.json({ok:!0,result:a})}catch(s){return mt(e,s,"Unable to run this connection action.")}});async function vn(e,t){let r=new URL(e.req.url);return r.pathname=t,await M.fetch(new Request(r,e.req.raw))}M.get("/integrations",e=>vn(e,"/schedule-connections"));M.get("/integrations/catalog",e=>vn(e,"/schedule-connection-catalog"));M.post("/integrations/session",e=>vn(e,"/schedule-connections/session"));M.post("/integrations/billing/reconcile",e=>vn(e,"/schedule-connections/billing/reconcile"));M.post("/integrations/:id/reconnect-session",e=>vn(e,`/schedule-connections/${encodeURIComponent(e.req.param("id"))}/reconnect-session`));M.post("/integrations/:id/enable-actions",e=>vn(e,`/schedule-connections/${encodeURIComponent(e.req.param("id"))}/enable-actions`));M.post("/integrations/:id/test",e=>vn(e,`/schedule-connections/${encodeURIComponent(e.req.param("id"))}/test`));M.patch("/integrations/:id/label",e=>vn(e,`/schedule-connections/${encodeURIComponent(e.req.param("id"))}/label`));M.delete("/integrations/:id",e=>vn(e,`/schedule-connections/${encodeURIComponent(e.req.param("id"))}`));M.post("/integrations/actions/read",e=>vn(e,"/schedule-connections/actions/read"));M.post("/integrations/actions/describe",e=>vn(e,"/schedule-connections/actions/describe"));M.post("/integrations/actions/call",e=>vn(e,"/schedule-connections/actions/call"));M.post("/integrations/actions/export",e=>vn(e,"/schedule-connections/actions/export"));M.post("/schedule-link",re,$u,async e=>{try{let t=e.get("user"),r=await zo(t);if(!r.ok)return e.json({ok:!1,code:"schedule_credentials_unavailable",error:r.error??"Scheduled Actions credentials are unavailable."},503);let{key:n,error:i}=await Re(t);if(!n)return e.json({error:i??"memory unavailable"},503);let o=await Y("getScheduleLinkTool",{},n);return e.json(o)}catch(t){let r=t instanceof Error?t.message:"Unable to get schedule link.";return console.error("[schedule-link]",r),e.json({error:r},500)}});M.get("/schedule-actions",re,async e=>{let t=e.get("user"),{key:r,error:n}=await Re(t);if(!r)return e.json({ok:!1,error:n??"memory unavailable"},503);let i=await Y("listScheduledActionsTool",{},r);if(!i.ok||!Array.isArray(i.actions))return e.json(i);try{let o=await Promise.all(i.actions.map(async s=>{let a=typeof s.id=="string"?s.id:"",c=s.executionMode==="connection_sync"?"connection_sync":"agent";if(!a)return{...s,executionMode:c,connections:[]};let l=await v0(t.email,a);return{...s,executionMode:c,connections:vH(l,a)}}));return e.json({...i,actions:o})}catch(o){return console.warn("[schedule-actions] connection bindings unavailable:",o instanceof Error?o.message:String(o)),e.json({...i,actions:i.actions.map(s=>({...s,executionMode:s.executionMode==="connection_sync"?"connection_sync":"agent",connections:[]})),connectionBindingsUnavailable:!0})}});M.post("/schedule-actions",re,$u,async e=>{let t=e.get("user"),r=await zo(t);if(!r.ok)return e.json({ok:!1,code:"schedule_credentials_unavailable",error:r.error??"Scheduled Actions credentials are unavailable."},503);let{key:n,error:i}=await Re(t);if(!n)return e.json({ok:!1,error:i??"memory unavailable"},503);let o=await e.req.json().catch(()=>({})),s=o.executionMode??"agent";if(s!=="agent"&&s!=="connection_sync")return e.json({ok:!1,error:'executionMode must be "agent" or "connection_sync".'},400);let a=Object.prototype.hasOwnProperty.call(o,"connections"),c;try{c=w0(o.connections)}catch(p){return mt(e,p,"Invalid service connections.")}if(s==="connection_sync"){let p=y0(c);if(p)return e.json({ok:!1,error:p},400)}let l={...o,executionMode:s};delete l.connections;let d=await Y("createScheduledActionTool",l,n),u={...d,executionMode:s};if(!d.ok||!a)return e.json(u);if(c.length===0)return e.json({...u,connections:[]});if(!d.id)return console.error("[schedule-actions] create returned no id while service connections were requested"),e.json({ok:!1,error:"The scheduled action was created without a usable id; service connections were not attached."},502);try{let p=await x0(t.email,d.id,c);return e.json({...u,connections:p})}catch(p){let m=await Y("deleteScheduledActionTool",{id:d.id},n).catch(()=>({ok:!1}));m.ok||console.error("[schedule-actions] failed to compensate action after connection binding failure",d.id);let g=mt(e,p,"The scheduled action was not created because its service connections could not be attached.");return g.headers.set("x-schedule-create-rolled-back",m.ok?"true":"false"),g}});M.post("/schedule-actions/propose",re,async e=>{let t=e.get("user"),{key:r,error:n}=await Re(t);if(!r)return e.json({ok:!1,error:n??"memory unavailable"},503);let i=await e.req.json().catch(()=>({})),o=await Y("proposeScheduledActionTool",i,r);return e.json(o)});M.get("/schedule-actions/:id/connections",re,async e=>{try{let t=e.get("user"),r=e.req.param("id"),n=await v0(t.email,r);return e.json({ok:!0,connections:vH(n,r)})}catch(t){return mt(e,t,"Unable to load scheduled action service connections.")}});M.put("/schedule-actions/:id/connections",re,cr,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n;try{n=w0(r.connections)}catch(i){return mt(e,i,"Invalid service connections.")}try{let i=e.req.param("id"),{key:o,error:s}=await Re(t);if(!o)return e.json({ok:!1,error:s??"memory unavailable"},503);let a=await Y("listScheduledActionsTool",{},o);if(!a.ok||!Array.isArray(a.actions))return e.json(a,502);let c=a.actions.find(d=>d.id===i);if(!c)return e.json({ok:!1,error:"scheduled action not found"},404);if(c.executionMode==="connection_sync"){let d=y0(n);if(d)return e.json({ok:!1,error:d},400)}let l=await x0(t.email,i,n);return e.json({ok:!0,connections:l})}catch(i){return mt(e,i,"Unable to update scheduled action service connections.")}});M.get("/schedule-actions/:id/runs",re,async e=>{let t=e.get("user"),{key:r,error:n}=await Re(t);if(!r)return e.json({ok:!1,error:n??"memory unavailable"},503);let i=Number(e.req.query("limit")??"50"),o=Number.isFinite(i)?Math.min(Math.max(Math.round(i),1),100):50,s=await Y("listScheduledActionRunsTool",{id:e.req.param("id"),limit:o},r);return e.json(s)});M.put("/schedule-actions/:id",re,$u,async e=>{let t=e.get("user"),{key:r,error:n}=await Re(t);if(!r)return e.json({ok:!1,error:n??"memory unavailable"},503);let i=await e.req.json().catch(()=>({})),o={};for(let a of["description","vault","cadence","timeOfDay","timezone","artifactSelection"])Object.prototype.hasOwnProperty.call(i,a)&&(o[a]=i[a]);if(Object.keys(o).length===0)return e.json({ok:!1,error:"Provide at least one of description, vault, cadence, timeOfDay, timezone, artifactSelection."},400);let s=await Y("updateScheduledActionTool",{id:e.req.param("id"),...o},r);return e.json(s)});M.post("/schedule-actions/:id/pause",re,async e=>{let t=e.get("user"),{key:r,error:n}=await Re(t);if(!r)return e.json({ok:!1,error:n??"memory unavailable"},503);let i=await Y("pauseScheduledActionTool",{id:e.req.param("id")},r);return e.json(i)});M.post("/schedule-actions/:id/resume",re,$u,async e=>{let t=e.get("user"),{key:r,error:n}=await Re(t);if(!r)return e.json({ok:!1,error:n??"memory unavailable"},503);let i=await Y("resumeScheduledActionTool",{id:e.req.param("id")},r);return e.json(i)});M.delete("/schedule-actions/:id",re,async e=>{let t=e.get("user"),{key:r,error:n}=await Re(t);if(!r)return e.json({ok:!1,error:n??"memory unavailable"},503);let i=e.req.param("id"),o=await Y("deleteScheduledActionTool",{id:i},r);if(!o.ok)return e.json(o);try{return await d$(t.email,i),e.json(o)}catch(s){return console.error("[schedule-actions] action deleted but connection binding cleanup failed",i,s instanceof Error?s.message:String(s)),e.json({...o,connectionBindingCleanupPending:!0})}});M.post("/schedule-link/revoke",re,async e=>{try{let t=e.get("user"),{key:r,error:n}=await Re(t);if(!r)return e.json({error:n??"memory unavailable"},503);let i=await Y("revokeScheduleLinkTool",{},r);return e.json(i)}catch(t){let r=t instanceof Error?t.message:"Unable to revoke schedule link.";return console.error("[schedule-link/revoke]",r),e.json({error:r},500)}});M.post("/support/submit",re,async e=>{try{let t=e.get("user"),n=(await e.req.json().catch(()=>({}))).message?.trim();if(!n)return e.json({error:"message is required"},400);if(n.length>8e3)return e.json({error:"message is too long (max 8000 characters)"},400);let i=process.env.SUPPORT_INBOX_MEMORY_KEY?.trim();if(!i)return e.json({error:"support inbox is not configured"},503);let o=new Date().toISOString(),s=`support/${o.replace(/[:.]/g,"-")}-user-${t.id}`,a=`Support request from ${t.email} (user #${t.id})`,c=[`**From:** ${t.email} (user #${t.id})`,`**Submitted:** ${o}`,"","**Message:**",n].join(`
|
|
4884
|
-
`),l=await Y("putTool",{vault:"Issues",path:s,title:a,content:c},i);return l.ok?e.json({ok:!0}):e.json({error:l.error??"failed to submit support request"},502)}catch(t){let r=t instanceof Error?t.message:"Unable to submit support request.";return console.error("[support/submit]",r),e.json({error:r},500)}});M.get("/memory/usage",re,async e=>{let t=e.get("user");try{let r=await Wx(io(t),hp(t));return e.json(r)}catch(r){return e.json({ok:!1,error:r instanceof Error?r.message:"could not load usage"},502)}});var Jae=(()=>{let e=process.env.SYNC_HARVEST_TIMEOUT_MS,t=e===void 0?NaN:Number(e);return Number.isFinite(t)&&t>0?t:null})();function Yae(e){let t=new AbortController,r=n=>{t.signal.aborted||t.abort(n.reason)};for(let n of e){if(n.aborted){r(n);break}n.addEventListener("abort",()=>r(n),{once:!0})}return t.signal}function pH(e){if(!e||typeof e!="object")return 0;let t=e;return typeof t.totalQuestions=="number"?t.totalQuestions:Array.isArray(t.flat)?t.flat.length:0}function wh(e){return D.paa_base+Math.max(1,e)*D.paa}async function SH(e,t){if(t&&await Jb(e.id,t,1800)||Rm(e))return null;let r=ia(e),n=await od(e.id);return n>=r?me({ok:!1,active:n,limit:r,operation:"harvest",retryAfterSeconds:30}):null}M.post("/harvest/internal/resume/:id",async e=>{let t=e.req.header("authorization");if(!process.env.CRON_SECRET||t!==`Bearer ${process.env.CRON_SECRET}`)return e.json({error:"Unauthorized"},401);let r=await Vt(e.req.param("id"));return!r||r.options.executionOwner!=="direct"||r.options.serpOnly===!0?e.json({error:"PAA job not found"},404):r.status!=="pending"?e.json({job_id:r.id,status:r.status},409):(O_(r.id),e.json({job_id:r.id,status:"pending",recovery_dispatched:!0},202))});M.post("/harvest",re,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=gw.safeParse(r);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let i=n.data;if(i.mode==="full"&&!i.serpOnly)return e.json({error:"Full mode requires serpOnly search."},400);if(i.mode==="full"&&i.serpIdentity)return e.json({error:"Full mode cannot use a saved search identity."},400);let o=i.serpIdentity?await Eo(t.id,i.serpIdentity):null;if(i.serpIdentity&&!o)return e.json({error:"SERP identity not found"},404);if(o&&o.status!=="ready")return e.json({error:"SERP identity is not ready"},409);let s=i.callback_url?.trim();if(s){let y=await fe(s,{field:"callback_url",requireHttps:!0});if(y.error)return e.json({error:y.error},400);i.callback_url=y.parsed?.href}let a={query:i.query,location:i.location,depth:Math.min(4,Math.max(1,i.depth??4)),maxQuestions:Math.min(100,Math.max(1,i.maxQuestions??30)),gl:i.gl??"us",hl:i.hl??"en",device:i.device??"desktop",proxyMode:o?"configured":i.proxyMode??Kn,serpIdentity:i.serpIdentity,proxyZip:i.proxyZip,debug:i.debug??!1,serpOnly:i.serpOnly??!1,pages:Math.min(2,Math.max(1,i.pages??1)),mode:i.mode??"light",includeAllSerpFeatures:i.includeAllSerpFeatures??!1,includeLocalPack:i.includeLocalPack??!1,includeForums:i.includeForums??!1,includeVideos:i.includeVideos??!1,includeAiOverview:i.includeAiOverview??!1,includeWhatPeopleSaying:i.includeWhatPeopleSaying??!1},c=e.req.header("idempotency-key")??e.req.header("x-idempotency-key"),l=null,d=null;if(c!=null){let y=c.trim();if(!y||y.length>200)return e.json({error:"Idempotency-Key must contain 1-200 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);l=`${a.serpOnly?"serp":"paa"}-${Hn(`${t.id}\0${y}`).slice(0,24)}`,d=Hn(JSON.stringify({options:a,callbackUrl:i.callback_url??null}));let b=await Vt(l,t.id);if(b)return b.options.requestFingerprint!==d?e.json({error:"Idempotency-Key was already used for a different asynchronous harvest call.",errorCode:"idempotency_conflict",retryable:!1},409):e.json({job_id:b.id,status:b.status,replayed:!0},["done","failed","cancelled"].includes(b.status)?200:202)}let u=await SH(t,e.req.header("x-mcp-scraper-concurrency-lock"));if(u)return e.json(u,429,{"Retry-After":"30"});let p=a.serpOnly?i.serpIdentity?D.serp_headful:Rl(a.pages,"brightdata"):wh(a.maxQuestions),m=l??mH(),g=`${a.serpOnly?"serp-search":"paa-harvest"}:${m}:hold`,f=`${a.serpOnly?"SERP search":"PAA harvest"}: ${a.query}`.slice(0,500),h=await vy({jobId:m,userId:t.id,tool:a.serpOnly?"search_serp":"harvest_paa",normalizedFlags:a,requestId:e.req.header("x-request-id")??e.req.header("x-vercel-id")??null});{let y=await _t(t.id,p,a.serpOnly?N.SERP:N.PAA,f,g);if(!y.ok)return await Si(h,"failed","insufficient_balance"),e.json(ce(y.balance_mc,p),402);await ao(h,g,"hold",p);try{await Ph(m,t.id,a.query,{...a,...a.serpOnly?{}:{executionOwner:"direct",paaRecoveryCount:0},billingHoldMc:p,billingDebitKey:g,...d?{requestFingerprint:d}:{}},i.callback_url)}catch(b){let _=l?await Vt(l,t.id).catch(()=>null):null;if(_&&_.options.requestFingerprint===d)return e.json({job_id:_.id,status:_.status,replayed:!0},["done","failed","cancelled"].includes(_.status)?200:202);throw await ze(t.id,g,0,a.serpOnly?N.SERP_REFUND:N.PAA_REFUND,`${a.serpOnly?"SERP":"PAA"} job creation failed`,a.serpOnly?"serp_search":"paa_harvest").catch(()=>{}),await Si(h,"failed","job_creation_failed"),b}a.serpOnly||O_(m)}if(a.serpOnly&&process.env.CRON_SECRET){let y=new URL(e.req.url);fetch(`${y.origin}/cron/tick`,{headers:{Authorization:`Bearer ${process.env.CRON_SECRET}`}}).catch(()=>{}),await new Promise(b=>setTimeout(b,80))}return e.json({job_id:m,status:"pending"},202)});M.post("/harvest/sync",re,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=gw.safeParse(r);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let i=n.data;if(i.mode==="full"&&!i.serpOnly)return e.json({error:"Full mode requires serpOnly search."},400);if(i.mode==="full"&&i.serpIdentity)return e.json({error:"Full mode cannot use a saved search identity."},400);let o=i.serpIdentity?await Eo(t.id,i.serpIdentity):null;if(i.serpIdentity&&!o)return e.json({error:"SERP identity not found"},404);if(o&&o.status!=="ready")return e.json({error:"SERP identity is not ready"},409);let s={query:i.query,location:i.location,depth:Math.min(4,Math.max(1,i.depth??4)),maxQuestions:Math.min(100,Math.max(1,i.maxQuestions??30)),gl:i.gl??"us",hl:i.hl??"en",device:i.device??"desktop",proxyMode:o?"configured":i.proxyMode??Kn,serpIdentity:i.serpIdentity,proxyZip:i.proxyZip,debug:i.debug??!1,serpOnly:i.serpOnly??!1,pages:Math.min(2,Math.max(1,i.pages??1)),mode:i.mode??"light",recency:i.recency,includeAllSerpFeatures:i.includeAllSerpFeatures??!1,includeLocalPack:i.includeLocalPack??!1,includeForums:i.includeForums??!1,includeVideos:i.includeVideos??!1,includeAiOverview:i.includeAiOverview??!1,includeWhatPeopleSaying:i.includeWhatPeopleSaying??!1},a=e.req.header("idempotency-key")??e.req.header("x-idempotency-key"),c=null,l=Hh(s);if(a!=null){let E=a.trim();if(!E||E.length>200)return e.json({error:"Idempotency-Key must contain 1-200 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);c=`sync_${Hn(`${t.id}\0${E}`).slice(0,24)}`;let v=await Vt(c,t.id);if(v){if((typeof v.options.requestFingerprint=="string"?v.options.requestFingerprint:Hh(v.options))!==l)return e.json({error:"Idempotency-Key was already used for a different harvest_paa/search_serp call.",errorCode:"idempotency_conflict",retryable:!1},409);if(v.status==="done"){let A=await Ho(c,t.id);return e.json({job_id:c,status:"done",result:Al(v.result),attempts:Zi(A),replayed:!0})}if(v.status==="running")return e.json({error:"This request is already running under the supplied Idempotency-Key.",errorCode:"idempotency_in_progress",retryable:!0,job_id:c},409,{"Retry-After":"10"});if(v.status==="failed"||v.status==="cancelled"){let A=await Ho(c,t.id);return e.json({job_id:c,status:v.status,...v.publicError??{},attempts:Zi(A),replayed:!0})}}}let d=await SH(t,e.req.header("x-mcp-scraper-concurrency-lock"));if(d)return e.json(d,429,{"Retry-After":"30"});let u=s.serpOnly&&!i.serpIdentity,p=s.serpOnly?u?Rl(s.pages,"brightdata"):D.serp_headful:wh(s.maxQuestions),m;if(c){let E=await Nh(c,t.id,s.query,{...s,requestFingerprint:l});if(!E.created){if(E.job.status==="done"){let v=await Ho(c,t.id);return e.json({job_id:c,status:"done",result:Al(E.job.result),attempts:Zi(v),replayed:!0})}if(E.job.status==="failed"||E.job.status==="cancelled"){let v=await Ho(c,t.id);return e.json({job_id:c,status:E.job.status,...E.job.publicError??{},attempts:Zi(v),replayed:!0})}return e.json({error:"This request is already running under the supplied Idempotency-Key.",errorCode:"idempotency_in_progress",retryable:!0,job_id:c},409,{"Retry-After":"10"})}m=E.job.id}else m=await xS(t.id,s.query,s);let g=await vy({jobId:m,userId:t.id,tool:s.serpOnly?"search_serp":"harvest_paa",normalizedFlags:s,requestId:e.req.header("x-request-id")??e.req.header("x-vercel-id")??null}),f=`harvest-sync:${m}:charge`,h=await _t(t.id,p,s.serpOnly?N.SERP:N.PAA,s.query,f);if(!h.ok)return await Si(g,"failed","insufficient_balance"),c&&await _r(m,JSON.stringify({error_code:"insufficient_balance"}),ei(xi(new Error("insufficient_balance")),{chargeStatus:"not_charged"})),e.json(ce(h.balance_mc,p),402);await ao(g,f,"hold",p);let y=kp(m,t.id,g),b=Jae??zu(s.maxQuestions,s.serpOnly,s.mode,s.pages).serverMs,_=Yae([...c?[]:[e.req.raw.signal],AbortSignal.timeout(b)]),S={value:null},x=null;try{let E=we(),v=u,k=process.env.APIFY_API_TOKEN?.trim();if(v&&(s.pages===2?!k:!TE()||s.mode!=="full"&&!k))throw new Error("Search providers are not configured");if(v&&!g)throw new Error("Search accounting is unavailable");let A=()=>Wo({...s,kernelApiKey:ae(),...o?{kernelProxyId:o.kernel_proxy_id,kernelProfileName:o.kernel_profile_name,kernelProfileSaveChanges:!1,kernelStealth:!0}:{},headless:!0,format:"json",outputDir:"/tmp/paa-output-api",signal:_,softDeadlineMs:Date.now()+(s.serpOnly?6500:Math.floor(b*.8)),onAttemptEvent:y}),C=await ve({...E,op:s.serpOnly?"serp":"paa",userId:t.id,headlessSentOut:S,operationRunId:g,operationAccountScope:"production-default"},()=>v?PE({query:s.query,pages:s.pages,mode:s.mode,token:k,operationRunId:g,accountScope:"production-default",signal:_}).then(I=>(x=I,I.result)):A());if(v&&!x)throw new Error("Search provider billing receipt is missing");i.serpIdentity&&await Jg(t.id,i.serpIdentity).catch(()=>{}),await Sl(m,C);let T=await Ho(m,t.id);if(f){let I=s.serpOnly?v?Rl(x.deliveredPages,x.deliveryProvider):Il(S.value):wh(pH(C));await ze(t.id,f,I,s.serpOnly?N.SERP_REFUND:N.PAA_REFUND,s.serpOnly?"SERP search settlement":"PAA harvest settlement",s.serpOnly?"serp_search_sync":"paa_harvest_sync"),await ao(g,f,"settlement",I)}else if(s.serpOnly){let I=v?Rl(x.deliveredPages,x.deliveryProvider):Il(S.value),L=p-I;L>0?await ie(t.id,L,N.SERP_REFUND,"headless-mode pricing settle"):L<0&&await oe(t.id,-L,N.SERP,s.query)}else{let I=wh(pH(C)),L=p-I;L>0?await ie(t.id,L,N.PAA_REFUND,"overestimate refund"):L<0&&await oe(t.id,-L,N.PAA,s.query)}return await Si(g,"succeeded"),e.json({job_id:m,status:"done",result:Al(C),attempts:Zi(T)})}catch(E){console.error("[harvest/sync] failed",{jobId:m,error:E instanceof Error?E.message:String(E)});let v=xi(E);await Si(g,v.error_code==="harvest_timeout"?"timed_out":v.terminalStatus,v.error_code);let k=xE(E),A={chargeStatus:"refunded",...k?{details:{retry_guidance:`SERP extraction stopped at the stable ${k} stage. Retry with the same idempotencyKey after correcting the request or transient condition.`}}:{}},C=vp(v,A),T=await Ho(m,t.id);return v.terminalStatus==="cancelled"||e.req.raw.signal.aborted?(await $h(m,so(v),ei(v,{...A,chargeStatus:"refund_pending"})),f?(await ze(t.id,f,0,N.REFUND,"cancelled call",s.serpOnly?"serp_search_sync":"paa_harvest_sync"),await ao(g,f,"refund",p)):await ie(t.id,p,N.REFUND,"cancelled call"),await $h(m,so(v),ei(v,A)),e.json({job_id:m,status:"cancelled",...C,attempts:Zi(T)},v.httpStatus)):(await _r(m,so(v),ei(v,{...A,chargeStatus:"refund_pending"})),f?(await ze(t.id,f,0,N.REFUND,"failed call",s.serpOnly?"serp_search_sync":"paa_harvest_sync"),await ao(g,f,"refund",p)):await ie(t.id,p,N.REFUND,"failed call"),await _r(m,so(v),ei(v,A)),e.json({job_id:m,status:"failed",...C,attempts:Zi(T)},v.httpStatus))}});function EH(e,t,r){return t||(!r||!["failed","cancelled"].includes(e)?null:$r({errorCode:"service_unavailable",retryable:!0}))}function kH(e){let{error:t,publicError:r,...n}=e,i=EH(e.status,r,!!t);return{...n,error:i?.message??null,...i??{}}}M.get("/jobs/:id",re,async e=>{let t=await Vt(e.req.param("id"),e.get("user").id);if(!t)return e.json({error:"Job not found"},404);let r=await Ho(t.id,e.get("user").id),n=t.result&&typeof t.result=="object"?Al(t.result):t.result;return e.json({...kH(t),result:n,attempts:Zi(r)})});M.get("/jobs",re,async e=>e.json((await Lh(e.get("user").id)).map(kH)));M.get("/history",re,async e=>{let t=e.get("user").id,[r,n]=await Promise.all([Lh(t),TS(t,100)]),i=r.map(l=>({id:l.id,ts:l.created_at,query:l.query,location:l.options?.location??"",source:l.options?.jobKind==="extract_url"?"extract_url":N.SERP,status:l.status,result_count:l.result?(l.result.flat?.length??0)+(l.result.organicResults?.length??0):0})),o=n.map(l=>({id:l.id,ts:l.created_at,query:l.query,location:l.location??"",source:l.source,status:l.status,result_count:l.result_count??0,error:l.error??null})),s=new Set(r.filter(l=>l.options?.jobKind==="extract_url"&&l.completed_at).map(l=>`${l.query}\0${l.completed_at}`)),a=o.filter(l=>l.source!=="extract_url"||!s.has(`${l.query}\0${l.ts}`)),c=[...i,...a].sort((l,d)=>String(d.ts).localeCompare(String(l.ts)));return e.json(c.slice(0,100))});M.get("/ledger",re,async e=>e.json(await Qu(e.get("user").id,100)));M.post("/admin/users",bo,async e=>{let{email:t,name:r,password:n}=await e.req.json();if(!t?.trim())return e.json({error:"email is required"},400);try{return e.json(await Ih(t.trim(),r?.trim(),n?.trim()),201)}catch(i){if((i instanceof Error?i.message:"").includes("UNIQUE"))return e.json({error:"Email already registered"},409);throw i}});M.get("/admin/users",bo,async e=>e.json(await Ch()));M.delete("/admin/users/:id",bo,async e=>(await Th(parseInt(e.req.param("id")??"0")),e.json({ok:!0})));M.post("/admin/backfill-signup-credits",bo,async e=>{let t=await Ch();return e.json({processed:t.length,credited:0,skipped:t.length,users_credited:[],retired:!0})});function _l(e){let t=typeof e.requestedDelivery=="string"?e.requestedDelivery:typeof e.delivery=="string"?e.delivery:"auto",r=e.depositToVault===!0?"memory":t,n=typeof e.preserveMedia=="boolean"?e.preserveMedia:e.downloadMedia===!0,i=String(e.url??"");try{i=new URL(i).href}catch{}return{url:i,screenshot:e.screenshot===!0,screenshotDevice:e.screenshotDevice==="mobile"?"mobile":"desktop",extractBranding:e.extractBranding===!0,includeFeaturedImage:e.includeFeaturedImage===!0,mediaTypes:Array.isArray(e.mediaTypes)?e.mediaTypes:["image","video","audio"],maxMediaAssets:typeof e.maxMediaAssets=="number"?e.maxMediaAssets:100,maxInlineImages:typeof e.maxInlineImages=="number"?e.maxInlineImages:3,requestedDelivery:t,delivery:r,preserveMedia:n,vaultName:typeof e.vaultName=="string"?e.vaultName:null,allowLocal:e.allowLocal===!0}}function _h(e,t){let r={jobId:e.id,job_id:e.id,status:e.status,statusTool:"extract_url_status",replayed:t};if(e.status==="done"&&e.result&&typeof e.result=="object")return{...r,result:e.result,error:null,billing:{chargeStatus:"charged",credits:D.page_scrape/rt}};if(e.status==="failed"||e.status==="cancelled"){let n=e.publicError??$r({errorCode:"service_unavailable",retryable:!0});return{...r,result:null,error:n,billing:{chargeStatus:n.charge_status??"unknown",credits:0}}}return{...r,result:null,error:null,billing:{chargeStatus:"charged",credits:D.page_scrape/rt},message:"Single-page extraction is running durably. Poll extract_url_status with this jobId; polling does not start or bill another extraction."}}M.post("/extract-url/start",re,async e=>{let t=await e.req.json().catch(()=>({})),r=Jx(e.req.raw),n=(r?hw:fw).safeParse(t);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let i=n.data,o=r&&"allowLocal"in i&&i.allowLocal===!0;if(i.depositToVault&&!["auto","memory"].includes(i.delivery??"auto"))return e.json({error:'depositToVault conflicts with the requested delivery; use delivery:"memory" instead.'},400);if(i.preserveMedia!==void 0&&i.downloadMedia!==void 0&&i.preserveMedia!==i.downloadMedia)return e.json({error:"preserveMedia conflicts with deprecated downloadMedia."},400);if(o)try{new URL(i.url)}catch{return e.json({error:"Invalid URL"},400)}else{let h=await fe(i.url,{field:"URL"});if(h.error||!h.parsed)return e.json({error:h.error??"Invalid URL"},400)}let a=(e.req.header("idempotency-key")??e.req.header("x-idempotency-key"))?.trim()||mH();if(a.length>200)return e.json({error:"Idempotency-Key must contain 1-200 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);let c=e.get("user"),l=(()=>{try{return new URL(i.url).href}catch{return i.url}})(),d=`exurl-${Hn(`${c.id}\0${a}`).slice(0,24)}`,u=_l(i),p=JSON.stringify(u),m=await Vt(d,c.id);if(m){if(JSON.stringify(_l(m.options))!==p)return e.json({error:"Idempotency-Key was already used for a different extract_url call.",errorCode:"idempotency_conflict",retryable:!1},409);if(m.status!=="pending")return e.json(_h(m,!0),m.status==="running"?202:200)}else{if(!Rm(c)){let y=ia(c),b=await od(c.id);if(b>=y)return e.json(me({ok:!1,active:b,limit:y,operation:"extract_url",retryAfterSeconds:30}),429,{"Retry-After":"30"})}let h=`extract-url:${d}:charge`;try{await Ph(d,c.id,l,{...u,executionOwner:"inngest",jobKind:"extract_url",billingDebitKey:h})}catch(y){if(m=await Vt(d,c.id),!m)throw y;if(JSON.stringify(_l(m.options))!==p)return e.json({error:"Idempotency-Key was already used for a different extract_url call.",errorCode:"idempotency_conflict",retryable:!1},409)}m=await Vt(d,c.id)}if(!m)throw new Error("Durable extraction job disappeared after admission.");let g=typeof m.options.billingDebitKey=="string"?m.options.billingDebitKey:`extract-url:${m.id}:charge`,f=await _t(c.id,D.page_scrape,N.EXTRACT_URL,new URL(l).hostname,g);if(!f.ok){let h=$r({errorCode:"insufficient_balance",retryable:!1,chargeStatus:"not_charged"});return await _r(m.id,JSON.stringify({error_code:"insufficient_balance"}),h),e.json(ce(f.balance_mc,D.page_scrape),402)}try{await at.send({id:m.id,name:"mcp-scraper/extract-url.requested",data:{jobId:m.id}}),await Va(m.id)}catch(h){let y=h instanceof Error?h.message:String(h);return await Va(m.id,y).catch(()=>!1),console.error("[extract-url/start] dispatch pending retry:",y),e.json({..._h(m,!0),error:"Extraction dispatch was not acknowledged; retry with the same Idempotency-Key or poll extract_url_status.",errorCode:"extract_dispatch_pending",retryable:!0},503)}return e.json(_h(m,!1),202)});M.get("/extract-url/status/:id",re,async e=>{let t=await Vt(e.req.param("id"),e.get("user").id);return!t||t.options.jobKind!=="extract_url"?e.json({error:"Extraction job not found"},404):e.json(_h(t,!0))});M.post("/extract-url",re,async e=>{let t=await e.req.json().catch(()=>({})),r=Jx(e.req.raw),n=(r?hw:fw).safeParse(t);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let{url:i,screenshot:o,screenshotDevice:s,extractBranding:a,includeFeaturedImage:c,mediaTypes:l}=n.data,d=r&&"allowLocal"in n.data&&n.data.allowLocal===!0,u=n.data.delivery??"auto";if(n.data.depositToVault&&!["auto","memory"].includes(u))return e.json({error:'depositToVault conflicts with the requested delivery; use delivery:"memory" instead.'},400);if(n.data.preserveMedia!==void 0&&n.data.downloadMedia!==void 0&&n.data.preserveMedia!==n.data.downloadMedia)return e.json({error:"preserveMedia conflicts with deprecated downloadMedia."},400);let p=n.data.depositToVault?"memory":u,m=n.data.preserveMedia??n.data.downloadMedia??!1,g=n.data.maxMediaAssets??100,f=n.data.maxInlineImages??3;if(d)try{new URL(i)}catch{return e.json({error:"Invalid URL"},400)}else{let C=await fe(i,{field:"URL"});if(C.error||!C.parsed)return e.json({error:C.error??"Invalid URL"},400)}let h=(()=>{try{return new URL(i).href}catch{return i}})(),b=ur(h)?.rawReplayUrl??h,_=e.get("user"),S=e.req.header("idempotency-key")??e.req.header("x-idempotency-key"),x=null;if(S!=null){let C=S.trim();if(!C||C.length>200)return e.json({error:"Idempotency-Key must contain 1-200 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);x=`exurl_${Hn(`${_.id}\0${C}`).slice(0,24)}`;let T=_l(n.data),I=JSON.stringify(T),L=await Vt(x,_.id);if(L){if(JSON.stringify(_l(L.options))!==I)return e.json({error:"Idempotency-Key was already used for a different extract_url call.",errorCode:"idempotency_conflict",retryable:!1},409);if(L.status==="done")return e.json({...L.result,replayed:!0});if(L.status==="running")return e.json({error:"This request is already running under the supplied Idempotency-Key.",errorCode:"idempotency_in_progress",retryable:!0,job_id:x},409,{"Retry-After":"10"});if(L.status==="failed"||L.status==="cancelled")return e.json({...L.publicError??{},job_id:x,status:L.status,replayed:!0})}}let E=await ge(_,"extract_url",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:h}});if(!E.ok)return e.json(me(E),429,{"Retry-After":String(E.retryAfterSeconds)});let v=!1,k=null,A=null;try{if(x){let L=await Nh(x,_.id,h,_l(n.data));if(k=L.job.id,!L.created)return L.job.status==="done"?e.json({...L.job.result,replayed:!0}):L.job.status==="failed"||L.job.status==="cancelled"?e.json({...L.job.publicError??{},job_id:L.job.id,status:L.job.status,replayed:!0}):e.json({error:"This request is already running under the supplied Idempotency-Key.",errorCode:"idempotency_in_progress",retryable:!0,job_id:L.job.id},409,{"Retry-After":"10"})}let C=k?`extract-url:${k}:charge`:null,T=C?await _t(_.id,D.page_scrape,N.EXTRACT_URL,new URL(h).hostname,C):await oe(_.id,D.page_scrape,N.EXTRACT_URL,new URL(h).hostname);if(!T.ok)return k&&await _r(k,JSON.stringify({error_code:"insufficient_balance"})),e.json(ce(T.balance_mc,D.page_scrape),402);v=!0;try{A=await sp({userId:_.id,jobId:k,billingDebitKey:C,normalizedFlags:{screenshot:o===!0,extractBranding:a===!0,preserveMedia:m}})}catch(L){console.warn(JSON.stringify({event:"extract_url_cost_start_failed",message:L instanceof Error?L.message:String(L)}))}let I={...await ve({...we(),op:"page_scrape",userId:_.id,operationRunId:A?.runId??null,operationRootAttemptId:A?.rootAttemptId??null,operationAccountScope:"production-default"},()=>hm(_,{canonicalUrl:h,screenshot:o===!0,screenshotDevice:s==="mobile"?"mobile":"desktop",extractBranding:a===!0,includeFeaturedImage:c===!0,mediaTypes:l??["image","video","audio"],maxMediaAssets:g,maxInlineImages:f,requestedDelivery:u,effectiveDelivery:p,preserveMedia:m,vaultName:n.data.vaultName})),...k?{job_id:k}:{}};return A&&await Xa({...A,status:"succeeded"}).catch(()=>{}),k&&await Sl(k,I),e.json(I)}catch(C){A&&await Xa({...A,status:"failed",failureClass:"extract_url_failed"}).catch(()=>{});let T=no(C),I=JSON.stringify({error_code:T.errorCode,http_status:T.httpStatus}),L=Zn(T,{chargeStatus:v?"refund_pending":"not_charged"}),{error:H,...B}=L;k&&await _r(k,I,B),v&&(k?await ze(_.id,`extract-url:${k}:charge`,0,N.EXTRACT_URL,"failed call","extract_url"):await ie(_.id,D.page_scrape,N.EXTRACT_URL,"failed call"));let F=Zn(T,{chargeStatus:v?"refunded":"not_charged"});if(k){let{error:U,...W}=F;await _r(k,I,W)}return await Z({userId:_.id,source:"extract_url",status:"failed",query:h,error:T.errorCode}),e.json({...F,...k?{job_id:k}:{}},iy(T))}finally{await te(E.lockId)}});M.post("/diff-page",re,async e=>{let t=await e.req.json().catch(()=>({})),r=Jx(e.req.raw),n=(r?pM:uM).safeParse(t);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let{url:i,resetBaseline:o}=n.data;if(r&&"allowLocal"in n.data&&n.data.allowLocal===!0)try{new URL(i)}catch{return e.json({error:"Invalid URL"},400)}else{let u=await fe(i,{field:"URL"});if(u.error||!u.parsed)return e.json({error:u.error??"Invalid URL"},400)}let a=(()=>{try{return new URL(i).href}catch{return i}})(),c=e.get("user"),l=await ge(c,"diff_page",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:a}});if(!l.ok)return e.json(me(l),429,{"Retry-After":String(l.retryAfterSeconds)});let d=!1;try{let{ok:u,balance_mc:p}=await oe(c.id,D.diff_page,N.DIFF_PAGE,new URL(a).hostname);if(!u)return e.json(ce(p,D.diff_page),402);d=!0;let m=await RS(c.id,a),g=o?null:m,f=ae(),h=await Qo({url:a,kernelApiKey:f}),y=h.bodyMarkdown??"",b=Hn(y),{value:_,truncated:S}=bB(y),x=new Date().toISOString();await CS(c.id,a,{contentHash:b,content:_,title:h.title??null,contentBytes:Buffer.byteLength(y,"utf8"),truncated:S});let E,v={hunks:[],linesAdded:0,linesRemoved:0,percentChanged:0,hunksTruncated:!1,hunksTruncatedReason:null,totalChangedLineCount:0};return g?g.contentHash===b?E="unchanged":(E="changed",v=wB(g.content,_)):(E="baseline",v.percentChanged=null),await Z({userId:c.id,source:"diff_page",status:"done",query:a,result:{status:E}}),e.json({url:a,title:h.title??null,status:E,isReset:!!o&&!!m,previousCheckedAt:g?.checkedAt??null,currentCheckedAt:x,contentHash:b,previousContentHash:g?.contentHash??null,summary:{linesAdded:v.linesAdded,linesRemoved:v.linesRemoved,percentChanged:v.percentChanged},hunks:v.hunks,contentTruncated:S,hunksTruncated:v.hunksTruncated,hunksTruncatedReason:v.hunksTruncatedReason,totalChangedLineCount:v.totalChangedLineCount})}catch(u){let p=u instanceof Error?u.message:String(u);return d&&await ie(c.id,D.diff_page,N.DIFF_PAGE,"failed call"),await Z({userId:c.id,source:"diff_page",status:"failed",query:a,error:p}),e.json({error:p},500)}finally{await te(l.lockId)}});M.post("/extract-site/read",re,async e=>{let t=yM.safeParse(await e.req.json().catch(()=>({})));if(!t.success){let r=$r({errorCode:"invalid_request",retryable:!1});return e.json({error:r.message,...r},400)}try{let r=await UB({ownerId:String(e.get("user").id),...t.data});if(!r){let n=$r({errorCode:"site_export_not_found",retryable:!1});return e.json({error:n.message,...n},404)}return e.json(r)}catch(r){if(r instanceof Uu){let i=$r({errorCode:"site_export_format_unavailable",retryable:!1});return e.json({error:i.message,...i},409)}console.error("[extract-site/read]",r instanceof Error?r.message:r);let n=$r({errorCode:"site_export_read_failed",retryable:!0});return e.json({error:n.message,...n},500)}});M.post("/extract-site/image",re,async e=>{let t=bM.safeParse(await e.req.json().catch(()=>({})));if(!t.success){let r=$r({errorCode:"invalid_request",retryable:!1});return e.json({error:r.message,...r},400)}try{let r=await jB({ownerId:String(e.get("user").id),...t.data});if(!r){let n=$r({errorCode:"site_export_image_not_found",retryable:!1});return e.json({error:n.message,...n},404)}return e.json({jobId:t.data.jobId,imageId:t.data.imageId,sourcePage:r.artifact.sourcePage,sourceUrl:r.artifact.sourceUrl,mimeType:r.artifact.contentType,bytes:r.bytes.length,sha256:r.artifact.sha256,dataBase64:r.bytes.toString("base64")})}catch(r){console.error("[extract-site/image]",r instanceof Error?r.message:r);let n=$r({errorCode:"site_export_read_failed",retryable:!0});return e.json({error:n.message,...n},500)}});M.post("/archive/read",re,async e=>{let t=await e.req.json().catch(()=>({})),r=mM.safeParse(t);if(!r.success)return e.json({error:"archive_invalid_request",error_code:"archive_invalid_request",retryable:!1,message:r.error.issues[0]?.message??"Invalid request"},400);let{artifactId:n,url:i,path:o,depositToLibrary:s}=r.data,a=e.get("user"),c=await ge(a,"archive_read",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:n?{artifactId:n}:{url:i}});if(!c.ok)return e.json(me(c),429,{"Retry-After":String(c.retryAfterSeconds)});try{let l=null;if(n){let f=n.startsWith(_g)?await hD({artifactId:n,ownerId:String(a.id)}):n.startsWith(mm)?await GP({artifactId:n,ownerId:String(a.id)}):await ok({artifactId:n,ownerId:String(a.id)});if(!f)throw new Ie("archive_not_found","Archive artifact was not found or has expired.",404);l={buffer:f,ref:n}}if(!o){let f=l?await Tx(l.buffer,l.ref,r.data.maxEntries??200):await TB(i,r.data.maxEntries??200);return e.json({mode:"list",...f,...n?{artifactId:n}:{}})}let d=l?await Px(l.buffer,l.ref,o,r.data.offset??0,r.data.maxBytes??5e4):await PB(i,o,r.data.offset??0,r.data.maxBytes??5e4);if(s&&d.fileBytes>Rx)throw new Ie("archive_library_size_limit",`ZIP entry exceeds the ${Math.round(Rx/1024/1024)} MB Library-ingest limit.`);let u=NB(d.archiveUrl,d.path),p=s?await qb(a,{title:LB(d.path),content:d.fullContent,source:u,vault:"Library",capturedAt:MB(d.fullContent),summary:`Source file \`${d.path}\` preserved from ZIP archive ${u.split("#")[0]}.`}):void 0,{fullContent:m,...g}=d;return e.json({mode:"read",...g,...n?{artifactId:n}:{},memory:p})}catch(l){let d=l instanceof Ie?l:new Ie("archive_read_failed",l instanceof Error?l.message:"Failed to read ZIP archive.",500,!0);return e.json({error:d.code,error_code:d.code,retryable:d.retryable,message:d.message},d.status)}finally{await te(c.lockId)}});M.post("/map-urls",re,async e=>{let t=await e.req.json().catch(()=>({})),r=gM.safeParse(t);if(!r.success)return e.json({error:r.error.issues[0]?.message??"Invalid request"},400);let n=r.data,i=await fe(n.url,{field:"URL"});if(i.error||!i.parsed)return e.json({error:i.error??"Invalid URL"},400);let o=i.parsed,s=e.get("user"),a=await ge(s,"map_urls",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:o.href}});if(!a.ok)return e.json(me(a),429,{"Retry-After":String(a.retryAfterSeconds)});let c=!1;try{let{ok:l,balance_mc:d}=await oe(s.id,D.url_map,N.URL_MAP,o.hostname);if(!l)return e.json(ce(d,D.url_map),402);c=!0;let u=await Xs({startUrl:o.href,maxUrls:Math.min(1e4,Math.max(1,n.maxUrls??500)),concurrency:Math.min(20,Math.max(1,n.concurrency??12)),kernelApiKey:n.browserFallback??n.kernelFallback?ae():void 0});return await Z({userId:s.id,source:"map_urls",status:"done",query:o.href,resultCount:Array.isArray(u.urls)?u.urls.length:null,result:u}),e.json(u)}catch(l){let d=l instanceof Error?l.message:String(l);return c&&await ie(s.id,D.url_map,N.URL_MAP_REFUND,"failed call"),await Z({userId:s.id,source:"map_urls",status:"failed",query:o.href,error:d}),e.json({error:d},500)}finally{await te(a.lockId)}});M.post("/wayback/snapshots",re,async e=>{let t=await e.req.json().catch(()=>({})),r=fM.safeParse(t);if(!r.success)return e.json({error:r.error.issues[0]?.message??"Invalid request"},400);let n=r.data,i=await fe(n.url,{field:"URL"});if(i.error||!i.parsed)return e.json({error:i.error??"Invalid URL"},400);for(let c of n.urls??[]){let l=await fe(c,{field:"Wayback selected URL"});if(l.error||!l.parsed)return e.json({error:l.error??"Invalid Wayback selected URL"},400)}let o=e.get("user"),s=await ge(o,"wayback_inventory",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:i.parsed.href}});if(!s.ok)return e.json(me(s),429,{"Retry-After":String(s.retryAfterSeconds)});let a=!1;try{let{ok:c,balance_mc:l}=await oe(o.id,D.url_map,N.URL_MAP,i.parsed.hostname);if(!c)return e.json(ce(l,D.url_map),402);a=!0;let d=await jT({url:i.parsed.href,scope:n.scope,urls:n.urls,from:n.from,to:n.to,successfulHtmlOnly:n.successfulHtmlOnly,maxCaptures:n.maxCaptures,includeCaptures:n.includeCaptures,maxCaptureRows:n.maxCaptureRows});return await Z({userId:o.id,source:"wayback_inventory",status:"done",query:d.url,resultCount:d.totalCaptures,result:{scope:d.scope,from:d.from,to:d.to,totalCaptures:d.totalCaptures,countType:d.countType,uniqueUrls:d.uniqueUrls,uniqueDigests:d.uniqueDigests}}),e.json(d)}catch(c){let l=c instanceof Error?c.message:String(c);return a&&await ie(o.id,D.url_map,N.URL_MAP_REFUND,"failed call"),await Z({userId:o.id,source:"wayback_inventory",status:"failed",query:i.parsed.href,error:l}),e.json({error:l},500)}finally{await te(s.lockId)}});M.post("/extract-site",re,async e=>{let t=await e.req.json().catch(()=>({})),r=hM.safeParse(t);if(!r.success)return e.json({error:r.error.issues[0]?.message??"Invalid request"},400);let n=r.data;if(n.semanticSimilarity&&!process.env.JINA_API_KEY?.trim())return e.json({error:"Semantic site similarity is not configured on this deployment.",errorCode:"semantic_similarity_unconfigured",retryable:!1},503);if(n.preserveMedia!==void 0&&n.downloadImages!==void 0&&n.preserveMedia!==n.downloadImages)return e.json({error:"preserveMedia conflicts with deprecated downloadImages."},400);let i=await fe(n.url,{field:"URL"});if(i.error||!i.parsed)return e.json({error:i.error??"Invalid URL"},400);let o=i.parsed,s=e.get("user"),a=ur(o.href),c=n.wayback?ub(n.wayback):[],l=a?.originalUrl??o.href;if(n.wayback?.urls){let S=new URL(l).hostname.replace(/^www\./,"").toLowerCase();for(let x of n.wayback.urls){let E=await fe(x,{field:"Wayback selected URL"});if(E.error||!E.parsed)return e.json({error:E.error??"Invalid Wayback selected URL"},400);if(E.parsed.hostname.replace(/^www\./,"").toLowerCase()!==S)return e.json({error:"Every selected Wayback URL must belong to the same site as url."},400)}}let d=Math.min(500,ph(n.maxPages)),u=n.wayback?Math.min(1e4,c.length*Math.min(d,n.wayback.urls?.length??d)):ph(n.maxPages),p=n.wayback?l:a?.rawReplayUrl??o.href,m=n.preserveMedia??n.downloadImages??!1;if(!!n.wayback||yB(n)){let S=e.req.header("idempotency-key")??e.req.header("x-idempotency-key");if(S==null)return e.json({error:"Idempotency-Key is required for background site extraction so a lost response can be retried without a second hold.",errorCode:"idempotency_key_required",retryable:!1},428);if(!S.trim()||S.length>500)return e.json({error:"Idempotency-Key must contain 1-500 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);let x=S.trim();e.header("Idempotency-Key",x);let E=`ext_${Hn(`${s.id}\0${x}`).slice(0,24)}`,v=Hn(JSON.stringify({url:o.href,waybackTimestamp:a?.timestamp??null,waybackOriginalUrl:a?.originalUrl??null,waybackTimeline:n.wayback??null,waybackMonths:c,waybackMaxPagesPerSnapshot:n.wayback?d:null,maxPages:u,rotateProxyEvery:n.rotateProxyEvery??10,formats:[...n.formats??[]].sort(),downloadImages:m,renderJavaScript:n.renderJavaScript===!0,captureRenderedDom:n.captureRenderedDom===!0,semanticSimilarity:n.semanticSimilarity===!0,similarityThreshold:n.similarityThreshold??null,similarityMaxPairs:n.similarityMaxPairs??null})),k=await ML({jobId:E,userId:s.id,balanceMc:s.balance_mc,startUrl:p,requestedMaxPages:u,idempotencyKey:x,requestFingerprint:v,concurrency:ia(s),urlsPerBrowser:n.rotateProxyEvery??10,formats:n.formats,downloadImages:m,renderJavaScript:n.renderJavaScript===!0,captureRenderedDom:n.captureRenderedDom===!0,semanticSimilarity:n.semanticSimilarity===!0,similarityThreshold:n.similarityThreshold,similarityMaxPairs:n.similarityMaxPairs,debitKey:`site-extract:${s.id}:${E}:hold`,waybackReplay:a&&!n.wayback?{timestamp:a.timestamp,originalUrl:a.originalUrl,replayUrl:a.replayUrl,rawReplayUrl:a.rawReplayUrl}:void 0,waybackTimeline:n.wayback?{rootUrl:l,timeline:n.wayback,maxPagesPerSnapshot:d}:void 0});if(k.kind==="conflict")return e.json({error:"Idempotency-Key was already used for different site extraction inputs.",errorCode:"idempotency_conflict",retryable:!1},409);if(k.kind==="insufficient")return e.json(ce(k.balanceMc,k.requiredMc),402);let A=k;if(["complete","partial","failed"].includes(A.job.status)){let H=Ws(A.job);return e.json({jobId:A.job.id,status:"pending",jobStatus:A.job.status,statusUrl:`/extract-site/status/${A.job.id}`,duplicate:!0,...H},202)}let C=Number(A.job.options.heldMc),T=typeof A.job.options.debitKey=="string"?A.job.options.debitKey:"";if(!Number.isSafeInteger(C)||C<=0||!T)return e.json({error:"Background crawl billing metadata is invalid.",errorCode:"extract_job_invalid",retryable:!1},500);let{ok:I,balance_mc:L}=await _t(s.id,C,N.EXTRACT_SITE_HOLD,o.hostname,T);if(!I)return await Ek(A.job.id,"Insufficient credits to fund the background crawl."),e.json(ce(L,C),402);try{await at.send({id:A.job.id,name:"mcp-scraper/extract.requested",data:{jobId:A.job.id}})}catch(H){return console.error("[extract-site/enqueue] dispatch pending retry:",H instanceof Error?H.message:String(H)),e.json({error:"Background crawl dispatch was not acknowledged; retry with the same Idempotency-Key or poll the job status.",errorCode:"extract_dispatch_pending",retryable:!0,jobId:A.job.id,statusUrl:`/extract-site/status/${A.job.id}`,...Ws(A.job)},503)}return e.json({jobId:A.job.id,status:"pending",statusUrl:`/extract-site/status/${A.job.id}`,duplicate:!A.created,...Ws(A.job)},202)}let f=Math.floor(s.balance_mc/D.page_scrape);if(f<1)return e.json(ce(s.balance_mc,D.page_scrape),402);let h=Math.min(Math.min(1e4,Math.max(1,n.maxPages??100)),f),y=h*D.page_scrape,b=await ge(s,"extract_site",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:o.href}});if(!b.ok)return e.json(me(b),429,{"Retry-After":String(b.retryAfterSeconds)});let _=!1;try{let{ok:S,balance_mc:x,billingKey:E}=await oe(s.id,y,N.EXTRACT_SITE_HOLD,o.hostname);return S?(_=!0,await pn({tool:"extract_site",op:"extract_site",userId:s.id,billingDebitKey:E,normalizedFlags:{maxPages:h,rotateProxies:n.rotateProxies===!0,browserFallback:!!(n.browserFallback||n.kernelFallback),renderJavaScript:!!n.renderJavaScript,captureRenderedDom:!!n.captureRenderedDom,semanticSimilarity:!!n.semanticSimilarity,waybackReplay:!!a}},async()=>{let v=n.rotateProxies===!0,k=n.formats?.includes("branding")===!0,A=a?await Pt("site_discovery",()=>Yp(a,h)):null,C=await rm({startUrl:p,maxPages:h,seedUrls:A?.map(F=>F.rawReplayUrl),kernelApiKey:v||k||n.browserFallback||n.kernelFallback||n.renderJavaScript||n.captureRenderedDom||n.semanticSimilarity?ae():void 0,formats:n.formats,forceBrowserRender:n.renderJavaScript||n.captureRenderedDom||n.semanticSimilarity,captureRenderedDom:n.captureRenderedDom,...v?{rotateProxyEvery:n.rotateProxyEvery??30,parallelism:ia(s)}:{}});a&&(C.startUrl=a.originalUrl,C.pages=C.pages.map(F=>{let U=ur(F.url);return U?{...F,archivedUrl:F.url,archiveTimestamp:U.timestamp,originalUrl:U.originalUrl,url:U.originalUrl,finalUrl:U.originalUrl}:F}));let T=C.pages?.length??1,L=C.pages.filter(rs).length*D.page_scrape,H=y-L;if(H>0)await ie(s.id,H,N.EXTRACT_SITE_REFUND,"overestimate refund");else if(H<0){let F=await oe(s.id,-H,N.EXTRACT_SITE,o.hostname);F.ok&&await un(F.billingKey,-H)}await Z({userId:s.id,source:"extract_site",status:"done",query:o.href,resultCount:T,result:C});let B=u>h;return e.json({...C,pages:C.pages.map(F=>{if(F.extractionStatus!=="failed")return F;let{failureCode:U,failureReason:W}=Ak(F.failureCode,F.failureReason);return{...F,failureCode:U,failureReason:W}}),requestedMaxPages:u,effectiveMaxPages:h,creditLimited:B,creditTruncated:B&&C.pages.length>=h})})):e.json(ce(x,y),402)}catch(S){let x=no(S);_&&await ie(s.id,y,N.EXTRACT_SITE_REFUND,"failed call"),await Z({userId:s.id,source:"extract_site",status:"failed",query:o.href,error:x.errorCode});let E=Zn(x,{chargeStatus:_?"refunded":"not_charged"});return e.json(E,iy(x))}finally{await te(b.lockId)}});M.get("/extract-site/status/:id",re,async e=>{let t=e.get("user"),r=await _i(e.req.param("id"));if(!r||r.userId!==t.id)return e.json({error:"Job not found"},404);let n=await Promise.all((r.artifacts??[]).map(async s=>{if(!s.key.startsWith(nk))return s;let a=await ik({artifactId:s.key,ownerId:String(t.id)}).catch(()=>null);return a?{...s,url:a.downloadUrl,expiresAt:a.expiresAt,downloadUrlExpiresAt:a.downloadUrlExpiresAt}:{...s,url:""}})),i=Ws(r),o=EH(r.status,r.publicError,!!r.error);return e.json({jobId:r.id,status:r.status,startUrl:r.startUrl,totalUrls:r.totalUrls,doneUrls:r.doneUrls,discovered:r.totalUrls,attempted:r.attemptedUrls,successful:r.successfulUrls,failed:r.failedUrls,remaining:r.remainingUrls,...i,artifacts:n,error:o?.message??null,...o??{},updatedAt:r.updatedAt})});M.post("/billing/checkout",cn,jo,async e=>{try{let t=e.get("sessionUser");if(!nb(t))return e.json({error:"One-time credit packs are available only on the Scale plan.",error_code:"scale_plan_required"},403);let r=await e.req.json().catch(()=>({})),n=wT(r.quantity),i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);let o=new sn(i,{apiVersion:an}),s=t.stripe_customer_id;s||(s=(await o.customers.create({email:t.email})).id,await qs(t.id,s));let a=_T(t.id,n),c=await o.checkout.sessions.create({customer:s,mode:"payment",payment_method_types:["card"],line_items:[{price_data:{currency:"usd",unit_amount:Nl*100,product_data:{name:"MCP Scraper Scale credit pack",description:`${cp.toLocaleString("en-US")} credits per $${Nl} increment`}},quantity:n}],metadata:a,payment_intent_data:{metadata:a,receipt_email:t.email},ui_mode:"embedded",automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},return_url:`${_n()}/billing?credit_pack=1&session_id={CHECKOUT_SESSION_ID}`});return e.json({clientSecret:c.client_secret,quantity:n,amountUsd:n*Nl,credits:n*cp})}catch(t){let r=t instanceof Error?t.message:"Unable to start credit-pack checkout.";return console.error("[billing/checkout]",r),r.startsWith("quantity must be")?e.json({error:r},400):e.json({error:r},500)}});M.post("/billing/concurrency/checkout",cn,jo,async e=>{try{let t=e.get("sessionUser"),r=await e.req.json().catch(()=>({})),n=yH(r.quantity),i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);let o=new sn(i,{apiVersion:an}),s=Ns(t),a=Ns(t,n),c=eo(t.subscription_tier);if(t.concurrency_stripe_sub_id){let u=await o.subscriptions.retrieve(t.concurrency_stripe_sub_id);if(t.stripe_customer_id&&Eh(u)!==t.stripe_customer_id)throw new Error("Concurrency subscription does not belong to this billing account.");let p=u.items.data.find(g=>g.price?.id===Za);if(!p)throw new Error("Concurrency subscription is missing its pack price.");let m=(p.quantity??1)!==n;return m&&await o.subscriptions.update(u.id,{items:[{id:p.id,quantity:n}],proration_behavior:"create_prorations"}),await El(t.id,n),e.json({updated:m,current:s,after:a,base_plan:c})}let l=t.stripe_customer_id;l||(l=(await o.customers.create({email:t.email})).id,await qs(t.id,l));let d=await o.checkout.sessions.create({customer:l,mode:"subscription",line_items:[{price:Za,quantity:n}],...wH(),ui_mode:"embedded",automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},return_url:`${_n()}/billing?slot_added=1`});return e.json({clientSecret:d.client_secret,current:s,after:a,base_plan:c})}catch(t){let r=t instanceof Error?t.message:"Unable to start checkout.";return console.error("[billing/concurrency/checkout]",r),r==="quantity must be a positive whole number"?e.json({error:r},400):e.json({error:r},500)}});async function AH(e,t,r,n,i){let o=Eh(r);if(!o||t.stripe_customer_id&&o!==t.stripe_customer_id)throw new Error("Subscription does not belong to this billing account.");let a=(await e.invoices.list({customer:o,status:"open",limit:100})).data.find(f=>(f.amount_remaining??0)>0);if(a)return{ok:!1,status:"outstanding_invoice",invoiceId:a.id,hostedInvoiceUrl:a.hosted_invoice_url,error:"Pay the outstanding invoice before changing plans."};let c=_p(r)?.price?.id,l=c?KE[c]:void 0,d=!l||i.credits_mc>l.credits_mc,u=await e.subscriptions.update(r.id,{items:[{id:n,price:i.price_id}],proration_behavior:"always_invoice",payment_behavior:"pending_if_incomplete",expand:["latest_invoice"]}),p=typeof u.latest_invoice=="object"&&u.latest_invoice&&!("deleted"in u.latest_invoice)?u.latest_invoice:u.latest_invoice?await e.invoices.retrieve(String(u.latest_invoice)):void 0;if(u.pending_update||!p||p.status!=="paid")return{ok:!1,status:"payment_required",invoiceId:p?.id,hostedInvoiceUrl:p?.hosted_invoice_url,error:"The prorated payment must succeed before the plan and Credits can be upgraded."};let m=await e.invoices.retrieve(p.id),g=await yT(e,m);if(d&&g.status==="not_applicable")throw new Error("The paid upgrade invoice did not produce the required Credit grant.");return await Bh(t.id,i.tier,i.concurrency,r.id),{ok:!0,status:g.status==="already_fulfilled"?"already_fulfilled":"fulfilled",invoiceId:p.id,hostedInvoiceUrl:m.hosted_invoice_url,amountPaidUsd:(m.amount_paid??0)/100,amountSettledUsd:(m.total??0)/100,creditsAdded:g.status==="not_applicable"?0:i.credits_mc/rt}}M.get("/billing/offer/:slug",async e=>{let t=GE(e.req.param("slug").trim().toLowerCase());return t?e.json(t):e.json({error:"Unknown offer."},404)});M.post("/billing/subscribe",cn,jo,async e=>{try{let t=e.get("sessionUser"),r=await e.req.json().catch(()=>({})),n=ap[String(r.tier??"")];if(!n)return e.json({error:"Invalid tier. Choose starter, growth, or scale."},400);if(ry(t.subscription_tier,n.tier)&&r.confirm_offer_change!==!0){let l=eo(t.subscription_tier),d=ty(t.subscription_tier);return e.json({error:`You are on the ${d?.label??"partner offer"} (${l?.label??t.subscription_tier}, $${l?.amount_usd??"?"} per ${l?.interval??"year"}). Switching to ${n.label} ends that offer and its non-expiring Credits, and it cannot be re-claimed automatically. Confirm to continue.`,error_code:"offer_change_requires_confirmation",current_plan:l,requested_plan:eo(n.tier)},409)}let i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);let o=new sn(i,{apiVersion:an}),s=t.stripe_customer_id;s||(s=(await o.customers.create({email:t.email})).id,await qs(t.id,s));let a=t.subscription_id;if(!a)try{a=await bH(o,s)??null}catch(l){console.warn("[billing/subscribe] could not check for existing subscription",l instanceof Error?l.message:l)}if(a){let l=await o.subscriptions.retrieve(a),d=_p(l);if(d?.price?.id===n.price_id)return e.json({updated:!1,tier:n.tier,already_current:!0});let u=d?.id;if(u){let p=await AH(o,t,l,u,n);return p.ok?e.json({updated:!0,tier:n.tier,billed:!0,amount_paid_usd:p.amountPaidUsd,amount_settled_usd:p.amountSettledUsd,credits_added:p.creditsAdded,invoice_id:p.invoiceId}):e.json({error:p.error,error_code:p.status,invoice_id:p.invoiceId,hosted_invoice_url:p.hostedInvoiceUrl},p.status==="outstanding_invoice"?409:402)}}let c=await o.checkout.sessions.create({customer:s,mode:"subscription",line_items:[{price:n.price_id,quantity:1}],...n.intro_coupon?{discounts:[{coupon:n.intro_coupon}]}:{},ui_mode:"embedded",automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},return_url:n.private_offer&&ey(n.tier)?`${_n()}${ey(n.tier)}?purchased=${n.tier}&session_id={CHECKOUT_SESSION_ID}`:`${_n()}/billing?subscribed=${n.tier}&session_id={CHECKOUT_SESSION_ID}`});return e.json({clientSecret:c.client_secret})}catch(t){let r=t instanceof Error?t.message:"Unable to start subscription.";return console.error("[billing/subscribe]",r),e.json({error:r},500)}});M.post("/billing/checkout/confirm",cn,jo,async e=>{let t=e.get("sessionUser"),r=await e.req.json().catch(()=>({})),n=typeof r.session_id=="string"?r.session_id.trim():"";if(!/^cs_(test_|live_)?[A-Za-z0-9]+$/.test(n))return e.json({error:"A valid Checkout session id is required."},400);let i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);try{let o=new sn(i,{apiVersion:an});if((await o.checkout.sessions.retrieve(n)).mode==="payment"){let c=await vT(o,n,t.stripe_customer_id);return c.status==="not_applicable"?e.json({ok:!1,status:"pending",message:"Stripe has not confirmed this payment yet."},202):e.json({ok:!0,status:c.status,credits:c.credits})}let a=await bT(o,n,t.stripe_customer_id);return a.status==="not_applicable"?e.json({ok:!1,status:"pending",message:"Stripe has not confirmed this payment yet."},202):e.json({ok:!0,status:a.status,tier:a.tier})}catch(o){let s=o instanceof Error?o.message:"Unable to confirm checkout.";return s.includes("does not belong")||s.includes("not a subscription")?e.json({error:s},409):(console.error("[billing/checkout/confirm]",n,s),e.json({error:"We could not confirm this checkout yet. Please retry."},503))}});M.post("/billing/portal",cn,jo,async e=>{try{let t=e.get("sessionUser");if(!t.stripe_customer_id)return e.json({error:"No billing account yet \u2014 subscribe first."},409);let r=Yx();if(!r)return e.json({error:"Stripe is not configured."},503);let i=await new sn(r,{apiVersion:an}).billingPortal.sessions.create({customer:t.stripe_customer_id,return_url:`${_n()}/billing`});return e.json({url:i.url})}catch(t){let r=t instanceof Error?t.message:"Unable to open billing portal.";return console.error("[billing/portal]",r),e.json({error:r},500)}});M.post("/billing/concurrency/terminal-checkout",re,async e=>{try{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=yH(r.quantity),i=Ll(),o=Ns(t),s=Ns(t,n),a=eo(t.subscription_tier);if(t.concurrency_stripe_sub_id){let p=process.env.STRIPE_SECRET_KEY?.trim();if(!p)return e.json({error:"Stripe is not configured."},503);let m=new sn(p,{apiVersion:an}),g=await m.subscriptions.retrieve(t.concurrency_stripe_sub_id);if(t.stripe_customer_id&&Eh(g)!==t.stripe_customer_id)throw new Error("Concurrency subscription does not belong to this billing account.");let f=g.items.data.find(y=>y.price?.id===Za);if(!f)throw new Error("Concurrency subscription is missing its pack price.");let h=(f.quantity??1)!==n;return h&&await m.subscriptions.update(g.id,{items:[{id:f.id,quantity:n}],proration_behavior:"create_prorations"}),await El(t.id,n),e.json({updated:h,price:i,current:o,after:s,base_plan:a,next_step:h?"The concurrency pack quantity was updated.":"The concurrency pack quantity was already current."})}let c=process.env.STRIPE_SECRET_KEY?.trim();if(!c)return e.json({error:"Stripe is not configured."},503);let l=new sn(c,{apiVersion:an}),d=t.stripe_customer_id;d||(d=(await l.customers.create({email:t.email})).id,await qs(t.id,d));let u=await l.checkout.sessions.create({customer:d,mode:"subscription",line_items:[{price:Za,quantity:n}],...wH(),automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},success_url:`${_n()}/billing?slot_added=1&source=terminal`,cancel_url:`${_n()}/billing?slot_cancelled=1&source=terminal`});return u.url?e.json({checkout_url:u.url,price:i,current:o,after:s,base_plan:a,next_step:"Open checkout_url in a browser, complete checkout, then restart or retry the MCP request."}):e.json({error:"Stripe did not return a checkout URL."},502)}catch(t){let r=t instanceof Error?t.message:"Unable to start terminal checkout.";return console.error("[billing/concurrency/terminal-checkout]",r),r==="quantity must be a positive whole number"?e.json({error:r},400):e.json({error:r},500)}});M.post("/billing/subscribe/terminal-checkout",re,async e=>{try{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=ap[String(r.tier??"")];if(!n)return e.json({error:"Invalid tier. Choose starter, growth, or scale.",tiers:Object.values(ap).filter(l=>!l.private_offer).map(l=>l.tier)},400);if(ry(t.subscription_tier,n.tier)&&r.confirm_offer_change!==!0){let l=eo(t.subscription_tier),d=ty(t.subscription_tier);return e.json({error:`You are on the ${d?.label??"partner offer"} (${l?.label??t.subscription_tier}, $${l?.amount_usd??"?"} per ${l?.interval??"year"}). Switching to ${n.label} ends that offer and its non-expiring Credits, and it cannot be re-claimed automatically. Retry with confirm_offer_change: true to continue.`,error_code:"offer_change_requires_confirmation",current_plan:l,requested_plan:eo(n.tier)},409)}let i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);let o=new sn(i,{apiVersion:an}),s=t.stripe_customer_id;s||(s=(await o.customers.create({email:t.email})).id,await qs(t.id,s));let a=t.subscription_id;if(!a)try{a=await bH(o,s)??null}catch(l){console.warn("[billing/subscribe] could not check for existing subscription",l instanceof Error?l.message:l)}if(a){let l=await o.subscriptions.retrieve(a),d=_p(l);if(d?.price?.id===n.price_id)return e.json({updated:!1,tier:n.tier,already_current:!0,message:`${n.label} is already active.`});let u=d?.id;if(u){let p=await AH(o,t,l,u,n);return p.ok?e.json({updated:!0,tier:n.tier,billed:!0,amount_paid_usd:p.amountPaidUsd,amount_settled_usd:p.amountSettledUsd,credits_added:p.creditsAdded,invoice_id:p.invoiceId,message:`Switched to ${n.label}, settled the prorated payment immediately, and added ${Number(p.creditsAdded??0).toLocaleString("en-US")} Credits.`}):e.json({error:p.error,error_code:p.status,invoice_id:p.invoiceId,hosted_invoice_url:p.hostedInvoiceUrl},p.status==="outstanding_invoice"?409:402)}}let c=await o.checkout.sessions.create({customer:s,mode:"subscription",line_items:[{price:n.price_id,quantity:1}],...n.intro_coupon?{discounts:[{coupon:n.intro_coupon}]}:{},automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},success_url:`${_n()}/billing?subscribed=${n.tier}&source=terminal&session_id={CHECKOUT_SESSION_ID}`,cancel_url:`${_n()}/billing?subscribe_cancelled=1&source=terminal`});return c.url?e.json({checkout_url:c.url,tier:n.tier,label:n.label,monthly_usd:n.monthly_usd,amount_usd:WE(n),billing_interval:HE(n),credits_per_interval:zE(n),credits_per_month:n.credits_mc/rt,credits_never_expire:!!n.credits_never_expire,includes_memory:!!n.includes_memory,concurrency:n.concurrency,intro:n.intro_coupon?"$1 first month":null,next_step:"Open checkout_url in a browser to complete payment."}):e.json({error:"Stripe did not return a checkout URL."},502)}catch(t){let r=t instanceof Error?t.message:"Unable to start subscription checkout.";return console.error("[billing/subscribe/terminal-checkout]",r),e.json({error:r},500)}});M.post("/billing/concurrency/cancel",cn,jo,async e=>{let t=e.get("sessionUser");if(!t.concurrency_stripe_sub_id)return e.json({error:"No active concurrency subscription."},404);let r=process.env.STRIPE_SECRET_KEY?.trim();if(!r)return e.json({error:"Stripe is not configured."},503);let n=new sn(r,{apiVersion:an}),i=await n.subscriptions.retrieve(t.concurrency_stripe_sub_id);if(t.stripe_customer_id&&Eh(i)!==t.stripe_customer_id)return e.json({error:"Concurrency subscription does not belong to this billing account."},409);if(!i.items.data.some(s=>s.price?.id===Za))return e.json({error:"The saved subscription is not a concurrency pack. No subscription was canceled."},409);await n.subscriptions.cancel(t.concurrency_stripe_sub_id),await El(t.id,0),await Fh(t.id,null);let o=Ns(t,0);return e.json({ok:!0,...o,base_plan:eo(t.subscription_tier)})});M.get("/billing/balance",re,async e=>{let t=e.get("user"),r=t.balance_mc,n=await Qu(t.id,20),i=await HB(t.id),o=Ns(t);return e.json({balance_mc:r,balance_credits:r/rt,free_credits:i,ledger:n,concurrency_pack_quantity:o.pack_quantity,slots_per_pack:o.slots_per_pack,extra_concurrency_slots:o.extra_concurrency_slots,concurrency_limit:o.effective_limit,concurrency_addon_monthly_usd:o.monthly_amount_usd,concurrency:{...o,has_subscription:!!t.concurrency_stripe_sub_id},scale_credit_pack:{eligible:nb(t),increment_usd:Nl,credits_per_increment:cp,max_quantity:ZE,expiry_months:3}})});M.get("/billing/summary",re,ZL);M.get("/billing/connected-usage/history",re,async e=>e.json({ok:!0,ratePolicy:Zh,receipts:await xA(e.get("user").id,100)}));M.post("/billing/credits",re,async e=>{let t=e.get("user"),r=t.balance_mc,n=await e.req.json().catch(()=>({})),i=n.item?.trim().toLowerCase(),o=Qh.map(({aliases:c,...l})=>l),s=i?Qh.find(c=>c.label.toLowerCase().includes(i)||c.key.toLowerCase()===i||c.aliases.some(l=>l.toLowerCase().includes(i)||i.includes(l.toLowerCase()))):void 0,a=n.includeLedger?(await Qu(t.id,10)).map(c=>({amount_mc:c.amount_mc,operation:c.operation,description:c.description,created_at:c.created_at})):void 0;return e.json({balance_mc:r,balance_credits:r/rt,item:n.item??null,matched_cost:s?(({aliases:c,...l})=>l)(s):null,costs:o,concurrency:{...Ns(t),has_subscription:!!t.concurrency_stripe_sub_id,upgrade:Ll()},connected_accounts:Zh,ledger:a})});M.post("/api/internal/extract-refinalize/:id",async e=>{let t=e.req.header("authorization");if(!process.env.CRON_SECRET||t!==`Bearer ${process.env.CRON_SECRET}`)return e.json({error:"Unauthorized"},401);let r=e.req.param("id"),{claimFailedExtractJobForRefinalize:n,failExtractJob:i,getExtractJob:o,finishExtractJob:s,terminalExtractJobStatus:a,extractJobLimitInfo:c}=await import("./site-extract-repository-2LFDP6FZ.js"),{assembleExtractArtifacts:l}=await import("./extract-bundle-MQOAKQDV.js"),d=await o(r);if(!d)return e.json({error:"job not found"},404);if(d.status!=="failed")return e.json({error:"Only failed site extraction jobs can be re-finalized.",status:d.status},409);let u=await n(r);if(!u){let S=await o(r);return e.json({error:"job state changed during re-finalization claim",status:S?.status??"missing"},409)}let p;try{p=await l(u,{})}catch(S){throw await i(r,S instanceof Error?S.message:String(S)).catch(()=>{}),S}let m=await o(r);if(!m)return e.json({error:"job disappeared during finalization"},500);let g=c(m),f=a(m,g.creditTruncated),h=f==="failed"?`No pages were extracted successfully (${m.failedUrls} failed, ${m.remainingUrls} remaining).`:f==="partial"?`${m.successfulUrls} pages succeeded; ${m.failedUrls} failed and ${m.remainingUrls} remain.${g.creditTruncated?` Credit availability limited this request from ${g.requestedMaxPages} to ${g.effectiveMaxPages} pages.`:""}`:null;if(!await s(r,p,f,h))return e.json({error:"job state changed during re-finalization"},409);let b="already_settled_or_refunded";if(u.billedMc==null&&u.userId!=null){let{settleExtractJob:S,countSuccessfulPages:x}=await import("./site-extract-repository-2LFDP6FZ.js"),E=Number(u.options.heldMc??0),v=await x(r),k=Math.min(v*D.page_scrape,E),A="site";try{A=new URL(u.startUrl).hostname}catch{}await S(r,u.userId,E-k,k,A),b=`settled: billed ${k} mc, refunded ${E-k} mc`}let _=p.find(S=>S.contentType==="application/zip"||S.filename?.endsWith(".zip"));return e.json({ok:!0,jobId:r,artifacts:p.length,settlement:b,bundleUrl:_?.url??null,bundleBytes:_?.bytes??null})});M.on(["GET","POST","PUT"],"/api/inngest",jae({client:at,functions:[MP,qP,QP,jN,rL,LL]}));M.route("/",Nu);M.route("/admin/credits",ds);M.route("/api/internal/site-architecture-auditor",fn);M.route("/api/internal/memory",ow);M.route("/api/internal/billing",Ri);M.route("/api/internal/inbox",$m);M.route("/inbox",ga);M.route("/youtube",zm);M.route("/screenshot",Gm);M.route("/facebook",yo);M.route("/tiktok",Pw);M.route("/google-ads",Ad);M.route("/instagram",cg);M.route("/reddit",fg);M.route("/kernel-reddit",hg);M.route("/",gr);M.route("/video",wg);M.route("/maps",vg);M.route("/trustpilot",u_);M.route("/g2",m_);M.route("/directory",Eg);M.route("/lead-list-input",qd);M.route("/lead-list-enrichment",Ud);M.route("/local-sourcebook",yn);M.route("/public/local-sourcebook",Lc);M.route("/admin/local-sourcebook",Mc);M.route("/locations",Oc);M.route("/workflows",Qr);M.route("/serp-intelligence",Hd);M.route("/mcp",G_);M.route("/agent",nU());M.route("/chat",Rr);M.route("/vault",qi);M.route("/schedule",ai);M.route("/resend",Iv);M.route("/workshops",mf);M.route("/editorial-reading-room",ru);M.route("/commons",Me);M.route("/analytics",O);M.route("/crm",lt);M.route("/research",ml);M.route("/",Or);M.route("/public",uh);M.route("/api/internal/scheduled-artifacts",Ax);M.get("/console",e=>e.html(sv()));M.use("/console/auth/*",re);M.get("/console/auth/:id",async e=>{let t=e.req.param("id");await vd();let r=await Sc(e.get("user").id,t),n=r?.browser_agent_session_id?await fa(r.browser_agent_session_id):null;return e.html(iU(t,{domain:r?.domain??null,status:r?.status??null,liveViewUrl:n?.live_view_url??null,sessionOpen:n?.status==="open"}))});M.post("/console/auth/:id/complete",async e=>{let t=e.req.param("id");await vd();let r=await tU(e.get("user").id,t);return r.ok?e.json({ok:!0}):e.json({error:r.error},404)});M.get("/console/:id",e=>e.html(sv(e.req.param("id"))));M.route("/stripe",xT);M.route("/",Hr);process.env.INNGEST_EVENT_KEY||(hB(),NL());process.env.NODE_ENV!=="production"&&M.get("/font-test",e=>e.html(`<!DOCTYPE html>
|
|
4884
|
+
`),l=await Y("putTool",{vault:"Issues",path:s,title:a,content:c},i);return l.ok?e.json({ok:!0}):e.json({error:l.error??"failed to submit support request"},502)}catch(t){let r=t instanceof Error?t.message:"Unable to submit support request.";return console.error("[support/submit]",r),e.json({error:r},500)}});M.get("/memory/usage",re,async e=>{let t=e.get("user");try{let r=await Wx(io(t),hp(t));return e.json(r)}catch(r){return e.json({ok:!1,error:r instanceof Error?r.message:"could not load usage"},502)}});var Jae=(()=>{let e=process.env.SYNC_HARVEST_TIMEOUT_MS,t=e===void 0?NaN:Number(e);return Number.isFinite(t)&&t>0?t:null})();function Yae(e){let t=new AbortController,r=n=>{t.signal.aborted||t.abort(n.reason)};for(let n of e){if(n.aborted){r(n);break}n.addEventListener("abort",()=>r(n),{once:!0})}return t.signal}function pH(e){if(!e||typeof e!="object")return 0;let t=e;return typeof t.totalQuestions=="number"?t.totalQuestions:Array.isArray(t.flat)?t.flat.length:0}function wh(e){return D.paa_base+Math.max(1,e)*D.paa}async function SH(e,t){if(t&&await Jb(e.id,t,1800)||Rm(e))return null;let r=ia(e),n=await od(e.id);return n>=r?me({ok:!1,active:n,limit:r,operation:"harvest",retryAfterSeconds:30}):null}M.post("/harvest/internal/resume/:id",async e=>{let t=e.req.header("authorization");if(!process.env.CRON_SECRET||t!==`Bearer ${process.env.CRON_SECRET}`)return e.json({error:"Unauthorized"},401);let r=await Vt(e.req.param("id"));return!r||r.options.executionOwner!=="direct"||r.options.serpOnly===!0?e.json({error:"PAA job not found"},404):r.status!=="pending"?e.json({job_id:r.id,status:r.status},409):(O_(r.id),e.json({job_id:r.id,status:"pending",recovery_dispatched:!0},202))});M.post("/harvest",re,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=gw.safeParse(r);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let i=n.data;if(i.mode==="full"&&!i.serpOnly)return e.json({error:"Full mode requires serpOnly search."},400);if(i.mode==="full"&&i.serpIdentity)return e.json({error:"Full mode cannot use a saved search identity."},400);let o=i.serpIdentity?await Eo(t.id,i.serpIdentity):null;if(i.serpIdentity&&!o)return e.json({error:"SERP identity not found"},404);if(o&&o.status!=="ready")return e.json({error:"SERP identity is not ready"},409);let s=i.callback_url?.trim();if(s){let y=await fe(s,{field:"callback_url",requireHttps:!0});if(y.error)return e.json({error:y.error},400);i.callback_url=y.parsed?.href}let a={query:i.query,location:i.location,depth:Math.min(4,Math.max(1,i.depth??4)),maxQuestions:Math.min(100,Math.max(1,i.maxQuestions??30)),gl:i.gl??"us",hl:i.hl??"en",device:i.device??"desktop",proxyMode:o?"configured":i.proxyMode??Kn,serpIdentity:i.serpIdentity,proxyZip:i.proxyZip,debug:i.debug??!1,serpOnly:i.serpOnly??!1,pages:Math.min(2,Math.max(1,i.pages??1)),mode:i.mode??"light",includeAllSerpFeatures:i.includeAllSerpFeatures??!1,includeLocalPack:i.includeLocalPack??!1,includeForums:i.includeForums??!1,includeVideos:i.includeVideos??!1,includeAiOverview:i.includeAiOverview??!1,includeWhatPeopleSaying:i.includeWhatPeopleSaying??!1},c=e.req.header("idempotency-key")??e.req.header("x-idempotency-key"),l=null,d=null;if(c!=null){let y=c.trim();if(!y||y.length>200)return e.json({error:"Idempotency-Key must contain 1-200 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);l=`${a.serpOnly?"serp":"paa"}-${Hn(`${t.id}\0${y}`).slice(0,24)}`,d=Hn(JSON.stringify({options:a,callbackUrl:i.callback_url??null}));let b=await Vt(l,t.id);if(b)return b.options.requestFingerprint!==d?e.json({error:"Idempotency-Key was already used for a different asynchronous harvest call.",errorCode:"idempotency_conflict",retryable:!1},409):e.json({job_id:b.id,status:b.status,replayed:!0},["done","failed","cancelled"].includes(b.status)?200:202)}let u=await SH(t,e.req.header("x-mcp-scraper-concurrency-lock"));if(u)return e.json(u,429,{"Retry-After":"30"});let p=a.serpOnly?i.serpIdentity?D.serp_headful:Rl(a.pages,"brightdata"):wh(a.maxQuestions),m=l??mH(),g=`${a.serpOnly?"serp-search":"paa-harvest"}:${m}:hold`,f=`${a.serpOnly?"SERP search":"PAA harvest"}: ${a.query}`.slice(0,500),h=await vy({jobId:m,userId:t.id,tool:a.serpOnly?"search_serp":"harvest_paa",normalizedFlags:a,requestId:e.req.header("x-request-id")??e.req.header("x-vercel-id")??null});{let y=await _t(t.id,p,a.serpOnly?N.SERP:N.PAA,f,g);if(!y.ok)return await Si(h,"failed","insufficient_balance"),e.json(ce(y.balance_mc,p),402);await ao(h,g,"hold",p);try{await Ph(m,t.id,a.query,{...a,...a.serpOnly?{}:{executionOwner:"direct",paaRecoveryCount:0},billingHoldMc:p,billingDebitKey:g,...d?{requestFingerprint:d}:{}},i.callback_url)}catch(b){let _=l?await Vt(l,t.id).catch(()=>null):null;if(_&&_.options.requestFingerprint===d)return e.json({job_id:_.id,status:_.status,replayed:!0},["done","failed","cancelled"].includes(_.status)?200:202);throw await ze(t.id,g,0,a.serpOnly?N.SERP_REFUND:N.PAA_REFUND,`${a.serpOnly?"SERP":"PAA"} job creation failed`,a.serpOnly?"serp_search":"paa_harvest").catch(()=>{}),await Si(h,"failed","job_creation_failed"),b}a.serpOnly||O_(m)}if(a.serpOnly&&process.env.CRON_SECRET){let y=new URL(e.req.url);fetch(`${y.origin}/cron/tick`,{headers:{Authorization:`Bearer ${process.env.CRON_SECRET}`}}).catch(()=>{}),await new Promise(b=>setTimeout(b,80))}return e.json({job_id:m,status:"pending"},202)});M.post("/harvest/sync",re,async e=>{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=gw.safeParse(r);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let i=n.data;if(i.mode==="full"&&!i.serpOnly)return e.json({error:"Full mode requires serpOnly search."},400);if(i.mode==="full"&&i.serpIdentity)return e.json({error:"Full mode cannot use a saved search identity."},400);let o=i.serpIdentity?await Eo(t.id,i.serpIdentity):null;if(i.serpIdentity&&!o)return e.json({error:"SERP identity not found"},404);if(o&&o.status!=="ready")return e.json({error:"SERP identity is not ready"},409);let s={query:i.query,location:i.location,depth:Math.min(4,Math.max(1,i.depth??4)),maxQuestions:Math.min(100,Math.max(1,i.maxQuestions??30)),gl:i.gl??"us",hl:i.hl??"en",device:i.device??"desktop",proxyMode:o?"configured":i.proxyMode??Kn,serpIdentity:i.serpIdentity,proxyZip:i.proxyZip,debug:i.debug??!1,serpOnly:i.serpOnly??!1,pages:Math.min(2,Math.max(1,i.pages??1)),mode:i.mode??"light",recency:i.recency,includeAllSerpFeatures:i.includeAllSerpFeatures??!1,includeLocalPack:i.includeLocalPack??!1,includeForums:i.includeForums??!1,includeVideos:i.includeVideos??!1,includeAiOverview:i.includeAiOverview??!1,includeWhatPeopleSaying:i.includeWhatPeopleSaying??!1},a=e.req.header("idempotency-key")??e.req.header("x-idempotency-key"),c=null,l=Hh(s);if(a!=null){let E=a.trim();if(!E||E.length>200)return e.json({error:"Idempotency-Key must contain 1-200 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);c=`sync_${Hn(`${t.id}\0${E}`).slice(0,24)}`;let v=await Vt(c,t.id);if(v){if((typeof v.options.requestFingerprint=="string"?v.options.requestFingerprint:Hh(v.options))!==l)return e.json({error:"Idempotency-Key was already used for a different harvest_paa/search_serp call.",errorCode:"idempotency_conflict",retryable:!1},409);if(v.status==="done"){let A=await Ho(c,t.id);return e.json({job_id:c,status:"done",result:Al(v.result),attempts:Zi(A),replayed:!0})}if(v.status==="running")return e.json({error:"This request is already running under the supplied Idempotency-Key.",errorCode:"idempotency_in_progress",retryable:!0,job_id:c},409,{"Retry-After":"10"});if(v.status==="failed"||v.status==="cancelled"){let A=await Ho(c,t.id);return e.json({job_id:c,status:v.status,...v.publicError??{},attempts:Zi(A),replayed:!0})}}}let d=await SH(t,e.req.header("x-mcp-scraper-concurrency-lock"));if(d)return e.json(d,429,{"Retry-After":"30"});let u=s.serpOnly&&!i.serpIdentity,p=s.serpOnly?u?Rl(s.pages,"brightdata"):D.serp_headful:wh(s.maxQuestions),m;if(c){let E=await Nh(c,t.id,s.query,{...s,requestFingerprint:l});if(!E.created){if(E.job.status==="done"){let v=await Ho(c,t.id);return e.json({job_id:c,status:"done",result:Al(E.job.result),attempts:Zi(v),replayed:!0})}if(E.job.status==="failed"||E.job.status==="cancelled"){let v=await Ho(c,t.id);return e.json({job_id:c,status:E.job.status,...E.job.publicError??{},attempts:Zi(v),replayed:!0})}return e.json({error:"This request is already running under the supplied Idempotency-Key.",errorCode:"idempotency_in_progress",retryable:!0,job_id:c},409,{"Retry-After":"10"})}m=E.job.id}else m=await xS(t.id,s.query,s);let g=await vy({jobId:m,userId:t.id,tool:s.serpOnly?"search_serp":"harvest_paa",normalizedFlags:s,requestId:e.req.header("x-request-id")??e.req.header("x-vercel-id")??null}),f=`harvest-sync:${m}:charge`,h=await _t(t.id,p,s.serpOnly?N.SERP:N.PAA,s.query,f);if(!h.ok)return await Si(g,"failed","insufficient_balance"),c&&await _r(m,JSON.stringify({error_code:"insufficient_balance"}),ei(xi(new Error("insufficient_balance")),{chargeStatus:"not_charged"})),e.json(ce(h.balance_mc,p),402);await ao(g,f,"hold",p);let y=kp(m,t.id,g),b=Jae??zu(s.maxQuestions,s.serpOnly,s.mode,s.pages).serverMs,_=Yae([...c?[]:[e.req.raw.signal],AbortSignal.timeout(b)]),S={value:null},x=null;try{let E=we(),v=u,k=process.env.APIFY_API_TOKEN?.trim();if(v&&(s.pages===2?!k:!TE()||s.mode!=="full"&&!k))throw new Error("Search providers are not configured");if(v&&!g)throw new Error("Search accounting is unavailable");let A=()=>Wo({...s,kernelApiKey:ae(),...o?{kernelProxyId:o.kernel_proxy_id,kernelProfileName:o.kernel_profile_name,kernelProfileSaveChanges:!1,kernelStealth:!0}:{},headless:!0,format:"json",outputDir:"/tmp/paa-output-api",signal:_,softDeadlineMs:Date.now()+(s.serpOnly?6500:Math.floor(b*.8)),onAttemptEvent:y}),C=await ve({...E,op:s.serpOnly?"serp":"paa",userId:t.id,headlessSentOut:S,operationRunId:g,operationAccountScope:"production-default"},()=>v?PE({query:s.query,pages:s.pages,mode:s.mode,token:k,operationRunId:g,accountScope:"production-default",signal:_}).then(I=>(x=I,I.result)):A());if(v&&!x)throw new Error("Search provider billing receipt is missing");i.serpIdentity&&await Jg(t.id,i.serpIdentity).catch(()=>{}),await Sl(m,C);let T=await Ho(m,t.id);if(f){let I=s.serpOnly?v?Rl(x.deliveredPages,x.deliveryProvider):Il(S.value):wh(pH(C));await ze(t.id,f,I,s.serpOnly?N.SERP_REFUND:N.PAA_REFUND,s.serpOnly?"SERP search settlement":"PAA harvest settlement",s.serpOnly?"serp_search_sync":"paa_harvest_sync"),await ao(g,f,"settlement",I)}else if(s.serpOnly){let I=v?Rl(x.deliveredPages,x.deliveryProvider):Il(S.value),L=p-I;L>0?await ie(t.id,L,N.SERP_REFUND,"headless-mode pricing settle"):L<0&&await oe(t.id,-L,N.SERP,s.query)}else{let I=wh(pH(C)),L=p-I;L>0?await ie(t.id,L,N.PAA_REFUND,"overestimate refund"):L<0&&await oe(t.id,-L,N.PAA,s.query)}return await Si(g,"succeeded"),e.json({job_id:m,status:"done",result:Al(C),attempts:Zi(T)})}catch(E){console.error("[harvest/sync] failed",{jobId:m,error:E instanceof Error?E.message:String(E)});let v=xi(E);await Si(g,v.error_code==="harvest_timeout"?"timed_out":v.terminalStatus,v.error_code);let k=xE(E),A={chargeStatus:"refunded",...k?{details:{retry_guidance:`SERP extraction stopped at the stable ${k} stage. Retry with the same idempotencyKey after correcting the request or transient condition.`}}:{}},C=vp(v,A),T=await Ho(m,t.id);return v.terminalStatus==="cancelled"||e.req.raw.signal.aborted?(await $h(m,so(v),ei(v,{...A,chargeStatus:"refund_pending"})),f?(await ze(t.id,f,0,N.REFUND,"cancelled call",s.serpOnly?"serp_search_sync":"paa_harvest_sync"),await ao(g,f,"refund",p)):await ie(t.id,p,N.REFUND,"cancelled call"),await $h(m,so(v),ei(v,A)),e.json({job_id:m,status:"cancelled",...C,attempts:Zi(T)},v.httpStatus)):(await _r(m,so(v),ei(v,{...A,chargeStatus:"refund_pending"})),f?(await ze(t.id,f,0,N.REFUND,"failed call",s.serpOnly?"serp_search_sync":"paa_harvest_sync"),await ao(g,f,"refund",p)):await ie(t.id,p,N.REFUND,"failed call"),await _r(m,so(v),ei(v,A)),e.json({job_id:m,status:"failed",...C,attempts:Zi(T)},v.httpStatus))}});function EH(e,t,r){return t||(!r||!["failed","cancelled"].includes(e)?null:$r({errorCode:"service_unavailable",retryable:!0}))}function kH(e){let{error:t,publicError:r,...n}=e,i=EH(e.status,r,!!t);return{...n,error:i?.message??null,...i??{}}}M.get("/jobs/:id",re,async e=>{let t=await Vt(e.req.param("id"),e.get("user").id);if(!t)return e.json({error:"Job not found"},404);let r=await Ho(t.id,e.get("user").id),n=t.result&&typeof t.result=="object"?Al(t.result):t.result;return e.json({...kH(t),result:n,attempts:Zi(r)})});M.get("/jobs",re,async e=>e.json((await Lh(e.get("user").id)).map(kH)));M.get("/history",re,async e=>{let t=e.get("user").id,[r,n]=await Promise.all([Lh(t),TS(t,100)]),i=r.map(l=>({id:l.id,ts:l.created_at,query:l.query,location:l.options?.location??"",source:l.options?.jobKind==="extract_url"?"extract_url":N.SERP,status:l.status,result_count:l.result?(l.result.flat?.length??0)+(l.result.organicResults?.length??0):0})),o=n.map(l=>({id:l.id,ts:l.created_at,query:l.query,location:l.location??"",source:l.source,status:l.status,result_count:l.result_count??0,error:l.error??null})),s=new Set(r.filter(l=>l.options?.jobKind==="extract_url"&&l.completed_at).map(l=>`${l.query}\0${l.completed_at}`)),a=o.filter(l=>l.source!=="extract_url"||!s.has(`${l.query}\0${l.ts}`)),c=[...i,...a].sort((l,d)=>String(d.ts).localeCompare(String(l.ts)));return e.json(c.slice(0,100))});M.get("/ledger",re,async e=>e.json(await Qu(e.get("user").id,100)));M.post("/admin/users",bo,async e=>{let{email:t,name:r,password:n}=await e.req.json();if(!t?.trim())return e.json({error:"email is required"},400);try{return e.json(await Ih(t.trim(),r?.trim(),n?.trim()),201)}catch(i){if((i instanceof Error?i.message:"").includes("UNIQUE"))return e.json({error:"Email already registered"},409);throw i}});M.get("/admin/users",bo,async e=>e.json(await Ch()));M.delete("/admin/users/:id",bo,async e=>(await Th(parseInt(e.req.param("id")??"0")),e.json({ok:!0})));M.post("/admin/backfill-signup-credits",bo,async e=>{let t=await Ch();return e.json({processed:t.length,credited:0,skipped:t.length,users_credited:[],retired:!0})});function _l(e){let t=typeof e.requestedDelivery=="string"?e.requestedDelivery:typeof e.delivery=="string"?e.delivery:"auto",r=e.depositToVault===!0?"memory":t,n=typeof e.preserveMedia=="boolean"?e.preserveMedia:e.downloadMedia===!0,i=String(e.url??"");try{i=new URL(i).href}catch{}return{url:i,screenshot:e.screenshot===!0,screenshotDevice:e.screenshotDevice==="mobile"?"mobile":"desktop",extractBranding:e.extractBranding===!0,includeFeaturedImage:e.includeFeaturedImage===!0,mediaTypes:Array.isArray(e.mediaTypes)?e.mediaTypes:["image","video","audio"],maxMediaAssets:typeof e.maxMediaAssets=="number"?e.maxMediaAssets:100,maxInlineImages:typeof e.maxInlineImages=="number"?e.maxInlineImages:3,requestedDelivery:t,delivery:r,preserveMedia:n,vaultName:typeof e.vaultName=="string"?e.vaultName:null,allowLocal:e.allowLocal===!0}}function _h(e,t){let r={jobId:e.id,job_id:e.id,status:e.status,statusTool:"extract_url_status",replayed:t};if(e.status==="done"&&e.result&&typeof e.result=="object")return{...r,result:e.result,error:null,billing:{chargeStatus:"charged",credits:D.page_scrape/rt}};if(e.status==="failed"||e.status==="cancelled"){let n=e.publicError??$r({errorCode:"service_unavailable",retryable:!0});return{...r,result:null,error:n,billing:{chargeStatus:n.charge_status??"unknown",credits:0}}}return{...r,result:null,error:null,billing:{chargeStatus:"charged",credits:D.page_scrape/rt},message:"Single-page extraction is running durably. Poll extract_url_status with this jobId; polling does not start or bill another extraction."}}M.post("/extract-url/start",re,async e=>{let t=await e.req.json().catch(()=>({})),r=Jx(e.req.raw),n=(r?hw:fw).safeParse(t);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let i=n.data,o=r&&"allowLocal"in i&&i.allowLocal===!0;if(i.depositToVault&&!["auto","memory"].includes(i.delivery??"auto"))return e.json({error:'depositToVault conflicts with the requested delivery; use delivery:"memory" instead.'},400);if(i.preserveMedia!==void 0&&i.downloadMedia!==void 0&&i.preserveMedia!==i.downloadMedia)return e.json({error:"preserveMedia conflicts with deprecated downloadMedia."},400);if(o)try{new URL(i.url)}catch{return e.json({error:"Invalid URL"},400)}else{let h=await fe(i.url,{field:"URL"});if(h.error||!h.parsed)return e.json({error:h.error??"Invalid URL"},400)}let a=(e.req.header("idempotency-key")??e.req.header("x-idempotency-key"))?.trim()||mH();if(a.length>200)return e.json({error:"Idempotency-Key must contain 1-200 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);let c=e.get("user"),l=(()=>{try{return new URL(i.url).href}catch{return i.url}})(),d=`exurl-${Hn(`${c.id}\0${a}`).slice(0,24)}`,u=_l(i),p=JSON.stringify(u),m=await Vt(d,c.id);if(m){if(JSON.stringify(_l(m.options))!==p)return e.json({error:"Idempotency-Key was already used for a different extract_url call.",errorCode:"idempotency_conflict",retryable:!1},409);if(m.status!=="pending")return e.json(_h(m,!0),m.status==="running"?202:200)}else{if(!Rm(c)){let y=ia(c),b=await od(c.id);if(b>=y)return e.json(me({ok:!1,active:b,limit:y,operation:"extract_url",retryAfterSeconds:30}),429,{"Retry-After":"30"})}let h=`extract-url:${d}:charge`;try{await Ph(d,c.id,l,{...u,executionOwner:"inngest",jobKind:"extract_url",billingDebitKey:h})}catch(y){if(m=await Vt(d,c.id),!m)throw y;if(JSON.stringify(_l(m.options))!==p)return e.json({error:"Idempotency-Key was already used for a different extract_url call.",errorCode:"idempotency_conflict",retryable:!1},409)}m=await Vt(d,c.id)}if(!m)throw new Error("Durable extraction job disappeared after admission.");let g=typeof m.options.billingDebitKey=="string"?m.options.billingDebitKey:`extract-url:${m.id}:charge`,f=await _t(c.id,D.page_scrape,N.EXTRACT_URL,new URL(l).hostname,g);if(!f.ok){let h=$r({errorCode:"insufficient_balance",retryable:!1,chargeStatus:"not_charged"});return await _r(m.id,JSON.stringify({error_code:"insufficient_balance"}),h),e.json(ce(f.balance_mc,D.page_scrape),402)}try{await at.send({id:m.id,name:"mcp-scraper/extract-url.requested",data:{jobId:m.id}}),await Va(m.id)}catch(h){let y=h instanceof Error?h.message:String(h);return await Va(m.id,y).catch(()=>!1),console.error("[extract-url/start] dispatch pending retry:",y),e.json({..._h(m,!0),error:"Extraction dispatch was not acknowledged; retry with the same Idempotency-Key or poll extract_url_status.",errorCode:"extract_dispatch_pending",retryable:!0},503)}return e.json(_h(m,!1),202)});M.get("/extract-url/status/:id",re,async e=>{let t=await Vt(e.req.param("id"),e.get("user").id);return!t||t.options.jobKind!=="extract_url"?e.json({error:"Extraction job not found"},404):e.json(_h(t,!0))});M.post("/extract-url",re,async e=>{let t=await e.req.json().catch(()=>({})),r=Jx(e.req.raw),n=(r?hw:fw).safeParse(t);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let{url:i,screenshot:o,screenshotDevice:s,extractBranding:a,includeFeaturedImage:c,mediaTypes:l}=n.data,d=r&&"allowLocal"in n.data&&n.data.allowLocal===!0,u=n.data.delivery??"auto";if(n.data.depositToVault&&!["auto","memory"].includes(u))return e.json({error:'depositToVault conflicts with the requested delivery; use delivery:"memory" instead.'},400);if(n.data.preserveMedia!==void 0&&n.data.downloadMedia!==void 0&&n.data.preserveMedia!==n.data.downloadMedia)return e.json({error:"preserveMedia conflicts with deprecated downloadMedia."},400);let p=n.data.depositToVault?"memory":u,m=n.data.preserveMedia??n.data.downloadMedia??!1,g=n.data.maxMediaAssets??100,f=n.data.maxInlineImages??3;if(d)try{new URL(i)}catch{return e.json({error:"Invalid URL"},400)}else{let C=await fe(i,{field:"URL"});if(C.error||!C.parsed)return e.json({error:C.error??"Invalid URL"},400)}let h=(()=>{try{return new URL(i).href}catch{return i}})(),b=ur(h)?.rawReplayUrl??h,_=e.get("user"),S=e.req.header("idempotency-key")??e.req.header("x-idempotency-key"),x=null;if(S!=null){let C=S.trim();if(!C||C.length>200)return e.json({error:"Idempotency-Key must contain 1-200 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);x=`exurl_${Hn(`${_.id}\0${C}`).slice(0,24)}`;let T=_l(n.data),I=JSON.stringify(T),L=await Vt(x,_.id);if(L){if(JSON.stringify(_l(L.options))!==I)return e.json({error:"Idempotency-Key was already used for a different extract_url call.",errorCode:"idempotency_conflict",retryable:!1},409);if(L.status==="done")return e.json({...L.result,replayed:!0});if(L.status==="running")return e.json({error:"This request is already running under the supplied Idempotency-Key.",errorCode:"idempotency_in_progress",retryable:!0,job_id:x},409,{"Retry-After":"10"});if(L.status==="failed"||L.status==="cancelled")return e.json({...L.publicError??{},job_id:x,status:L.status,replayed:!0})}}let E=await ge(_,"extract_url",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:h}});if(!E.ok)return e.json(me(E),429,{"Retry-After":String(E.retryAfterSeconds)});let v=!1,k=null,A=null;try{if(x){let L=await Nh(x,_.id,h,_l(n.data));if(k=L.job.id,!L.created)return L.job.status==="done"?e.json({...L.job.result,replayed:!0}):L.job.status==="failed"||L.job.status==="cancelled"?e.json({...L.job.publicError??{},job_id:L.job.id,status:L.job.status,replayed:!0}):e.json({error:"This request is already running under the supplied Idempotency-Key.",errorCode:"idempotency_in_progress",retryable:!0,job_id:L.job.id},409,{"Retry-After":"10"})}let C=k?`extract-url:${k}:charge`:null,T=C?await _t(_.id,D.page_scrape,N.EXTRACT_URL,new URL(h).hostname,C):await oe(_.id,D.page_scrape,N.EXTRACT_URL,new URL(h).hostname);if(!T.ok)return k&&await _r(k,JSON.stringify({error_code:"insufficient_balance"})),e.json(ce(T.balance_mc,D.page_scrape),402);v=!0;try{A=await sp({userId:_.id,jobId:k,billingDebitKey:C,normalizedFlags:{screenshot:o===!0,extractBranding:a===!0,preserveMedia:m}})}catch(L){console.warn(JSON.stringify({event:"extract_url_cost_start_failed",message:L instanceof Error?L.message:String(L)}))}let I={...await ve({...we(),op:"page_scrape",userId:_.id,operationRunId:A?.runId??null,operationRootAttemptId:A?.rootAttemptId??null,operationAccountScope:"production-default"},()=>hm(_,{canonicalUrl:h,screenshot:o===!0,screenshotDevice:s==="mobile"?"mobile":"desktop",extractBranding:a===!0,includeFeaturedImage:c===!0,mediaTypes:l??["image","video","audio"],maxMediaAssets:g,maxInlineImages:f,requestedDelivery:u,effectiveDelivery:p,preserveMedia:m,vaultName:n.data.vaultName})),...k?{job_id:k}:{}};return A&&await Xa({...A,status:"succeeded"}).catch(()=>{}),k&&await Sl(k,I),e.json(I)}catch(C){A&&await Xa({...A,status:"failed",failureClass:"extract_url_failed"}).catch(()=>{});let T=no(C),I=JSON.stringify({error_code:T.errorCode,http_status:T.httpStatus}),L=Zn(T,{chargeStatus:v?"refund_pending":"not_charged"}),{error:H,...B}=L;k&&await _r(k,I,B),v&&(k?await ze(_.id,`extract-url:${k}:charge`,0,N.EXTRACT_URL,"failed call","extract_url"):await ie(_.id,D.page_scrape,N.EXTRACT_URL,"failed call"));let F=Zn(T,{chargeStatus:v?"refunded":"not_charged"});if(k){let{error:U,...W}=F;await _r(k,I,W)}return await Z({userId:_.id,source:"extract_url",status:"failed",query:h,error:T.errorCode}),e.json({...F,...k?{job_id:k}:{}},iy(T))}finally{await te(E.lockId)}});M.post("/diff-page",re,async e=>{let t=await e.req.json().catch(()=>({})),r=Jx(e.req.raw),n=(r?pM:uM).safeParse(t);if(!n.success)return e.json({error:n.error.issues[0]?.message??"Invalid request"},400);let{url:i,resetBaseline:o}=n.data;if(r&&"allowLocal"in n.data&&n.data.allowLocal===!0)try{new URL(i)}catch{return e.json({error:"Invalid URL"},400)}else{let u=await fe(i,{field:"URL"});if(u.error||!u.parsed)return e.json({error:u.error??"Invalid URL"},400)}let a=(()=>{try{return new URL(i).href}catch{return i}})(),c=e.get("user"),l=await ge(c,"diff_page",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:a}});if(!l.ok)return e.json(me(l),429,{"Retry-After":String(l.retryAfterSeconds)});let d=!1;try{let{ok:u,balance_mc:p}=await oe(c.id,D.diff_page,N.DIFF_PAGE,new URL(a).hostname);if(!u)return e.json(ce(p,D.diff_page),402);d=!0;let m=await RS(c.id,a),g=o?null:m,f=ae(),h=await Qo({url:a,kernelApiKey:f}),y=h.bodyMarkdown??"",b=Hn(y),{value:_,truncated:S}=bB(y),x=new Date().toISOString();await CS(c.id,a,{contentHash:b,content:_,title:h.title??null,contentBytes:Buffer.byteLength(y,"utf8"),truncated:S});let E,v={hunks:[],linesAdded:0,linesRemoved:0,percentChanged:0,hunksTruncated:!1,hunksTruncatedReason:null,totalChangedLineCount:0};return g?g.contentHash===b?E="unchanged":(E="changed",v=wB(g.content,_)):(E="baseline",v.percentChanged=null),await Z({userId:c.id,source:"diff_page",status:"done",query:a,result:{status:E}}),e.json({url:a,title:h.title??null,status:E,isReset:!!o&&!!m,previousCheckedAt:g?.checkedAt??null,currentCheckedAt:x,contentHash:b,previousContentHash:g?.contentHash??null,summary:{linesAdded:v.linesAdded,linesRemoved:v.linesRemoved,percentChanged:v.percentChanged},hunks:v.hunks,contentTruncated:S,hunksTruncated:v.hunksTruncated,hunksTruncatedReason:v.hunksTruncatedReason,totalChangedLineCount:v.totalChangedLineCount})}catch(u){let p=u instanceof Error?u.message:String(u);return d&&await ie(c.id,D.diff_page,N.DIFF_PAGE,"failed call"),await Z({userId:c.id,source:"diff_page",status:"failed",query:a,error:p}),e.json({error:p},500)}finally{await te(l.lockId)}});M.post("/extract-site/read",re,async e=>{let t=yM.safeParse(await e.req.json().catch(()=>({})));if(!t.success){let r=$r({errorCode:"invalid_request",retryable:!1});return e.json({error:r.message,...r},400)}try{let r=await UB({ownerId:String(e.get("user").id),...t.data});if(!r){let n=$r({errorCode:"site_export_not_found",retryable:!1});return e.json({error:n.message,...n},404)}return e.json(r)}catch(r){if(r instanceof Uu){let i=$r({errorCode:"site_export_format_unavailable",retryable:!1});return e.json({error:i.message,...i},409)}console.error("[extract-site/read]",r instanceof Error?r.message:r);let n=$r({errorCode:"site_export_read_failed",retryable:!0});return e.json({error:n.message,...n},500)}});M.post("/extract-site/image",re,async e=>{let t=bM.safeParse(await e.req.json().catch(()=>({})));if(!t.success){let r=$r({errorCode:"invalid_request",retryable:!1});return e.json({error:r.message,...r},400)}try{let r=await jB({ownerId:String(e.get("user").id),...t.data});if(!r){let n=$r({errorCode:"site_export_image_not_found",retryable:!1});return e.json({error:n.message,...n},404)}return e.json({jobId:t.data.jobId,imageId:t.data.imageId,sourcePage:r.artifact.sourcePage,sourceUrl:r.artifact.sourceUrl,mimeType:r.artifact.contentType,bytes:r.bytes.length,sha256:r.artifact.sha256,dataBase64:r.bytes.toString("base64")})}catch(r){console.error("[extract-site/image]",r instanceof Error?r.message:r);let n=$r({errorCode:"site_export_read_failed",retryable:!0});return e.json({error:n.message,...n},500)}});M.post("/archive/read",re,async e=>{let t=await e.req.json().catch(()=>({})),r=mM.safeParse(t);if(!r.success)return e.json({error:"archive_invalid_request",error_code:"archive_invalid_request",retryable:!1,message:r.error.issues[0]?.message??"Invalid request"},400);let{artifactId:n,url:i,path:o,depositToLibrary:s}=r.data,a=e.get("user"),c=await ge(a,"archive_read",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:n?{artifactId:n}:{url:i}});if(!c.ok)return e.json(me(c),429,{"Retry-After":String(c.retryAfterSeconds)});try{let l=null;if(n){let f=n.startsWith(_g)?await hD({artifactId:n,ownerId:String(a.id)}):n.startsWith(mm)?await GP({artifactId:n,ownerId:String(a.id)}):await ok({artifactId:n,ownerId:String(a.id)});if(!f)throw new Ie("archive_not_found","Archive artifact was not found or has expired.",404);l={buffer:f,ref:n}}if(!o){let f=l?await Tx(l.buffer,l.ref,r.data.maxEntries??200):await TB(i,r.data.maxEntries??200);return e.json({mode:"list",...f,...n?{artifactId:n}:{}})}let d=l?await Px(l.buffer,l.ref,o,r.data.offset??0,r.data.maxBytes??5e4):await PB(i,o,r.data.offset??0,r.data.maxBytes??5e4);if(s&&d.fileBytes>Rx)throw new Ie("archive_library_size_limit",`ZIP entry exceeds the ${Math.round(Rx/1024/1024)} MB Library-ingest limit.`);let u=NB(d.archiveUrl,d.path),p=s?await qb(a,{title:LB(d.path),content:d.fullContent,source:u,vault:"Library",capturedAt:MB(d.fullContent),summary:`Source file \`${d.path}\` preserved from ZIP archive ${u.split("#")[0]}.`}):void 0,{fullContent:m,...g}=d;return e.json({mode:"read",...g,...n?{artifactId:n}:{},memory:p})}catch(l){let d=l instanceof Ie?l:new Ie("archive_read_failed",l instanceof Error?l.message:"Failed to read ZIP archive.",500,!0);return e.json({error:d.code,error_code:d.code,retryable:d.retryable,message:d.message},d.status)}finally{await te(c.lockId)}});M.post("/map-urls",re,async e=>{let t=await e.req.json().catch(()=>({})),r=gM.safeParse(t);if(!r.success)return e.json({error:r.error.issues[0]?.message??"Invalid request"},400);let n=r.data,i=await fe(n.url,{field:"URL"});if(i.error||!i.parsed)return e.json({error:i.error??"Invalid URL"},400);let o=i.parsed,s=e.get("user"),a=await ge(s,"map_urls",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:o.href}});if(!a.ok)return e.json(me(a),429,{"Retry-After":String(a.retryAfterSeconds)});let c=!1;try{let{ok:l,balance_mc:d}=await oe(s.id,D.url_map,N.URL_MAP,o.hostname);if(!l)return e.json(ce(d,D.url_map),402);c=!0;let u=await Xs({startUrl:o.href,maxUrls:Math.min(1e4,Math.max(1,n.maxUrls??500)),concurrency:Math.min(20,Math.max(1,n.concurrency??12)),kernelApiKey:n.browserFallback??n.kernelFallback?ae():void 0});return await Z({userId:s.id,source:"map_urls",status:"done",query:o.href,resultCount:Array.isArray(u.urls)?u.urls.length:null,result:u}),e.json(u)}catch(l){let d=l instanceof Error?l.message:String(l);return c&&await ie(s.id,D.url_map,N.URL_MAP_REFUND,"failed call"),await Z({userId:s.id,source:"map_urls",status:"failed",query:o.href,error:d}),e.json({error:d},500)}finally{await te(a.lockId)}});M.post("/wayback/snapshots",re,async e=>{let t=await e.req.json().catch(()=>({})),r=fM.safeParse(t);if(!r.success)return e.json({error:r.error.issues[0]?.message??"Invalid request"},400);let n=r.data,i=await fe(n.url,{field:"URL"});if(i.error||!i.parsed)return e.json({error:i.error??"Invalid URL"},400);for(let c of n.urls??[]){let l=await fe(c,{field:"Wayback selected URL"});if(l.error||!l.parsed)return e.json({error:l.error??"Invalid Wayback selected URL"},400)}let o=e.get("user"),s=await ge(o,"wayback_inventory",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:i.parsed.href}});if(!s.ok)return e.json(me(s),429,{"Retry-After":String(s.retryAfterSeconds)});let a=!1;try{let{ok:c,balance_mc:l}=await oe(o.id,D.url_map,N.URL_MAP,i.parsed.hostname);if(!c)return e.json(ce(l,D.url_map),402);a=!0;let d=await jT({url:i.parsed.href,scope:n.scope,urls:n.urls,from:n.from,to:n.to,successfulHtmlOnly:n.successfulHtmlOnly,maxCaptures:n.maxCaptures,includeCaptures:n.includeCaptures,maxCaptureRows:n.maxCaptureRows});return await Z({userId:o.id,source:"wayback_inventory",status:"done",query:d.url,resultCount:d.totalCaptures,result:{scope:d.scope,from:d.from,to:d.to,totalCaptures:d.totalCaptures,countType:d.countType,uniqueUrls:d.uniqueUrls,uniqueDigests:d.uniqueDigests}}),e.json(d)}catch(c){let l=c instanceof Error?c.message:String(c);return a&&await ie(o.id,D.url_map,N.URL_MAP_REFUND,"failed call"),await Z({userId:o.id,source:"wayback_inventory",status:"failed",query:i.parsed.href,error:l}),e.json({error:l},500)}finally{await te(s.lockId)}});M.post("/extract-site",re,async e=>{let t=await e.req.json().catch(()=>({})),r=hM.safeParse(t);if(!r.success)return e.json({error:r.error.issues[0]?.message??"Invalid request"},400);let n=r.data;if(n.semanticSimilarity&&!process.env.JINA_API_KEY?.trim())return e.json({error:"Semantic site similarity is not configured on this deployment.",errorCode:"semantic_similarity_unconfigured",retryable:!1},503);if(n.preserveMedia!==void 0&&n.downloadImages!==void 0&&n.preserveMedia!==n.downloadImages)return e.json({error:"preserveMedia conflicts with deprecated downloadImages."},400);let i=await fe(n.url,{field:"URL"});if(i.error||!i.parsed)return e.json({error:i.error??"Invalid URL"},400);let o=i.parsed,s=e.get("user"),a=ur(o.href),c=n.wayback?ub(n.wayback):[],l=a?.originalUrl??o.href;if(n.wayback?.urls){let S=new URL(l).hostname.replace(/^www\./,"").toLowerCase();for(let x of n.wayback.urls){let E=await fe(x,{field:"Wayback selected URL"});if(E.error||!E.parsed)return e.json({error:E.error??"Invalid Wayback selected URL"},400);if(E.parsed.hostname.replace(/^www\./,"").toLowerCase()!==S)return e.json({error:"Every selected Wayback URL must belong to the same site as url."},400)}}let d=Math.min(500,ph(n.maxPages)),u=n.wayback?Math.min(1e4,c.length*Math.min(d,n.wayback.urls?.length??d)):ph(n.maxPages),p=n.wayback?l:a?.rawReplayUrl??o.href,m=n.preserveMedia??n.downloadImages??!1;if(!!n.wayback||yB(n)){let S=e.req.header("idempotency-key")??e.req.header("x-idempotency-key");if(S==null)return e.json({error:"Idempotency-Key is required for background site extraction so a lost response can be retried without a second hold.",errorCode:"idempotency_key_required",retryable:!1},428);if(!S.trim()||S.length>500)return e.json({error:"Idempotency-Key must contain 1-500 characters.",errorCode:"invalid_idempotency_key",retryable:!1},400);let x=S.trim();e.header("Idempotency-Key",x);let E=`ext_${Hn(`${s.id}\0${x}`).slice(0,24)}`,v=Hn(JSON.stringify({url:o.href,waybackTimestamp:a?.timestamp??null,waybackOriginalUrl:a?.originalUrl??null,waybackTimeline:n.wayback??null,waybackMonths:c,waybackMaxPagesPerSnapshot:n.wayback?d:null,maxPages:u,rotateProxyEvery:n.rotateProxyEvery??10,formats:[...n.formats??[]].sort(),downloadImages:m,renderJavaScript:n.renderJavaScript===!0,captureRenderedDom:n.captureRenderedDom===!0,semanticSimilarity:n.semanticSimilarity===!0,similarityThreshold:n.similarityThreshold??null,similarityMaxPairs:n.similarityMaxPairs??null})),k=await ML({jobId:E,userId:s.id,balanceMc:s.balance_mc,startUrl:p,requestedMaxPages:u,idempotencyKey:x,requestFingerprint:v,concurrency:ia(s),urlsPerBrowser:n.rotateProxyEvery??10,formats:n.formats,downloadImages:m,renderJavaScript:n.renderJavaScript===!0,captureRenderedDom:n.captureRenderedDom===!0,semanticSimilarity:n.semanticSimilarity===!0,similarityThreshold:n.similarityThreshold,similarityMaxPairs:n.similarityMaxPairs,debitKey:`site-extract:${s.id}:${E}:hold`,waybackReplay:a&&!n.wayback?{timestamp:a.timestamp,originalUrl:a.originalUrl,replayUrl:a.replayUrl,rawReplayUrl:a.rawReplayUrl}:void 0,waybackTimeline:n.wayback?{rootUrl:l,timeline:n.wayback,maxPagesPerSnapshot:d}:void 0});if(k.kind==="conflict")return e.json({error:"Idempotency-Key was already used for different site extraction inputs.",errorCode:"idempotency_conflict",retryable:!1},409);if(k.kind==="insufficient")return e.json(ce(k.balanceMc,k.requiredMc),402);let A=k;if(["complete","partial","failed"].includes(A.job.status)){let H=Ws(A.job);return e.json({jobId:A.job.id,status:"pending",jobStatus:A.job.status,statusUrl:`/extract-site/status/${A.job.id}`,duplicate:!0,...H},202)}let C=Number(A.job.options.heldMc),T=typeof A.job.options.debitKey=="string"?A.job.options.debitKey:"";if(!Number.isSafeInteger(C)||C<=0||!T)return e.json({error:"Background crawl billing metadata is invalid.",errorCode:"extract_job_invalid",retryable:!1},500);let{ok:I,balance_mc:L}=await _t(s.id,C,N.EXTRACT_SITE_HOLD,o.hostname,T);if(!I)return await Ek(A.job.id,"Insufficient credits to fund the background crawl."),e.json(ce(L,C),402);try{await at.send({id:A.job.id,name:"mcp-scraper/extract.requested",data:{jobId:A.job.id}})}catch(H){return console.error("[extract-site/enqueue] dispatch pending retry:",H instanceof Error?H.message:String(H)),e.json({error:"Background crawl dispatch was not acknowledged; retry with the same Idempotency-Key or poll the job status.",errorCode:"extract_dispatch_pending",retryable:!0,jobId:A.job.id,statusUrl:`/extract-site/status/${A.job.id}`,...Ws(A.job)},503)}return e.json({jobId:A.job.id,status:"pending",statusUrl:`/extract-site/status/${A.job.id}`,duplicate:!A.created,...Ws(A.job)},202)}let f=Math.floor(s.balance_mc/D.page_scrape);if(f<1)return e.json(ce(s.balance_mc,D.page_scrape),402);let h=Math.min(Math.min(1e4,Math.max(1,n.maxPages??100)),f),y=h*D.page_scrape,b=await ge(s,"extract_site",{reuseLockId:e.req.header("x-mcp-scraper-concurrency-lock"),metadata:{url:o.href}});if(!b.ok)return e.json(me(b),429,{"Retry-After":String(b.retryAfterSeconds)});let _=!1;try{let{ok:S,balance_mc:x,billingKey:E}=await oe(s.id,y,N.EXTRACT_SITE_HOLD,o.hostname);return S?(_=!0,await pn({tool:"extract_site",op:"extract_site",userId:s.id,billingDebitKey:E,normalizedFlags:{maxPages:h,rotateProxies:n.rotateProxies===!0,browserFallback:!!(n.browserFallback||n.kernelFallback),renderJavaScript:!!n.renderJavaScript,captureRenderedDom:!!n.captureRenderedDom,semanticSimilarity:!!n.semanticSimilarity,waybackReplay:!!a}},async()=>{let v=n.rotateProxies===!0,k=n.formats?.includes("branding")===!0,A=a?await Pt("site_discovery",()=>Yp(a,h)):null,C=await rm({startUrl:p,maxPages:h,seedUrls:A?.map(F=>F.rawReplayUrl),kernelApiKey:v||k||n.browserFallback||n.kernelFallback||n.renderJavaScript||n.captureRenderedDom||n.semanticSimilarity?ae():void 0,formats:n.formats,forceBrowserRender:n.renderJavaScript||n.captureRenderedDom||n.semanticSimilarity,captureRenderedDom:n.captureRenderedDom,...v?{rotateProxyEvery:n.rotateProxyEvery??30,parallelism:ia(s)}:{}});a&&(C.startUrl=a.originalUrl,C.pages=C.pages.map(F=>{let U=ur(F.url);return U?{...F,archivedUrl:F.url,archiveTimestamp:U.timestamp,originalUrl:U.originalUrl,url:U.originalUrl,finalUrl:U.originalUrl}:F}));let T=C.pages?.length??1,L=C.pages.filter(rs).length*D.page_scrape,H=y-L;if(H>0)await ie(s.id,H,N.EXTRACT_SITE_REFUND,"overestimate refund");else if(H<0){let F=await oe(s.id,-H,N.EXTRACT_SITE,o.hostname);F.ok&&await un(F.billingKey,-H)}await Z({userId:s.id,source:"extract_site",status:"done",query:o.href,resultCount:T,result:C});let B=u>h;return e.json({...C,pages:C.pages.map(F=>{if(F.extractionStatus!=="failed")return F;let{failureCode:U,failureReason:W}=Ak(F.failureCode,F.failureReason);return{...F,failureCode:U,failureReason:W}}),requestedMaxPages:u,effectiveMaxPages:h,creditLimited:B,creditTruncated:B&&C.pages.length>=h})})):e.json(ce(x,y),402)}catch(S){let x=no(S);_&&await ie(s.id,y,N.EXTRACT_SITE_REFUND,"failed call"),await Z({userId:s.id,source:"extract_site",status:"failed",query:o.href,error:x.errorCode});let E=Zn(x,{chargeStatus:_?"refunded":"not_charged"});return e.json(E,iy(x))}finally{await te(b.lockId)}});M.get("/extract-site/status/:id",re,async e=>{let t=e.get("user"),r=await _i(e.req.param("id"));if(!r||r.userId!==t.id)return e.json({error:"Job not found"},404);let n=await Promise.all((r.artifacts??[]).map(async s=>{if(!s.key.startsWith(nk))return s;let a=await ik({artifactId:s.key,ownerId:String(t.id)}).catch(()=>null);return a?{...s,url:a.downloadUrl,expiresAt:a.expiresAt,downloadUrlExpiresAt:a.downloadUrlExpiresAt}:{...s,url:""}})),i=Ws(r),o=EH(r.status,r.publicError,!!r.error);return e.json({jobId:r.id,status:r.status,startUrl:r.startUrl,totalUrls:r.totalUrls,doneUrls:r.doneUrls,discovered:r.totalUrls,attempted:r.attemptedUrls,successful:r.successfulUrls,failed:r.failedUrls,remaining:r.remainingUrls,...i,artifacts:n,error:o?.message??null,...o??{},updatedAt:r.updatedAt})});M.post("/billing/checkout",cn,jo,async e=>{try{let t=e.get("sessionUser");if(!nb(t))return e.json({error:"One-time credit packs are available only on the Scale plan.",error_code:"scale_plan_required"},403);let r=await e.req.json().catch(()=>({})),n=wT(r.quantity),i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);let o=new sn(i,{apiVersion:an}),s=t.stripe_customer_id;s||(s=(await o.customers.create({email:t.email})).id,await qs(t.id,s));let a=_T(t.id,n),c=await o.checkout.sessions.create({customer:s,mode:"payment",payment_method_types:["card"],line_items:[{price_data:{currency:"usd",unit_amount:Nl*100,product_data:{name:"MCP Scraper Scale credit pack",description:`${cp.toLocaleString("en-US")} credits per $${Nl} increment`}},quantity:n}],metadata:a,payment_intent_data:{metadata:a,receipt_email:t.email},ui_mode:"embedded",automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},return_url:`${_n()}/billing?credit_pack=1&session_id={CHECKOUT_SESSION_ID}`});return e.json({clientSecret:c.client_secret,quantity:n,amountUsd:n*Nl,credits:n*cp})}catch(t){let r=t instanceof Error?t.message:"Unable to start credit-pack checkout.";return console.error("[billing/checkout]",r),r.startsWith("quantity must be")?e.json({error:r},400):e.json({error:r},500)}});M.post("/billing/concurrency/checkout",cn,jo,async e=>{try{let t=e.get("sessionUser"),r=await e.req.json().catch(()=>({})),n=yH(r.quantity),i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);let o=new sn(i,{apiVersion:an}),s=Ns(t),a=Ns(t,n),c=eo(t.subscription_tier);if(t.concurrency_stripe_sub_id){let u=await o.subscriptions.retrieve(t.concurrency_stripe_sub_id);if(t.stripe_customer_id&&Eh(u)!==t.stripe_customer_id)throw new Error("Concurrency subscription does not belong to this billing account.");let p=u.items.data.find(g=>g.price?.id===Za);if(!p)throw new Error("Concurrency subscription is missing its pack price.");let m=(p.quantity??1)!==n;return m&&await o.subscriptions.update(u.id,{items:[{id:p.id,quantity:n}],proration_behavior:"create_prorations"}),await El(t.id,n),e.json({updated:m,current:s,after:a,base_plan:c})}let l=t.stripe_customer_id;l||(l=(await o.customers.create({email:t.email})).id,await qs(t.id,l));let d=await o.checkout.sessions.create({customer:l,mode:"subscription",line_items:[{price:Za,quantity:n}],...wH(),ui_mode:"embedded",automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},return_url:`${_n()}/billing?slot_added=1`});return e.json({clientSecret:d.client_secret,current:s,after:a,base_plan:c})}catch(t){let r=t instanceof Error?t.message:"Unable to start checkout.";return console.error("[billing/concurrency/checkout]",r),r==="quantity must be a positive whole number"?e.json({error:r},400):e.json({error:r},500)}});async function AH(e,t,r,n,i){let o=Eh(r);if(!o||t.stripe_customer_id&&o!==t.stripe_customer_id)throw new Error("Subscription does not belong to this billing account.");let a=(await e.invoices.list({customer:o,status:"open",limit:100})).data.find(f=>(f.amount_remaining??0)>0);if(a)return{ok:!1,status:"outstanding_invoice",invoiceId:a.id,hostedInvoiceUrl:a.hosted_invoice_url,error:"Pay the outstanding invoice before changing plans."};let c=_p(r)?.price?.id,l=c?KE[c]:void 0,d=!l||i.credits_mc>l.credits_mc,u=await e.subscriptions.update(r.id,{items:[{id:n,price:i.price_id}],proration_behavior:"always_invoice",payment_behavior:"pending_if_incomplete",expand:["latest_invoice"]}),p=typeof u.latest_invoice=="object"&&u.latest_invoice&&!("deleted"in u.latest_invoice)?u.latest_invoice:u.latest_invoice?await e.invoices.retrieve(String(u.latest_invoice)):void 0;if(u.pending_update||!p||p.status!=="paid")return{ok:!1,status:"payment_required",invoiceId:p?.id,hostedInvoiceUrl:p?.hosted_invoice_url,error:"The prorated payment must succeed before the plan and Credits can be upgraded."};let m=await e.invoices.retrieve(p.id),g=await yT(e,m);if(d&&g.status==="not_applicable")throw new Error("The paid upgrade invoice did not produce the required Credit grant.");return await Bh(t.id,i.tier,i.concurrency,r.id),{ok:!0,status:g.status==="already_fulfilled"?"already_fulfilled":"fulfilled",invoiceId:p.id,hostedInvoiceUrl:m.hosted_invoice_url,amountPaidUsd:(m.amount_paid??0)/100,amountSettledUsd:(m.total??0)/100,creditsAdded:g.status==="not_applicable"?0:i.credits_mc/rt}}M.get("/billing/offer/:slug",async e=>{let t=GE(e.req.param("slug").trim().toLowerCase());return t?e.json(t):e.json({error:"Unknown offer."},404)});M.post("/billing/subscribe",cn,jo,async e=>{try{let t=e.get("sessionUser"),r=await e.req.json().catch(()=>({})),n=ap[String(r.tier??"")];if(!n)return e.json({error:"Invalid tier. Choose starter, growth, or scale."},400);if(ry(t.subscription_tier,n.tier)&&r.confirm_offer_change!==!0){let l=eo(t.subscription_tier),d=ty(t.subscription_tier);return e.json({error:`You are on the ${d?.label??"partner offer"} (${l?.label??t.subscription_tier}, $${l?.amount_usd??"?"} per ${l?.interval??"year"}). Switching to ${n.label} ends that offer and its non-expiring Credits, and it cannot be re-claimed automatically. Confirm to continue.`,error_code:"offer_change_requires_confirmation",current_plan:l,requested_plan:eo(n.tier)},409)}let i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);let o=new sn(i,{apiVersion:an}),s=t.stripe_customer_id;s||(s=(await o.customers.create({email:t.email})).id,await qs(t.id,s));let a=t.subscription_id;if(!a)try{a=await bH(o,s)??null}catch(l){console.warn("[billing/subscribe] could not check for existing subscription",l instanceof Error?l.message:l)}if(a){let l=await o.subscriptions.retrieve(a),d=_p(l);if(d?.price?.id===n.price_id)return e.json({updated:!1,tier:n.tier,already_current:!0});let u=d?.id;if(u){let p=await AH(o,t,l,u,n);return p.ok?e.json({updated:!0,tier:n.tier,billed:!0,amount_paid_usd:p.amountPaidUsd,amount_settled_usd:p.amountSettledUsd,credits_added:p.creditsAdded,invoice_id:p.invoiceId}):e.json({error:p.error,error_code:p.status,invoice_id:p.invoiceId,hosted_invoice_url:p.hostedInvoiceUrl},p.status==="outstanding_invoice"?409:402)}}let c=await o.checkout.sessions.create({customer:s,mode:"subscription",line_items:[{price:n.price_id,quantity:1}],...n.intro_coupon?{discounts:[{coupon:n.intro_coupon}]}:{},ui_mode:"embedded",automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},return_url:n.private_offer&&ey(n.tier)?`${_n()}${ey(n.tier)}?purchased=${n.tier}&session_id={CHECKOUT_SESSION_ID}`:`${_n()}/billing?subscribed=${n.tier}&session_id={CHECKOUT_SESSION_ID}`});return e.json({clientSecret:c.client_secret})}catch(t){let r=t instanceof Error?t.message:"Unable to start subscription.";return console.error("[billing/subscribe]",r),e.json({error:r},500)}});M.post("/billing/checkout/confirm",cn,jo,async e=>{let t=e.get("sessionUser"),r=await e.req.json().catch(()=>({})),n=typeof r.session_id=="string"?r.session_id.trim():"";if(!/^cs_(test_|live_)?[A-Za-z0-9]+$/.test(n))return e.json({error:"A valid Checkout session id is required."},400);let i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);try{let o=new sn(i,{apiVersion:an});if((await o.checkout.sessions.retrieve(n)).mode==="payment"){let c=await vT(o,n,t.stripe_customer_id);return c.status==="not_applicable"?e.json({ok:!1,status:"pending",message:"Stripe has not confirmed this payment yet."},202):e.json({ok:!0,status:c.status,credits:c.credits})}let a=await bT(o,n,t.stripe_customer_id);return a.status==="not_applicable"?e.json({ok:!1,status:"pending",message:"Stripe has not confirmed this payment yet."},202):e.json({ok:!0,status:a.status,tier:a.tier})}catch(o){let s=o instanceof Error?o.message:"Unable to confirm checkout.";return s.includes("does not belong")||s.includes("not a subscription")?e.json({error:s},409):(console.error("[billing/checkout/confirm]",n,s),e.json({error:"We could not confirm this checkout yet. Please retry."},503))}});M.post("/billing/portal",cn,jo,async e=>{try{let t=e.get("sessionUser");if(!t.stripe_customer_id)return e.json({error:"No billing account yet \u2014 subscribe first."},409);let r=Yx();if(!r)return e.json({error:"Stripe is not configured."},503);let i=await new sn(r,{apiVersion:an}).billingPortal.sessions.create({customer:t.stripe_customer_id,return_url:`${_n()}/billing`});return e.json({url:i.url})}catch(t){let r=t instanceof Error?t.message:"Unable to open billing portal.";return console.error("[billing/portal]",r),e.json({error:r},500)}});M.post("/billing/concurrency/terminal-checkout",re,async e=>{try{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=yH(r.quantity),i=Ll(),o=Ns(t),s=Ns(t,n),a=eo(t.subscription_tier);if(t.concurrency_stripe_sub_id){let p=process.env.STRIPE_SECRET_KEY?.trim();if(!p)return e.json({error:"Stripe is not configured."},503);let m=new sn(p,{apiVersion:an}),g=await m.subscriptions.retrieve(t.concurrency_stripe_sub_id);if(t.stripe_customer_id&&Eh(g)!==t.stripe_customer_id)throw new Error("Concurrency subscription does not belong to this billing account.");let f=g.items.data.find(y=>y.price?.id===Za);if(!f)throw new Error("Concurrency subscription is missing its pack price.");let h=(f.quantity??1)!==n;return h&&await m.subscriptions.update(g.id,{items:[{id:f.id,quantity:n}],proration_behavior:"create_prorations"}),await El(t.id,n),e.json({updated:h,price:i,current:o,after:s,base_plan:a,next_step:h?"The concurrency pack quantity was updated.":"The concurrency pack quantity was already current."})}let c=process.env.STRIPE_SECRET_KEY?.trim();if(!c)return e.json({error:"Stripe is not configured."},503);let l=new sn(c,{apiVersion:an}),d=t.stripe_customer_id;d||(d=(await l.customers.create({email:t.email})).id,await qs(t.id,d));let u=await l.checkout.sessions.create({customer:d,mode:"subscription",line_items:[{price:Za,quantity:n}],...wH(),automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},success_url:`${_n()}/billing?slot_added=1&source=terminal`,cancel_url:`${_n()}/billing?slot_cancelled=1&source=terminal`});return u.url?e.json({checkout_url:u.url,price:i,current:o,after:s,base_plan:a,next_step:"Open checkout_url in a browser, complete checkout, then restart or retry the MCP request."}):e.json({error:"Stripe did not return a checkout URL."},502)}catch(t){let r=t instanceof Error?t.message:"Unable to start terminal checkout.";return console.error("[billing/concurrency/terminal-checkout]",r),r==="quantity must be a positive whole number"?e.json({error:r},400):e.json({error:r},500)}});M.post("/billing/subscribe/terminal-checkout",re,async e=>{try{let t=e.get("user"),r=await e.req.json().catch(()=>({})),n=ap[String(r.tier??"")];if(!n)return e.json({error:"Invalid tier. Choose starter, growth, or scale.",tiers:Object.values(ap).filter(l=>!l.private_offer).map(l=>l.tier)},400);if(ry(t.subscription_tier,n.tier)&&r.confirm_offer_change!==!0){let l=eo(t.subscription_tier),d=ty(t.subscription_tier);return e.json({error:`You are on the ${d?.label??"partner offer"} (${l?.label??t.subscription_tier}, $${l?.amount_usd??"?"} per ${l?.interval??"year"}). Switching to ${n.label} ends that offer and its non-expiring Credits, and it cannot be re-claimed automatically. Retry with confirm_offer_change: true to continue.`,error_code:"offer_change_requires_confirmation",current_plan:l,requested_plan:eo(n.tier)},409)}let i=process.env.STRIPE_SECRET_KEY?.trim();if(!i)return e.json({error:"Stripe is not configured."},503);let o=new sn(i,{apiVersion:an}),s=t.stripe_customer_id;s||(s=(await o.customers.create({email:t.email})).id,await qs(t.id,s));let a=t.subscription_id;if(!a)try{a=await bH(o,s)??null}catch(l){console.warn("[billing/subscribe] could not check for existing subscription",l instanceof Error?l.message:l)}if(a){let l=await o.subscriptions.retrieve(a),d=_p(l);if(d?.price?.id===n.price_id)return e.json({updated:!1,tier:n.tier,already_current:!0,message:`${n.label} is already active.`});let u=d?.id;if(u){let p=await AH(o,t,l,u,n);return p.ok?e.json({updated:!0,tier:n.tier,billed:!0,amount_paid_usd:p.amountPaidUsd,amount_settled_usd:p.amountSettledUsd,credits_added:p.creditsAdded,invoice_id:p.invoiceId,message:`Switched to ${n.label}, settled the prorated payment immediately, and added ${Number(p.creditsAdded??0).toLocaleString("en-US")} Credits.`}):e.json({error:p.error,error_code:p.status,invoice_id:p.invoiceId,hosted_invoice_url:p.hostedInvoiceUrl},p.status==="outstanding_invoice"?409:402)}}let c=await o.checkout.sessions.create({customer:s,mode:"subscription",line_items:[{price:n.price_id,quantity:1}],...n.intro_coupon?{discounts:[{coupon:n.intro_coupon}]}:{},automatic_tax:{enabled:!0},billing_address_collection:"required",customer_update:{address:"auto",name:"auto"},tax_id_collection:{enabled:!0},success_url:`${_n()}/billing?subscribed=${n.tier}&source=terminal&session_id={CHECKOUT_SESSION_ID}`,cancel_url:`${_n()}/billing?subscribe_cancelled=1&source=terminal`});return c.url?e.json({checkout_url:c.url,tier:n.tier,label:n.label,monthly_usd:n.monthly_usd,amount_usd:WE(n),billing_interval:HE(n),credits_per_interval:zE(n),credits_per_month:n.credits_mc/rt,credits_never_expire:!!n.credits_never_expire,includes_memory:!!n.includes_memory,concurrency:n.concurrency,intro:n.intro_coupon?"$1 first month":null,next_step:"Open checkout_url in a browser to complete payment."}):e.json({error:"Stripe did not return a checkout URL."},502)}catch(t){let r=t instanceof Error?t.message:"Unable to start subscription checkout.";return console.error("[billing/subscribe/terminal-checkout]",r),e.json({error:r},500)}});M.post("/billing/concurrency/cancel",cn,jo,async e=>{let t=e.get("sessionUser");if(!t.concurrency_stripe_sub_id)return e.json({error:"No active concurrency subscription."},404);let r=process.env.STRIPE_SECRET_KEY?.trim();if(!r)return e.json({error:"Stripe is not configured."},503);let n=new sn(r,{apiVersion:an}),i=await n.subscriptions.retrieve(t.concurrency_stripe_sub_id);if(t.stripe_customer_id&&Eh(i)!==t.stripe_customer_id)return e.json({error:"Concurrency subscription does not belong to this billing account."},409);if(!i.items.data.some(s=>s.price?.id===Za))return e.json({error:"The saved subscription is not a concurrency pack. No subscription was canceled."},409);await n.subscriptions.cancel(t.concurrency_stripe_sub_id),await El(t.id,0),await Fh(t.id,null);let o=Ns(t,0);return e.json({ok:!0,...o,base_plan:eo(t.subscription_tier)})});M.get("/billing/balance",re,async e=>{let t=e.get("user"),r=t.balance_mc,n=await Qu(t.id,20),i=await HB(t.id),o=Ns(t);return e.json({balance_mc:r,balance_credits:r/rt,free_credits:i,ledger:n,concurrency_pack_quantity:o.pack_quantity,slots_per_pack:o.slots_per_pack,extra_concurrency_slots:o.extra_concurrency_slots,concurrency_limit:o.effective_limit,concurrency_addon_monthly_usd:o.monthly_amount_usd,concurrency:{...o,has_subscription:!!t.concurrency_stripe_sub_id},scale_credit_pack:{eligible:nb(t),increment_usd:Nl,credits_per_increment:cp,max_quantity:ZE,expiry_months:3}})});M.get("/billing/summary",re,ZL);M.get("/billing/connected-usage/history",re,async e=>e.json({ok:!0,ratePolicy:Zh,receipts:await xA(e.get("user").id,100)}));M.post("/billing/credits",re,async e=>{let t=e.get("user"),r=t.balance_mc,n=await e.req.json().catch(()=>({})),i=n.item?.trim().toLowerCase(),o=Qh.map(({aliases:c,...l})=>l),s=i?Qh.find(c=>c.label.toLowerCase().includes(i)||c.key.toLowerCase()===i||c.aliases.some(l=>l.toLowerCase().includes(i)||i.includes(l.toLowerCase()))):void 0,a=n.includeLedger?(await Qu(t.id,10)).map(c=>({amount_mc:c.amount_mc,operation:c.operation,description:c.description,created_at:c.created_at})):void 0;return e.json({balance_mc:r,balance_credits:r/rt,item:n.item??null,matched_cost:s?(({aliases:c,...l})=>l)(s):null,costs:o,concurrency:{...Ns(t),has_subscription:!!t.concurrency_stripe_sub_id,upgrade:Ll()},connected_accounts:Zh,ledger:a})});M.post("/api/internal/extract-refinalize/:id",async e=>{let t=e.req.header("authorization");if(!process.env.CRON_SECRET||t!==`Bearer ${process.env.CRON_SECRET}`)return e.json({error:"Unauthorized"},401);let r=e.req.param("id"),{claimFailedExtractJobForRefinalize:n,failExtractJob:i,getExtractJob:o,finishExtractJob:s,terminalExtractJobStatus:a,extractJobLimitInfo:c}=await import("./site-extract-repository-2LFDP6FZ.js"),{assembleExtractArtifacts:l}=await import("./extract-bundle-C3N6E6V7.js"),d=await o(r);if(!d)return e.json({error:"job not found"},404);if(d.status!=="failed")return e.json({error:"Only failed site extraction jobs can be re-finalized.",status:d.status},409);let u=await n(r);if(!u){let S=await o(r);return e.json({error:"job state changed during re-finalization claim",status:S?.status??"missing"},409)}let p;try{p=await l(u,{})}catch(S){throw await i(r,S instanceof Error?S.message:String(S)).catch(()=>{}),S}let m=await o(r);if(!m)return e.json({error:"job disappeared during finalization"},500);let g=c(m),f=a(m,g.creditTruncated),h=f==="failed"?`No pages were extracted successfully (${m.failedUrls} failed, ${m.remainingUrls} remaining).`:f==="partial"?`${m.successfulUrls} pages succeeded; ${m.failedUrls} failed and ${m.remainingUrls} remain.${g.creditTruncated?` Credit availability limited this request from ${g.requestedMaxPages} to ${g.effectiveMaxPages} pages.`:""}`:null;if(!await s(r,p,f,h))return e.json({error:"job state changed during re-finalization"},409);let b="already_settled_or_refunded";if(u.billedMc==null&&u.userId!=null){let{settleExtractJob:S,countSuccessfulPages:x}=await import("./site-extract-repository-2LFDP6FZ.js"),E=Number(u.options.heldMc??0),v=await x(r),k=Math.min(v*D.page_scrape,E),A="site";try{A=new URL(u.startUrl).hostname}catch{}await S(r,u.userId,E-k,k,A),b=`settled: billed ${k} mc, refunded ${E-k} mc`}let _=p.find(S=>S.contentType==="application/zip"||S.filename?.endsWith(".zip"));return e.json({ok:!0,jobId:r,artifacts:p.length,settlement:b,bundleUrl:_?.url??null,bundleBytes:_?.bytes??null})});M.on(["GET","POST","PUT"],"/api/inngest",jae({client:at,functions:[MP,qP,QP,jN,rL,LL]}));M.route("/",Nu);M.route("/admin/credits",ds);M.route("/api/internal/site-architecture-auditor",fn);M.route("/api/internal/memory",ow);M.route("/api/internal/billing",Ri);M.route("/api/internal/inbox",$m);M.route("/inbox",ga);M.route("/youtube",zm);M.route("/screenshot",Gm);M.route("/facebook",yo);M.route("/tiktok",Pw);M.route("/google-ads",Ad);M.route("/instagram",cg);M.route("/reddit",fg);M.route("/kernel-reddit",hg);M.route("/",gr);M.route("/video",wg);M.route("/maps",vg);M.route("/trustpilot",u_);M.route("/g2",m_);M.route("/directory",Eg);M.route("/lead-list-input",qd);M.route("/lead-list-enrichment",Ud);M.route("/local-sourcebook",yn);M.route("/public/local-sourcebook",Lc);M.route("/admin/local-sourcebook",Mc);M.route("/locations",Oc);M.route("/workflows",Qr);M.route("/serp-intelligence",Hd);M.route("/mcp",G_);M.route("/agent",nU());M.route("/chat",Rr);M.route("/vault",qi);M.route("/schedule",ai);M.route("/resend",Iv);M.route("/workshops",mf);M.route("/editorial-reading-room",ru);M.route("/commons",Me);M.route("/analytics",O);M.route("/crm",lt);M.route("/research",ml);M.route("/",Or);M.route("/public",uh);M.route("/api/internal/scheduled-artifacts",Ax);M.get("/console",e=>e.html(sv()));M.use("/console/auth/*",re);M.get("/console/auth/:id",async e=>{let t=e.req.param("id");await vd();let r=await Sc(e.get("user").id,t),n=r?.browser_agent_session_id?await fa(r.browser_agent_session_id):null;return e.html(iU(t,{domain:r?.domain??null,status:r?.status??null,liveViewUrl:n?.live_view_url??null,sessionOpen:n?.status==="open"}))});M.post("/console/auth/:id/complete",async e=>{let t=e.req.param("id");await vd();let r=await tU(e.get("user").id,t);return r.ok?e.json({ok:!0}):e.json({error:r.error},404)});M.get("/console/:id",e=>e.html(sv(e.req.param("id"))));M.route("/stripe",xT);M.route("/",Hr);process.env.INNGEST_EVENT_KEY||(hB(),NL());process.env.NODE_ENV!=="production"&&M.get("/font-test",e=>e.html(`<!DOCTYPE html>
|
|
4885
4885
|
<html lang="en">
|
|
4886
4886
|
<head>
|
|
4887
4887
|
<meta charset="UTF-8">
|