awwwards-mcp 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/server.js ADDED
@@ -0,0 +1,488 @@
1
+ import { parseCategories, parseDetail, parseElements, parseListing } from "./parsers.js";
2
+ import { BlockedError, buildFilterUrl, elementPosterPath, elementUrl, } from "./awwwards.js";
3
+ const AWARD_FILTER_LABELS = {
4
+ sotd: "Site of the Day",
5
+ developer: "Developer Award",
6
+ honorable: "Honorable Mention",
7
+ };
8
+ export const SITE_TTL_MS = 7 * 24 * 60 * 60 * 1000;
9
+ export const CATEGORY_TTL_MS = 30 * 24 * 60 * 60 * 1000;
10
+ // The text block lists every element; only this many posters are fetched inline.
11
+ export const MAX_INLINE_POSTERS = 8;
12
+ export function slugifyTag(s) {
13
+ return s.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
14
+ }
15
+ // Free-text queries match token-wise: every whitespace-separated token must
16
+ // substring-match the site's title+tags. A single token behaves exactly like
17
+ // the old whole-query substring check.
18
+ export function tokenizeQuery(q) {
19
+ return q.toLowerCase().split(/\s+/).filter(Boolean);
20
+ }
21
+ // Zero-result suggestions: rank taxonomy slugs against the query's tokens.
22
+ // +2 per token (>=4 chars) the slug contains; +1 when slug and token share a
23
+ // >=4-char prefix (compared via their first 5 chars, in either direction).
24
+ // Ties rank stably by slug; only positive scores are suggested.
25
+ export function suggestTags(tokens, taxonomy, limit = 6) {
26
+ const scored = [];
27
+ for (const slug of taxonomy) {
28
+ let score = 0;
29
+ for (const t of tokens) {
30
+ if (t.length < 4)
31
+ continue;
32
+ if (slug.includes(t))
33
+ score += 2;
34
+ else if (slug.startsWith(t.slice(0, 5)) || t.startsWith(slug.slice(0, 5)))
35
+ score += 1;
36
+ }
37
+ if (score > 0)
38
+ scored.push([slug, score]);
39
+ }
40
+ return scored
41
+ .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))
42
+ .slice(0, limit)
43
+ .map(([s]) => s);
44
+ }
45
+ function text(t) {
46
+ return { type: "text", text: t };
47
+ }
48
+ function summarizeSite(s) {
49
+ const award = s.awards.length ? ` [${s.awards.join(", ")}]` : "";
50
+ return `- ${s.title} (slug: ${s.slug})${award}\n live: ${s.liveUrl ?? "unknown"}\n awwwards: https://www.awwwards.com${s.detailPath}\n tags: ${s.tags.join(", ")}`;
51
+ }
52
+ // An all-empty parse means the layout changed (or the page was not found):
53
+ // neither tool may cache such a parse, so the mismatch can still be surfaced.
54
+ function isAllEmptyDetail(d) {
55
+ return (d.palette.length === 0 &&
56
+ d.technologies.length === 0 &&
57
+ d.elements.length === 0 &&
58
+ d.awards.length === 0 &&
59
+ !d.description);
60
+ }
61
+ function errorResponse(err) {
62
+ const message = err instanceof BlockedError
63
+ ? err.message
64
+ : `awwwards-mcp request failed: ${err instanceof Error ? err.message : String(err)}`;
65
+ return { content: [text(message)], isError: true };
66
+ }
67
+ export function createHandlers(deps) {
68
+ const { client, cache } = deps;
69
+ // Which filter wins the URL (combined filter URLs 404 on awwwards.com).
70
+ const urlSource = (f) => f.color ? "color" : f.award ? "award" : f.technology ? "technology" : f.tags?.length ? "tag" : "none";
71
+ // Two modes:
72
+ // - honorUrlSource=true: rows freshly scraped from the filter page — the URL
73
+ // really did apply the highest-priority filter (color > award > technology
74
+ // > first tag), so skip re-checking that one and verify the rest.
75
+ // - honorUrlSource=false: cache/index rows — nothing guarantees the URL
76
+ // filter was applied, so check every client-checkable filter. Color is
77
+ // never client-checkable (site rows carry no colors); it is handled by
78
+ // never serving color searches from cache (see search_sites).
79
+ function matchesFilters(s, f, honorUrlSource) {
80
+ const source = urlSource(f);
81
+ if (f.tags?.length) {
82
+ const tagsToCheck = honorUrlSource && source === "tag" ? f.tags.slice(1) : f.tags;
83
+ for (const t of tagsToCheck) {
84
+ const slug = t.toLowerCase();
85
+ if (!s.tags.some((st) => slugifyTag(st).includes(slug)))
86
+ return false;
87
+ }
88
+ }
89
+ if (f.technology && !(honorUrlSource && source === "technology")) {
90
+ const slug = f.technology.toLowerCase();
91
+ if (!s.tags.some((st) => slugifyTag(st).includes(slug)))
92
+ return false;
93
+ }
94
+ if (f.award && !(honorUrlSource && source === "award")) {
95
+ if (!s.awards.includes(AWARD_FILTER_LABELS[f.award]))
96
+ return false;
97
+ }
98
+ if (f.query) {
99
+ const queryTokens = tokenizeQuery(f.query);
100
+ if (queryTokens.length) {
101
+ const hay = (s.title + " " + s.tags.join(" ")).toLowerCase();
102
+ if (!queryTokens.every((t) => hay.includes(t)))
103
+ return false;
104
+ }
105
+ }
106
+ return true;
107
+ }
108
+ async function siteImage(s) {
109
+ try {
110
+ const buf = await cache.getImage(s.thumbnailPath, () => client.getThumbnail(s.thumbnailPath, 880));
111
+ return { type: "image", data: buf.toString("base64"), mimeType: "image/jpeg" };
112
+ }
113
+ catch {
114
+ return null; // thumbnail failures degrade to metadata-only cards
115
+ }
116
+ }
117
+ // sortBy: "score" view. Scores live in detail-meta entries written by
118
+ // get_site_details, so this reads the cache only — never a live fetch during
119
+ // search. SiteDetails gains its score field in a later v1.5.0 task, so the
120
+ // meta read is typed locally. Scored sites come first (desc); unscored ones
121
+ // follow, newest-first.
122
+ function orderByScore(list, sortBy) {
123
+ if (sortBy !== "score")
124
+ return list;
125
+ const withScores = list.map((s) => ({
126
+ s,
127
+ score: cache.getMeta(`detail:${s.slug}`, SITE_TTL_MS)?.score ?? null,
128
+ }));
129
+ withScores.sort((a, b) => (b.score ?? -1) - (a.score ?? -1) || b.s.createdAt - a.s.createdAt);
130
+ return withScores.map((w) => w.s);
131
+ }
132
+ async function search_sites(args) {
133
+ const count = Math.min(Math.max(args.count ?? 6, 1), 12);
134
+ const page = Math.max(args.page ?? 1, 1);
135
+ try {
136
+ // Color can't be verified client-side (site rows carry no colors), so a
137
+ // color search always scrapes its filter page; everything else is
138
+ // client-checkable against the index.
139
+ let sites = args.color
140
+ ? []
141
+ : cache.getSites(SITE_TTL_MS).filter((s) => matchesFilters(s, args, false));
142
+ // A scrape normally REPLACES the matched rows (fresh rows are only
143
+ // guaranteed the URL filter), so it must run when the cache cannot serve
144
+ // the requested page window at all. When the window is only PARTIALLY
145
+ // filled (e.g. a tokenized query matched 3 cached rows for count 6),
146
+ // replacing would discard already-verified matches — those runs scrape
147
+ // the filter page and MERGE instead: dedupe by slug, verified cache rows
148
+ // win duplicate slugs, scraped-only rows appended, all newest-first like
149
+ // getSites; the merge result is client-checked as before (cache rows
150
+ // already passed matchesFilters(false), scraped rows matchesFilters(true)).
151
+ const pageWindowEmpty = sites.slice((page - 1) * count, page * count).length === 0;
152
+ const pageWindowPartial = !pageWindowEmpty && sites.length < page * count;
153
+ if (pageWindowEmpty) {
154
+ const html = await client.getHtml(buildFilterUrl(args));
155
+ const parsed = parseListing(html);
156
+ if (parsed.length === 0) {
157
+ return {
158
+ content: [
159
+ text("Awwwards layout may have changed: parsed 0 site cards. " +
160
+ "The awwwards-mcp parser likely needs an update."),
161
+ ],
162
+ isError: true,
163
+ };
164
+ }
165
+ cache.upsertSites(parsed);
166
+ // Freshly parsed rows carry the URL filter by construction
167
+ // (honorUrlSource: true); serve only those, newest-first like getSites.
168
+ sites = parsed
169
+ .filter((s) => matchesFilters(s, args, true))
170
+ .sort((a, b) => b.createdAt - a.createdAt);
171
+ }
172
+ else if (pageWindowPartial) {
173
+ // Top-up scrape: best-effort in the fullest sense — a fetch failure
174
+ // (HTTP error, BlockedError) must not escape to the outer catch, whose
175
+ // stale-fallback always slices page 1 and would silently discard the
176
+ // requested page window's cached rows. Keep those rows instead.
177
+ try {
178
+ const html = await client.getHtml(buildFilterUrl(args));
179
+ const parsed = parseListing(html);
180
+ if (parsed.length > 0) {
181
+ cache.upsertSites(parsed);
182
+ const fresh = parsed.filter((s) => matchesFilters(s, args, true));
183
+ const bySlug = new Map();
184
+ // Scraped rows seed the map; verified cache rows then overwrite any
185
+ // duplicate slug, so they always win.
186
+ for (const s of fresh)
187
+ bySlug.set(s.slug, s);
188
+ for (const s of sites)
189
+ bySlug.set(s.slug, s);
190
+ sites = [...bySlug.values()].sort((a, b) => b.createdAt - a.createdAt);
191
+ }
192
+ }
193
+ catch {
194
+ /* keep cached rows; an empty parse is equally tolerated above */
195
+ }
196
+ }
197
+ const ordered = orderByScore(sites, args.sortBy);
198
+ const slice = ordered.slice((page - 1) * count, page * count);
199
+ if (slice.length === 0) {
200
+ if (sites.length === 0) {
201
+ // True zero-result search: suggest the closest taxonomy tags. The
202
+ // taxonomy comes from the cache only — never a live fetch just to
203
+ // phrase a suggestion; without a cached taxonomy, keep today's text.
204
+ const cats = cache.getMeta("categories", CATEGORY_TTL_MS);
205
+ const suggestions = cats
206
+ ? suggestTags(tokenizeQuery(args.query ?? ""), cats.filters)
207
+ : [];
208
+ if (suggestions.length > 0) {
209
+ return {
210
+ content: [
211
+ text(`No sites matched the search. Closest filter tags: ${suggestions.join(", ")}. ` +
212
+ "Run list_categories for the full taxonomy."),
213
+ ],
214
+ };
215
+ }
216
+ }
217
+ return {
218
+ content: [
219
+ text("No sites matched the search on this page. Try fewer filters or run list_categories. " +
220
+ "(Deep pagination is unavailable by design: awwwards.com's robots.txt disallows it.)"),
221
+ ],
222
+ };
223
+ }
224
+ const images = await Promise.all(slice.map(siteImage));
225
+ const content = [
226
+ text(`${ordered.length} site(s) matched; showing ${(page - 1) * count + 1}-${(page - 1) * count + slice.length}:\n\n` +
227
+ slice.map(summarizeSite).join("\n\n")),
228
+ ...images.filter((b) => b !== null),
229
+ ];
230
+ return { content };
231
+ }
232
+ catch (err) {
233
+ // Spec: on live-request failure, serve stale cache if present. The store
234
+ // itself may be the failure source, so this lookup is guarded too.
235
+ let stale = [];
236
+ try {
237
+ stale = cache.getSites(Infinity).filter((s) => matchesFilters(s, args, false));
238
+ }
239
+ catch {
240
+ stale = [];
241
+ }
242
+ if (stale.length > 0) {
243
+ const orderedStale = orderByScore(stale, args.sortBy);
244
+ const slice = orderedStale.slice(0, count);
245
+ const images = await Promise.all(slice.map(siteImage));
246
+ return {
247
+ content: [
248
+ text(`The live awwwards.com request failed (${err instanceof Error ? err.message : String(err)}). ` +
249
+ `Serving ${slice.length} result(s) from stale cache instead:\n\n` +
250
+ slice.map(summarizeSite).join("\n\n")),
251
+ ...images.filter((b) => b !== null),
252
+ ],
253
+ };
254
+ }
255
+ return errorResponse(err);
256
+ }
257
+ }
258
+ async function get_site_details(args) {
259
+ try {
260
+ const metaKey = `detail:${args.slug}`;
261
+ let d = cache.getMeta(metaKey, SITE_TTL_MS);
262
+ if (!d) {
263
+ const html = await client.getHtml(`/sites/${args.slug}`);
264
+ d = parseDetail(html, args.slug);
265
+ if (isAllEmptyDetail(d)) {
266
+ return {
267
+ content: [
268
+ text("Awwwards layout may have changed: parsed no design data for " +
269
+ args.slug +
270
+ ". The awwwards-mcp parser likely needs an update (or the site page was not found)."),
271
+ ],
272
+ isError: true,
273
+ };
274
+ }
275
+ cache.setMeta(metaKey, d);
276
+ // One fetch feeds both caches: seed the elements cache from the same
277
+ // HTML. Null (no section) caches as a legitimate empty; a zero-blob
278
+ // parse is left uncached for get_site_elements to surface as a mismatch.
279
+ if (cache.getMeta(`elements:${args.slug}`, SITE_TTL_MS) === null) {
280
+ const els = parseElements(html);
281
+ if (els === null)
282
+ cache.setMeta(`elements:${args.slug}`, []);
283
+ else if (els.length > 0)
284
+ cache.setMeta(`elements:${args.slug}`, els);
285
+ }
286
+ }
287
+ const cachedSite = cache.getSite(args.slug, SITE_TTL_MS);
288
+ const liveUrl = d.liveUrl ?? cachedSite?.liveUrl ?? null;
289
+ const content = [
290
+ text([
291
+ `# ${d.title ?? args.slug}`,
292
+ liveUrl ? `Live site: ${liveUrl}` : null,
293
+ d.awards.length
294
+ ? `Awards: ${d.awards.map((a) => `${a.title} (${a.date})`).join(", ")}`
295
+ : null,
296
+ d.score != null ? `Jury score: ${d.score.toFixed(2)}/10` : null,
297
+ d.palette.length ? `Color palette: ${d.palette.join(", ")}` : null,
298
+ d.technologies.length ? `Technologies & tools: ${d.technologies.join(", ")}` : null,
299
+ d.elements.length ? `Design elements: ${d.elements.join(", ")}` : null,
300
+ d.description ? `Description: ${d.description}` : null,
301
+ d.ogImage ? `Full-size screenshot: ${d.ogImage}` : null,
302
+ ]
303
+ .filter(Boolean)
304
+ .join("\n")),
305
+ ];
306
+ if (cachedSite) {
307
+ const img = await siteImage(cachedSite);
308
+ if (img)
309
+ content.push(img);
310
+ }
311
+ return { content };
312
+ }
313
+ catch (err) {
314
+ return errorResponse(err);
315
+ }
316
+ }
317
+ async function get_site_elements(args) {
318
+ try {
319
+ const elementsKey = `elements:${args.slug}`;
320
+ let elements = cache.getMeta(elementsKey, SITE_TTL_MS);
321
+ if (elements === null) {
322
+ const html = await client.getHtml(`/sites/${args.slug}`);
323
+ const parsed = parseElements(html);
324
+ if (parsed === null) {
325
+ elements = [];
326
+ cache.setMeta(elementsKey, elements);
327
+ }
328
+ else if (parsed.length === 0) {
329
+ return {
330
+ content: [
331
+ text("Awwwards layout may have changed: found an Elements section but parsed 0 elements. " +
332
+ "The awwwards-mcp parser likely needs an update."),
333
+ ],
334
+ isError: true,
335
+ };
336
+ }
337
+ else {
338
+ elements = parsed;
339
+ cache.setMeta(elementsKey, elements);
340
+ }
341
+ // One fetch feeds both caches: seed the detail cache from the same
342
+ // HTML unless it is an all-empty parse (never cached, per contract).
343
+ if (cache.getMeta(`detail:${args.slug}`, SITE_TTL_MS) === null) {
344
+ const d = parseDetail(html, args.slug);
345
+ if (!isAllEmptyDetail(d))
346
+ cache.setMeta(`detail:${args.slug}`, d);
347
+ }
348
+ }
349
+ const cachedSite = cache.getSite(args.slug, SITE_TTL_MS);
350
+ const title = cache.getMeta(`detail:${args.slug}`, SITE_TTL_MS)?.title ??
351
+ cachedSite?.title ??
352
+ args.slug;
353
+ if (elements.length === 0) {
354
+ return { content: [text(`No design elements listed for ${title} (${args.slug}).`)] };
355
+ }
356
+ const shown = elements.slice(0, MAX_INLINE_POSTERS);
357
+ const lines = elements.map((el, i) => {
358
+ const isVideo = el.mediaPath.endsWith(".mp4");
359
+ return `${i + 1}. ${el.title} (${isVideo ? "video" : "image"})` +
360
+ (isVideo ? ` — ${elementUrl(el.mediaPath)}` : "");
361
+ });
362
+ const posters = await Promise.all(shown.map(async (el) => {
363
+ try {
364
+ const poster = elementPosterPath(el.mediaPath);
365
+ const buf = await cache.getImage(poster, () => client.getAsset(poster));
366
+ return { type: "image", data: buf.toString("base64"), mimeType: "image/jpeg" };
367
+ }
368
+ catch {
369
+ return null; // poster failures degrade to text-only listings
370
+ }
371
+ }));
372
+ return {
373
+ content: [
374
+ text(`${title}: ${elements.length} design element(s):\n\n${lines.join("\n")}`),
375
+ ...posters.filter((b) => b !== null),
376
+ ],
377
+ };
378
+ }
379
+ catch (err) {
380
+ return errorResponse(err);
381
+ }
382
+ }
383
+ async function list_categories() {
384
+ try {
385
+ let cats = cache.getMeta("categories", CATEGORY_TTL_MS);
386
+ if (!cats) {
387
+ cats = parseCategories(await client.getHtml("/websites/"));
388
+ if (cats.colors.length === 0 && cats.filters.length === 0) {
389
+ return {
390
+ content: [
391
+ text("Awwwards layout may have changed: parsed 0 categories. " +
392
+ "The awwwards-mcp parser likely needs an update."),
393
+ ],
394
+ isError: true,
395
+ };
396
+ }
397
+ cache.setMeta("categories", cats);
398
+ }
399
+ return {
400
+ content: [
401
+ text(JSON.stringify({
402
+ colorCount: cats.colors.length,
403
+ colors: cats.colors,
404
+ filterCount: cats.filters.length,
405
+ filters: cats.filters,
406
+ usage: "Pass one of: color (hex), award (sotd|developer|honorable), technology or a tag slug to search_sites. Combine at most one URL filter with client-side tags.",
407
+ }, null, 1)),
408
+ ],
409
+ };
410
+ }
411
+ catch (err) {
412
+ return errorResponse(err);
413
+ }
414
+ }
415
+ async function capture_live_site(args) {
416
+ try {
417
+ // Lazy default: playwright is only touched when the tool actually runs.
418
+ // The default is wrapped because captureLiveSite's third positional is
419
+ // the injectable playwright loader — opts must land in fourth place.
420
+ const capture = deps.captureFn ??
421
+ ((url, imagesDir, opts) => import("./capture.js").then((m) => m.captureLiveSite(url, imagesDir, undefined, opts)));
422
+ const result = await capture(args.url, cache.imagesDir, { waitStrategy: args.waitStrategy });
423
+ if ("error" in result)
424
+ return { content: [text(result.error)], isError: true };
425
+ return {
426
+ content: [
427
+ text(`Full-page capture of ${args.url} saved to ${result.file}`),
428
+ { type: "image", data: result.base64, mimeType: "image/png" },
429
+ ],
430
+ };
431
+ }
432
+ catch (err) {
433
+ return errorResponse(err);
434
+ }
435
+ }
436
+ async function analyze_page_structure(args) {
437
+ try {
438
+ // Lazy default: playwright is only touched when the tool actually runs.
439
+ // maxBands from the tool schema is forwarded so the analyzer honors the
440
+ // caller's cap (falling back to the analyzer's own default of 40).
441
+ const analyze = deps.analyzeFn ??
442
+ ((url, maxBands, opts) => import("./structure.js").then((m) => m.analyzePageStructure(url, undefined, maxBands, opts)));
443
+ const structure = await analyze(args.url, args.maxBands, {
444
+ waitStrategy: args.waitStrategy,
445
+ });
446
+ if ("error" in structure)
447
+ return { content: [text(structure.error)], isError: true };
448
+ return { content: [text(JSON.stringify(structure, null, 1))] };
449
+ }
450
+ catch (err) {
451
+ return errorResponse(err);
452
+ }
453
+ }
454
+ async function record_site_motion(args) {
455
+ try {
456
+ // Lazy default: playwright/ffmpeg are only touched when the tool runs.
457
+ // The default forwards motionOpts wholesale, so waitStrategy flows into
458
+ // recordSiteMotion's MotionOpts (which already accepts it).
459
+ const motion = deps.motionFn ??
460
+ ((url, motionOpts) => import("./motion.js").then((m) => m.recordSiteMotion(url, motionOpts)));
461
+ const result = await motion(args.url, {
462
+ cacheImagesDir: cache.imagesDir,
463
+ frames: args.frames,
464
+ waitStrategy: args.waitStrategy,
465
+ });
466
+ if ("error" in result)
467
+ return { content: [text(result.error)], isError: true };
468
+ return {
469
+ content: [
470
+ text(`Motion recording saved to ${result.file}`),
471
+ { type: "image", data: result.base64, mimeType: "image/jpeg" },
472
+ ],
473
+ };
474
+ }
475
+ catch (err) {
476
+ return errorResponse(err);
477
+ }
478
+ }
479
+ return {
480
+ search_sites,
481
+ get_site_details,
482
+ get_site_elements,
483
+ list_categories,
484
+ capture_live_site,
485
+ analyze_page_structure,
486
+ record_site_motion,
487
+ };
488
+ }