@hoststack.dev/mcp 0.19.0 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -3,7 +3,7 @@ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
3
3
  import { HostStack } from "@hoststack.dev/sdk";
4
4
 
5
5
  // src/version.ts
6
- var MCP_VERSION = true ? "0.19.0" : "0.0.0-dev";
6
+ var MCP_VERSION = true ? "0.20.0" : "0.0.0-dev";
7
7
  var USER_AGENT = `hoststack-mcp/${MCP_VERSION}`;
8
8
 
9
9
  // src/api-client.ts
@@ -1724,9 +1724,11 @@ defineTool({
1724
1724
  "",
1725
1725
  "When to use: enumerate live domains, audit DNS verification status, or find which service a hostname resolves to before troubleshooting routing.",
1726
1726
  "",
1727
- "Returns: { items: Domain[] } \u2014 id, publicId, hostname, serviceId, verified, dnsTargets, sslStatus, createdAt.",
1727
+ "Returns: { items: Domain[] } \u2014 id, publicId, hostname, serviceId, verified, dnsTargets, sslStatus, isPrimary, createdAt.",
1728
1728
  "",
1729
- 'Example: list_domains() \u2192 { items: [{ hostname: "api.example.com", verified: true, sslStatus: "active", \u2026 }] }'
1729
+ "`isPrimary` is the one the platform treats as the service's address: what `${service.url}` resolves to, what its uptime check probes, and what the monitoring page shows. When NO domain on a service is primary \u2014 the default, since nothing sets it automatically \u2014 that choice falls to the OLDEST domain on the service, which on a renamed host is usually the retired alias rather than the canonical name. Nominate one with update_domain.",
1730
+ "",
1731
+ 'Example: list_domains() \u2192 { items: [{ hostname: "api.example.com", verified: true, sslStatus: "active", isPrimary: false, \u2026 }] }'
1730
1732
  ].join("\n"),
1731
1733
  input: {},
1732
1734
  handler: async (_args, ctx) => {
@@ -1802,6 +1804,54 @@ defineTool({
1802
1804
  });
1803
1805
  }
1804
1806
  });
1807
+ defineTool({
1808
+ name: "update_domain",
1809
+ category: "domains",
1810
+ description: [
1811
+ "Change a domain's settings \u2014 most usefully, nominate it as the service's PRIMARY hostname.",
1812
+ "",
1813
+ "When to use: a service answers on more than one hostname and you need to say which one is canonical \u2014 before setting an uptime check on it, after renaming a storefront, or when `${service.url}` is resolving to the wrong name. Also for toggling https on a domain or turning one into a redirect.",
1814
+ "",
1815
+ "WHY PRIMARY MATTERS. A service can answer on several hostnames that behave differently \u2014 the canonical one serves the page, a retired alias 301s to it \u2014 and three parts of the platform have to pick ONE of them: the address a service advertises to itself in `${service.url}`, the host its uptime check probes, and the hostname shown on the monitoring page. All three pick the primary, and fall back to the OLDEST domain when no primary is nominated. So a service with two hostnames and no primary has its uptime check pointed at whichever domain was added first, which may well be the alias that redirects \u2014 and `set_uptime_check` takes a path, never a host, so nominating the primary here is the only way to say which name is checked.",
1816
+ "",
1817
+ "One primary per service: promoting a domain demotes its sibling in the same write, so there is never a pair to choose between.",
1818
+ "",
1819
+ "Inputs:",
1820
+ " - domain_id: publicId of the domain (from list_domains).",
1821
+ " - isPrimary: make this the service's primary hostname.",
1822
+ " - sslEnabled: serve it over https. Applies on the next deploy, which this triggers.",
1823
+ " - redirectTo: send every request to an absolute http(s) URL instead of the service. Pass null to clear.",
1824
+ "",
1825
+ "Returns: { domain } \u2014 the updated row, including isPrimary.",
1826
+ "",
1827
+ 'Example: update_domain({ domain_id: "dom_xyz", isPrimary: true }) \u2192 { domain: { hostname: "shop.example.com", isPrimary: true } }'
1828
+ ].join("\n"),
1829
+ input: {
1830
+ domain_id: z9.string().describe("Domain publicId."),
1831
+ isPrimary: z9.boolean().optional(),
1832
+ sslEnabled: z9.boolean().optional(),
1833
+ redirectTo: z9.string().max(2e3).nullable().optional().describe("Absolute http(s) URL, or null to clear.")
1834
+ },
1835
+ handler: async (args, ctx) => {
1836
+ const { domain_id: domainId, ...patch } = args;
1837
+ const body = Object.fromEntries(
1838
+ Object.entries(patch).filter(([, v]) => v !== void 0)
1839
+ );
1840
+ if (Object.keys(body).length === 0) {
1841
+ return respond({
1842
+ summary: "Nothing to change \u2014 pass isPrimary, sslEnabled or redirectTo.",
1843
+ data: { ok: false }
1844
+ });
1845
+ }
1846
+ const teamId = await ctx.resolveTeamId();
1847
+ const response = await ctx.hoststack.domains.update(teamId, domainId, body);
1848
+ const domain = shapeDomain(response.domain);
1849
+ return respond({
1850
+ summary: body.isPrimary === true ? `${String(domain.hostname ?? domainId)} is now the primary hostname for its service \u2014 it is what \`\${service.url}\` resolves to and what the uptime check probes.` : `Updated ${String(domain.hostname ?? domainId)}.`,
1851
+ data: { domain }
1852
+ });
1853
+ }
1854
+ });
1805
1855
  defineTool({
1806
1856
  name: "remove_domain",
1807
1857
  category: "domains",
@@ -2157,6 +2207,15 @@ defineTool({
2157
2207
 
2158
2208
  // src/tools/analytics.ts
2159
2209
  import { z as z12 } from "zod";
2210
+
2211
+ // src/lib/format.ts
2212
+ var MCP_LOCALE = "en-IE";
2213
+ var COUNT_FORMAT = new Intl.NumberFormat(MCP_LOCALE);
2214
+ function formatCount(value) {
2215
+ return COUNT_FORMAT.format(value);
2216
+ }
2217
+
2218
+ // src/tools/analytics.ts
2160
2219
  async function resolveSiteIds(domains, teamId, api) {
2161
2220
  if (!domains || domains.length === 0) return void 0;
2162
2221
  const { sites } = await api.get(`/api/analytics/${teamId}/sites`);
@@ -2175,6 +2234,10 @@ async function resolveSiteIds(domains, teamId, api) {
2175
2234
  }
2176
2235
  return ids.join(",");
2177
2236
  }
2237
+ var ALLOWED_ORIGINS_INPUT = z12.array(z12.string().trim().min(1).max(253)).max(20).describe(
2238
+ 'Extra origins allowed to post events, beyond the site domain and its subdomains (both are always allowed). Bare hostnames, e.g. ["staging.example.com", "localhost:5173"]. Replaces the whole list; pass [] to clear it.'
2239
+ );
2240
+ var RETENTION_DAYS_INPUT = z12.number().int().positive().describe("How many days of raw events to keep. Rollups outlive this.");
2178
2241
  defineTool({
2179
2242
  name: "list_analytics_sites",
2180
2243
  category: "analytics",
@@ -2210,7 +2273,9 @@ defineTool({
2210
2273
  "",
2211
2274
  "The ingest endpoint answers HTTP 204 to a real site key and to a typo alike \u2014 deliberately, so it cannot be used to enumerate keys \u2014 which means five completely different situations produce the same empty chart: a stale or mistyped key in the deployed snippet, an origin the site does not allow, a blown hourly quota, events blocked in the browser before they leave (CORS, ad blocker, CSP), and simply no visitors yet. This tool tells them apart.",
2212
2275
  "",
2213
- "Returns: health ('receiving' | 'quiet' | 'refusing' | 'never'), a headline and detail sentence, lastEventAt, lastRefusalAt/Reason/Origin, refusedRecently (a 7-day tally by reason), the site key the snippet must carry, allowed origins, and quota usage this hour.",
2276
+ "Returns: health ('receiving' | 'quiet' | 'refusing' | 'never'), a headline and detail sentence, lastEventAt, lastRefusalAt/Reason/Origin, refusedRecently (a 7-day tally by reason), refusedOrigins (which origins were turned away for bad_origin and how often, over the same window), the site key the snippet must carry, allowed origins, and quota usage this hour.",
2277
+ "",
2278
+ "Use refusedOrigins, not lastRefusalOrigin, to decide what to allow: lastRefusalOrigin is the newest refusal of ANY reason, and bots outnumber everything, so one crawler hit overwrites the origin behind a real outage. Feed what you find to update_analytics_site \u2014 and if the origin is NOT the user's, the site key is public by design and somebody has pasted it into their own page, so the answer is to rotate the key rather than allow them.",
2214
2279
  "",
2215
2280
  "health='never' means nothing has EVER reached ingest for this key, so the request is not leaving the browser \u2014 check the snippet is in the deployed HTML and tell the user to add `data-debug` to the script tag, which makes the tracker log its resolved endpoint and every accepted 204 to the console.",
2216
2281
  "",
@@ -2304,7 +2369,7 @@ defineTool({
2304
2369
  "the filters you passed were NOT applied \u2014 they only work on shorter ranges"
2305
2370
  );
2306
2371
  }
2307
- const summary = `${current.pageviews.toLocaleString()} pageviews from ${current.visitors.toLocaleString()} visitors over ${response.range}${caveats.length > 0 ? `. Note: ${caveats.join("; ")}.` : "."}`;
2372
+ const summary = `${formatCount(current.pageviews)} pageviews from ${formatCount(current.visitors)} visitors over ${response.range}${caveats.length > 0 ? `. Note: ${caveats.join("; ")}.` : "."}`;
2308
2373
  return respond({ summary, data: response });
2309
2374
  }
2310
2375
  });
@@ -2318,7 +2383,9 @@ defineTool({
2318
2383
  "",
2319
2384
  "If the domain is already attached to a service on this team, the new site links itself to that service and appears on its Analytics tab. Nothing is counted until the snippet is actually on the page \u2014 creating the site alone produces an empty dashboard, which is expected, not a fault.",
2320
2385
  "",
2321
- 'Inputs: domain (required, bare hostname \u2014 "example.com", not a URL), name (optional display name, defaults to the domain).',
2386
+ 'Inputs: domain (required, bare hostname \u2014 "example.com", not a URL), name (optional display name, defaults to the domain), allowed_origins (optional extra origins beyond the domain and its subdomains), retention_days (optional raw-event retention).',
2387
+ "",
2388
+ "To change any of these later \u2014 or to attach the site to a service \u2014 use update_analytics_site. Creating is not the only chance to set them.",
2322
2389
  "",
2323
2390
  "Returns: the created site (id, domain, name, ingestKey, retentionDays) and the ready-made `snippet` to paste into the page \u2014 the key is in that snippet, so hand it over rather than describing it.",
2324
2391
  "",
@@ -2326,13 +2393,17 @@ defineTool({
2326
2393
  ].join("\n"),
2327
2394
  input: {
2328
2395
  domain: z12.string().min(1).max(253).describe("Bare hostname, e.g. example.com"),
2329
- name: z12.string().min(1).max(100).optional().describe("Display name. Defaults to the domain.")
2396
+ name: z12.string().min(1).max(100).optional().describe("Display name. Defaults to the domain."),
2397
+ allowed_origins: ALLOWED_ORIGINS_INPUT.optional(),
2398
+ retention_days: RETENTION_DAYS_INPUT.optional()
2330
2399
  },
2331
2400
  handler: async (args, ctx) => {
2332
2401
  const teamId = await ctx.resolveTeamId();
2333
2402
  const site = await ctx.api.post(`/api/analytics/${teamId}/sites`, {
2334
2403
  domain: args.domain,
2335
- ...args.name ? { name: args.name } : {}
2404
+ ...args.name ? { name: args.name } : {},
2405
+ ...args.allowed_origins ? { allowedOrigins: args.allowed_origins } : {},
2406
+ ...args.retention_days ? { retentionDays: args.retention_days } : {}
2336
2407
  });
2337
2408
  const snippet = `<script defer src="https://hoststack.dev/t.js" data-site-key="${site.ingestKey}"></script>`;
2338
2409
  return respond({
@@ -2341,6 +2412,53 @@ defineTool({
2341
2412
  });
2342
2413
  }
2343
2414
  });
2415
+ defineTool({
2416
+ name: "update_analytics_site",
2417
+ category: "analytics",
2418
+ description: [
2419
+ "Change an existing analytics site: its allowed origins, display name, retention, or which service it belongs to.",
2420
+ "",
2421
+ "When to use: events are being refused with `bad_origin` because they come from an origin the site does not list (check_analytics_site names it), a site needs a clearer name, retention should change, or the site should appear on a service's Analytics tab.",
2422
+ "",
2423
+ "Inputs: domain (required \u2014 the site to change, by its bare hostname), then any of allowed_origins, name, retention_days, service_id. Everything is optional except domain; only the fields you pass are touched.",
2424
+ "",
2425
+ "allowed_origins REPLACES the list rather than appending, so read the current one from list_analytics_sites first and send it back with your addition. The site domain and its subdomains are always allowed and never need listing.",
2426
+ "",
2427
+ "Returns: { site } \u2014 the site as it now stands.",
2428
+ "",
2429
+ "Example: update_analytics_site({ domain: 'example.com', allowed_origins: ['staging.example.com'] }) \u2192 { site: { domain: 'example.com', allowedOrigins: ['staging.example.com'], \u2026 } }."
2430
+ ].join("\n"),
2431
+ input: {
2432
+ domain: z12.string().min(1).max(253).describe("Bare hostname of a site this team tracks."),
2433
+ allowed_origins: ALLOWED_ORIGINS_INPUT.optional(),
2434
+ name: z12.string().trim().min(1).max(100).optional().describe("New display name."),
2435
+ retention_days: RETENTION_DAYS_INPUT.optional(),
2436
+ service_id: z12.number().int().positive().nullable().optional().describe("Numeric service id to attach this site to, or null to detach it.")
2437
+ },
2438
+ handler: async (args, ctx) => {
2439
+ const teamId = await ctx.resolveTeamId();
2440
+ const siteId = await resolveSiteIds([args.domain], teamId, ctx.api);
2441
+ const patch = {};
2442
+ if (args.allowed_origins !== void 0) patch["allowedOrigins"] = args.allowed_origins;
2443
+ if (args.name !== void 0) patch["name"] = args.name;
2444
+ if (args.retention_days !== void 0) patch["retentionDays"] = args.retention_days;
2445
+ if (args.service_id !== void 0) patch["serviceId"] = args.service_id;
2446
+ if (Object.keys(patch).length === 0) {
2447
+ throw new Error(
2448
+ "Nothing to update. Pass at least one of allowed_origins, name, retention_days or service_id."
2449
+ );
2450
+ }
2451
+ const site = await ctx.api.patch(
2452
+ `/api/analytics/${teamId}/sites/${siteId}`,
2453
+ patch
2454
+ );
2455
+ const changed = Object.keys(patch).join(", ");
2456
+ return respond({
2457
+ summary: `Updated ${site.domain} (${changed}).`,
2458
+ data: { site: shape(site) }
2459
+ });
2460
+ }
2461
+ });
2344
2462
  defineTool({
2345
2463
  name: "verify_site_domain",
2346
2464
  category: "analytics",
@@ -2386,6 +2504,42 @@ defineTool({
2386
2504
  });
2387
2505
  }
2388
2506
  });
2507
+ defineTool({
2508
+ name: "get_site_uptime_check",
2509
+ category: "analytics",
2510
+ description: [
2511
+ "Read a site's uptime check without touching it.",
2512
+ "",
2513
+ 'When to use: answer "has it probed yet?", "is it up?", "how many failures in a row?" \u2014 the site counterpart of get_uptime_check.',
2514
+ "",
2515
+ 'READ THIS BEFORE REACHING FOR set_site_uptime_check TO INSPECT ONE. The setter RESETS the accumulated state it returns (status back to "unknown", consecutiveFailures to 0, lastCheckedAt to null), because a check whose shape just changed has not observed the new shape failing. So using it to look at a check is what stops the check ever showing a probe: every read restarts the measurement, and a working check reads as one that never runs. This tool exists because that trap cost a real investigation an afternoon.',
2516
+ "",
2517
+ "Returns: { check } \u2014 enabled, path, method, expectedStatus, timeoutMs, intervalSeconds, failureThreshold, plus live state: status ('up' | 'down' | 'unknown'), consecutiveFailures, lastCheckedAt, lastStatusCode, lastLatencyMs, lastError, lastChangedAt. `null` when the site has no check.",
2518
+ "",
2519
+ "Example: get_site_uptime_check({ siteId: 12 }) \u2192 { check: { status: 'up', lastStatusCode: 200, lastCheckedAt: '2026-09-09T05:27:34Z' } }."
2520
+ ].join("\n"),
2521
+ input: {
2522
+ siteId: z12.number().int().positive()
2523
+ },
2524
+ handler: async (args, ctx) => {
2525
+ const teamId = await ctx.resolveTeamId();
2526
+ const response = await ctx.api.get(
2527
+ `/api/analytics/${teamId}/sites/${args.siteId}/uptime-check`
2528
+ );
2529
+ if (!response.check) {
2530
+ return respond({
2531
+ summary: "No uptime check on this site \u2014 nothing is watching it.",
2532
+ data: { check: null }
2533
+ });
2534
+ }
2535
+ const check = shape(response.check);
2536
+ const lastCheckedAt = check.lastCheckedAt;
2537
+ return respond({
2538
+ summary: `Uptime check is ${String(check.status ?? "unknown")}${lastCheckedAt ? `, last probed ${String(lastCheckedAt)}` : ", not probed yet"}.`,
2539
+ data: { check }
2540
+ });
2541
+ }
2542
+ });
2389
2543
  defineTool({
2390
2544
  name: "set_site_uptime_check",
2391
2545
  category: "analytics",
@@ -2394,7 +2548,9 @@ defineTool({
2394
2548
  "",
2395
2549
  "When to use: the user has a site running somewhere else (their own box, another provider) and wants to know when it goes down. This is the one observability capability an off-platform site cannot provide for itself: analytics is a script tag and error reporting is an HTTP POST, but an outside-in probe has to come from outside.",
2396
2550
  "",
2397
- "For a site that IS a HostStack service, use set_uptime_check with its serviceId instead \u2014 that one follows the service if its domain changes.",
2551
+ "Works on a site HostStack DOES host too, and that is not a worse option: this check names ONE hostname \u2014 the site's own proven domain \u2014 where set_uptime_check follows whichever of the service's domains is primary at probe time. A service answering on a canonical host plus a 301 alias has no single correct expectedStatus, so a site check per hostname is how each one gets asserted, and nothing in the prober treats a service-backed site differently. Use set_uptime_check when the check should follow the service's domain; use this one when a named host is the point.",
2552
+ "",
2553
+ 'To READ a check, call get_site_uptime_check. Changing the shape of a check RESETS its accumulated state (status to "unknown", consecutiveFailures to 0, lastCheckedAt to null) \u2014 the new shape has not been measured yet, so carrying failures forward would alert about a condition nobody observed. A write that resolves to the shape ALREADY STORED changes nothing and leaves that state alone. The inputs are defaults rather than a partial edit, though: omitting intervalSeconds on a check currently set to 30 resolves to 60, which IS a change.',
2398
2554
  "",
2399
2555
  "REQUIRES a proven domain. Call verify_site_domain first if `domainProven` is false; an unverified target is refused with 400. This is not paperwork: an uptime check makes the control plane fetch the hostname every interval, forever, from our IP, so it must be a name the team has shown it owns.",
2400
2556
  "",
@@ -2423,9 +2579,11 @@ defineTool({
2423
2579
  `/api/analytics/${teamId}/sites/${siteId}/uptime-check`,
2424
2580
  body
2425
2581
  );
2582
+ const check = shape(response.check);
2583
+ const probed = check["lastCheckedAt"] !== null && check["lastCheckedAt"] !== void 0;
2426
2584
  return respond({
2427
- summary: "Uptime check saved. It will start reporting within a minute or two.",
2428
- data: { check: shape(response.check) }
2585
+ summary: probed ? `Uptime check saved \u2014 the shape was already stored, so nothing was reset: still ${String(check["status"])}, last probed ${String(check["lastCheckedAt"])}.` : 'Uptime check saved, and not probed yet \u2014 "unknown" with no lastCheckedAt is the state of a check that has not run once, not a broken one. It reports within a minute or two: read it with get_site_uptime_check, never by saving it again.',
2586
+ data: { check }
2429
2587
  });
2430
2588
  }
2431
2589
  });
@@ -2853,11 +3011,17 @@ var NOTIFICATION_EVENTS = [
2853
3011
  "service.suspended",
2854
3012
  "service.resumed",
2855
3013
  "service.restart_failed",
2856
- "service.auto_suspended",
3014
+ "service.no_running_container",
3015
+ "service.health_check_failed",
2857
3016
  "service.acme_cert_failed",
2858
3017
  "service.resource_alert",
3018
+ "service.pressure_sustained",
3019
+ "service.pressure_recovered",
2859
3020
  "service.uptime_down",
2860
3021
  "service.uptime_recovered",
3022
+ "watchdog.reported_down",
3023
+ "watchdog.reported_recovered",
3024
+ "watchdog.silent",
2861
3025
  "error.issue_new",
2862
3026
  "error.issue_regressed",
2863
3027
  "git.auth_failed",
@@ -2869,7 +3033,13 @@ var NOTIFICATION_EVENTS = [
2869
3033
  "devenv.task.needs_input",
2870
3034
  "devenv.task.finished",
2871
3035
  "database.backup_failed",
3036
+ "database.backup_overdue",
3037
+ "database.failed",
2872
3038
  "database.restore_failed",
3039
+ "volume.backup_failed",
3040
+ "volume.backup_overdue",
3041
+ "domain.registrant_verification_lapsed",
3042
+ "service.auto_restarted",
2873
3043
  "machine.offline",
2874
3044
  "machine.online",
2875
3045
  "billing.invoice",
@@ -4092,10 +4262,12 @@ defineTool({
4092
4262
  "",
4093
4263
  "This resizes a cloud Dev Box, NOT a project deploy environment.",
4094
4264
  "",
4095
- "When to use: a Dev Box OOM-killed (see exitReason/recommendedSize from list_dev_environments), or you just want more headroom. This changes the SIZE TIER \u2014 unlike per-config memory/CPU overrides, which are clamped to the current tier and so cannot grow a box past it.",
4265
+ "When to use: a Dev Box OOM-killed (see exitReason/recommendedSize from list_dev_environments), or you just want more headroom. This changes the SIZE TIER.",
4096
4266
  "",
4097
4267
  "How it applies: the new tier's memory + CPU take effect LIVE on the running container (no recreate, no dropped shell sessions); a larger disk takes effect on the next recreate (suspend\u2192resume). Dev boxes are floored to the OOM-safe minimum size server-side.",
4098
4268
  "",
4269
+ "The tier is not always the container's limit, so check `effectiveMemoryMb` from list_dev_environments rather than assuming: a dedicated (Pro) tier resolves ~1.5 GB below nominal for agent headroom, and a per-service memory override REPLACES the tier figure rather than merely clamping to it. An UPSIZE now retires an override that would hold the box below its new tier; a re-assert of the SAME tier does not, so on a box already at the top tier a resize cannot lift an override \u2014 clear it with a config PATCH of memoryMb to 512 (the legacy floor, which means \"no override\"), or use the Remove-cap button on the box's Settings \u2192 General tab. `memoryPinnedBelowTier: true` is how you tell that case apart from a box that is genuinely maxed out.",
4270
+ "",
4099
4271
  "Inputs:",
4100
4272
  " - service_id: the box to resize \u2014 numeric id or publicId.",
4101
4273
  ' - size: target tier \u2014 one of the service catalog sizes (e.g. "standard", "large", "xlarge").',
@@ -4133,7 +4305,9 @@ defineTool({
4133
4305
  "",
4134
4306
  `When to use: "show my dev environments", before opening/tearing one down, to find a box's id.`,
4135
4307
  "",
4136
- 'Returns: { items: [{ ...service, devUrl, databases, exitReason, recommendedSize }] } where `databases` lists the companion engines wired into the box (e.g. ["postgres","redis"]). `exitReason` is "oom_killed" / "crashed" / null for the box\'s last container exit; when it is "oom_killed", `recommendedSize` is the next tier up to rescale to (use resize_dev_environment).',
4308
+ 'Returns: { items: [{ ...service, devUrl, databases, exitReason, recommendedSize, effectiveMemoryMb, memoryPinnedBelowTier }] } where `databases` lists the companion engines wired into the box (e.g. ["postgres","redis"]). `exitReason` is "oom_killed" / "crashed" / null for the box\'s last container exit; when it is "oom_killed", `recommendedSize` is the next tier up to rescale to (use resize_dev_environment).',
4309
+ "",
4310
+ '`effectiveMemoryMb` is the container\'s REAL ceiling \u2014 what `/sys/fs/cgroup/memory.max` reads inside the box \u2014 and is the number to size any in-box work against. Do NOT derive it from `plan`: a dedicated (Pro) tier sits ~1.5 GB below nominal, and a per-service override replaces the tier figure outright. When `memoryPinnedBelowTier` is true an override, not the tier, is the cap, so `recommendedSize: null` means "nothing bigger to sell you" rather than "nothing you can do" \u2014 see resize_dev_environment for how to lift it.',
4137
4311
  "",
4138
4312
  "Example: list_dev_environments() \u2192 every dev box for the active team."
4139
4313
  ].join("\n"),
@@ -4150,7 +4324,12 @@ defineTool({
4150
4324
  devUrl: env.devUrl ?? null,
4151
4325
  databases: env.databases ?? [],
4152
4326
  exitReason: env.exitReason ?? null,
4153
- recommendedSize: env.recommendedSize ?? null
4327
+ recommendedSize: env.recommendedSize ?? null,
4328
+ // The number an agent should size its work against. Absent it,
4329
+ // the only honest way to learn a box's ceiling was to shell in
4330
+ // and read the cgroup — so every box rediscovered its own.
4331
+ effectiveMemoryMb: env.effectiveMemoryMb ?? null,
4332
+ memoryPinnedBelowTier: env.memoryPinnedBelowTier ?? false
4154
4333
  }))
4155
4334
  }
4156
4335
  });
@@ -4336,7 +4515,7 @@ defineTool({
4336
4515
  "Inputs:",
4337
4516
  ' - service_id: publicId of the service (e.g. "svc_abc123").',
4338
4517
  "",
4339
- 'Returns: { service: Service, config: ServiceConfig } \u2014 service has type/status/runtime/repoUrl/branch/autoDeploy/region/plan/timestamps; config has memoryMb, cpuShares, diskSizeGb, port, protocol, healthCheckEnabled, healthCheckInterval, healthCheckTimeout, healthCheckGracePeriodSec, restartPolicy, deployStrategy ("rolling" | "recreate"), preDeployCommand, min/maxInstances, scale thresholds.',
4518
+ 'Returns: { service: Service, config: ServiceConfig } \u2014 service has type/status/runtime/repoUrl/branch/autoDeploy/region/plan/timestamps; config has memoryMb, cpuShares, diskSizeGb, port, protocol, healthCheckEnabled, healthCheckInterval, healthCheckTimeout, healthCheckGracePeriodSec, allowSearchIndexing, restartPolicy, deployStrategy ("rolling" | "recreate"), preDeployCommand, min/maxInstances, scale thresholds.',
4340
4519
  "",
4341
4520
  'Example: get_service({ service_id: "svc_abc" }) \u2192 { service: { type: "web", status: "running", \u2026 }, config: { healthCheckGracePeriodSec: 120, \u2026 } }'
4342
4521
  ].join("\n"),
@@ -4395,11 +4574,15 @@ defineTool({
4395
4574
  ' - from: ISO-8601 lower bound OR relative offset like "-15m" / "-2h" / "-7d".',
4396
4575
  " - to: ISO-8601 upper bound (or relative offset). Defaults to now.",
4397
4576
  "",
4398
- "Resolution: \u22647d \u2192 raw samples (~minute granularity), \u226430d \u2192 hourly pre-aggregates, >30d \u2192 daily. Up to ~500 points returned.",
4577
+ "Resolution: \u22647d \u2192 raw samples (one per ~30s), >7d \u2192 hourly pre-aggregates, >30d \u2192 daily. At most 300 points are returned, so a window wider than ~2.5h is thinned to fit.",
4578
+ "",
4579
+ "`cpuPercent` is the PEAK of whatever the point covers, and 100 means one host CORE \u2014 not 100% of the service. A `micro` (500 millicores) is capped by its cgroup at ~50, so ~50 IS that plan pegged. To compare against a `service.pressure_sustained` alert, which reports a share of the plan allowance, multiply by 1000/cpuMillicores (the alert ships that denominator in its metadata).",
4580
+ "",
4581
+ "Memory is the reading at that point, NOT a peak \u2014 so a memory number here can sit below an alert that averaged the hour.",
4399
4582
  "",
4400
4583
  "Returns: { history: Array<{ timestamp, cpuPercent, memoryUsedMb, memoryLimitMb, networkRxBytes, networkTxBytes, diskUsedMb }> }.",
4401
4584
  "",
4402
- 'Example: get_service_metrics_history({ service_id: "svc_abc", from: "-1h" }) \u2192 60-ish points for the last hour.'
4585
+ 'Example: get_service_metrics_history({ service_id: "svc_abc", from: "-1h" }) \u2192 ~120 points for the last hour.'
4403
4586
  ].join("\n"),
4404
4587
  input: {
4405
4588
  service_id: z19.string().describe("Service publicId."),
@@ -4477,6 +4660,7 @@ defineTool({
4477
4660
  " - auto_deploy (optional): boolean \u2014 auto-deploy on git push.",
4478
4661
  ' - health_check_path (optional): HTTP path the platform GETs to verify liveness (e.g. "/health"). Pass null for TCP-only check.',
4479
4662
  " - health_check_enabled (optional): boolean \u2014 toggle health checking on/off.",
4663
+ " - allow_search_indexing (optional): boolean \u2014 let search engines index the free *.hoststack.dev platform URL. Off by default; the platform URL is served with X-Robots-Tag: noindex so a site does not rank on a hostname it does not own. Custom domains are always indexable and unaffected. Applies on the next deploy.",
4480
4664
  " - health_check_interval (optional): integer 5\u2013300 seconds \u2014 how often the check runs.",
4481
4665
  " - health_check_timeout (optional): integer 1\u201360 seconds \u2014 single-attempt timeout.",
4482
4666
  ' - health_check_grace_period_sec (optional): integer 1\u20131800 seconds \u2014 startup tolerance before failures count. RAISE THIS (e.g. 180) when the agent reports "Health check timed out" on a cold-boot app (Bun + Vite SSR typically need 90\u2013180s).',
@@ -4491,7 +4675,7 @@ defineTool({
4491
4675
  " - instance_count (optional): integer 1\u201350 \u2014 pin both min and max instances to this value.",
4492
4676
  " - min_instances, max_instances (optional): integers \u2014 autoscale bounds. Use instead of instance_count when you want a range.",
4493
4677
  " - scale_cpu_threshold, scale_memory_threshold (optional): integer 10\u2013100 \u2014 autoscale trigger percentage.",
4494
- ' - log_filter_rules (optional): list of { pattern, action } rules applied to runtime logs at query time. Pattern matches the message by case-insensitive substring; action is "drop" (filter out) or "downgrade" (flip stderr \u2192 stdout so it stops looking like an error). Pass [] to clear all rules. Capped at 50 rules.',
4678
+ ` - log_filter_rules (optional): list of { pattern, action } rules applied at read time \u2014 to get_service_logs (tail AND count) and to the dashboard's live stream alike. Pattern matches the message by case-insensitive substring, literally (a % or _ is not a wildcard); action is "drop" (filter out) or "downgrade" (flip stderr \u2192 stdout so it stops looking like an error). Stored lines are never altered, so removing a rule re-exposes everything it was hiding. Pass [] to clear all rules. Capped at 50 rules.`,
4495
4679
  "",
4496
4680
  "Returns: { service?: Service, config?: ServiceConfig } \u2014 whichever rows were touched.",
4497
4681
  "",
@@ -4508,6 +4692,9 @@ defineTool({
4508
4692
  auto_deploy: z19.boolean().optional().describe("Auto-deploy on push."),
4509
4693
  health_check_path: z19.string().nullable().optional().describe('HTTP health-check path (e.g. "/health"). Null = TCP-only check.'),
4510
4694
  health_check_enabled: z19.boolean().optional().describe("Toggle health checking on/off."),
4695
+ allow_search_indexing: z19.boolean().optional().describe(
4696
+ "Let search engines index the free *.hoststack.dev platform URL (off by default). Custom domains are always indexable. Applies on the next deploy."
4697
+ ),
4511
4698
  health_check_interval: z19.number().int().min(5).max(300).optional().describe("How often the check runs, in seconds (5\u2013300)."),
4512
4699
  health_check_timeout: z19.number().int().min(1).max(60).optional().describe("Single-attempt timeout in seconds (1\u201360)."),
4513
4700
  health_check_grace_period_sec: z19.number().int().min(1).max(1800).optional().describe(
@@ -4554,6 +4741,8 @@ defineTool({
4554
4741
  const configUpdate = {};
4555
4742
  if (args.health_check_enabled !== void 0)
4556
4743
  configUpdate["healthCheckEnabled"] = args.health_check_enabled;
4744
+ if (args.allow_search_indexing !== void 0)
4745
+ configUpdate["allowSearchIndexing"] = args.allow_search_indexing;
4557
4746
  if (args.health_check_interval !== void 0)
4558
4747
  configUpdate["healthCheckInterval"] = args.health_check_interval;
4559
4748
  if (args.health_check_timeout !== void 0)
@@ -4704,7 +4893,9 @@ defineTool({
4704
4893
  ' - stream (optional): "stdout" | "stderr". Omit to combine.',
4705
4894
  " - level (optional): real log level \u2014 trace/debug/info/warn/error/fatal. For structured JSON logs (pino, bunyan, OpenTelemetry severity) this filters on the parsed inner level field. For plain text logs it falls back to a stream-alias hint (info/debug \u2192 stdout, warn/error/fatal \u2192 stderr).",
4706
4895
  " - search (optional): case-insensitive substring grep, \u2264100 chars.",
4707
- ' - count_only (optional): when true, returns { count } only \u2014 much cheaper for "how many error lines in last 5m" polling.',
4896
+ ' - count_only (optional): when true, returns { count } only \u2014 much cheaper for "how many error lines in last 5m" polling. It counts exactly what the same call would return, so `count` and the entry count agree whenever count <= lines.',
4897
+ "",
4898
+ "If the service has log_filter_rules with action='drop' (see update_service_config), the lines they match are excluded from BOTH the tail and the count \u2014 so a count lower than the raw log volume is the rules working, not lines going missing. `stream` selects on the stream shown in the response, which a 'downgrade' rule may have flipped from stderr to stdout.",
4708
4899
  "",
4709
4900
  "Returns: { logs: LogEntry[] | string } when count_only is false. Each entry has { timestamp, level?, stream, message }. `level` is the parsed inner level when the message is a structured JSON envelope (pino numeric or string), and undefined for plain-text logs. `stream` is always one of stdout/stderr. Or { count: number } when count_only is true.",
4710
4901
  "",
@@ -4842,9 +5033,11 @@ defineTool({
4842
5033
  "",
4843
5034
  "Inputs: serviceId (required).",
4844
5035
  "",
4845
- 'Returns: { check } or { check: null } when none is configured. The check carries path, method, expectedStatus, intervalSeconds, failureThreshold, status, consecutiveFailures, lastCheckedAt, lastStatusCode, lastLatencyMs, lastError, lastChangedAt (the "down since" timestamp).',
5036
+ 'Returns: { check } or { check: null } when none is configured. The check carries path, method, expectedStatus, intervalSeconds, failureThreshold, status, consecutiveFailures, lastCheckedAt, lastStatusCode, lastLatencyMs, lastError, lastChangedAt (the "down since" timestamp), and lastTargetUrl.',
5037
+ "",
5038
+ "WHICH HOSTNAME IS PROBED: `lastTargetUrl` is the URL the last probe actually dialled \u2014 read it here rather than inferring it. A check names a PATH and the platform picks the host: the service's primary domain, or the OLDEST domain when no primary is nominated, which on a renamed host is usually the retired alias. So a service answering on two names that behave differently (canonical serves, alias 301s) can have a check that asserts the wrong status code without anything being wrong with the service. If lastTargetUrl is not the name you meant to watch, nominate the right one with update_domain({ domain_id, isPrimary: true }) \u2014 set_uptime_check takes a path and never a host.",
4846
5039
  "",
4847
- "Example: get_uptime_check({ serviceId: 48 }) \u2192 { check: { path: '/healthz', status: 'down', consecutiveFailures: 5, lastError: 'No response within 10000ms', lastChangedAt: '2026-08-21T04:12:00Z' } }."
5040
+ "Example: get_uptime_check({ serviceId: 48 }) \u2192 { check: { path: '/healthz', status: 'down', consecutiveFailures: 5, lastError: 'No response within 10000ms', lastChangedAt: '2026-08-21T04:12:00Z', lastTargetUrl: 'https://grundfast.dk/healthz' } }."
4848
5041
  ].join("\n"),
4849
5042
  input: { serviceId: z20.number().int().positive() },
4850
5043
  handler: async (args, ctx) => {
@@ -4877,9 +5070,9 @@ defineTool({
4877
5070
  "",
4878
5071
  "For a site HostStack does NOT host, there is no serviceId to pass: use set_site_uptime_check with its analytics siteId instead. That path needs the domain proven first (verify_site_domain), because the host it probes comes from the site row rather than from a domain the platform already vouches for.",
4879
5072
  "",
4880
- "IMPORTANT: changing the shape of a check RESETS its accumulated state (status, consecutive failures, open alert). A check whose path or expected status just changed has not observed the new check failing, so carrying failures forward would alert about a condition that was never measured. Pass the full shape you want, not a partial edit of an unknown current state \u2014 read it with get_uptime_check first if that matters.",
5073
+ "IMPORTANT: changing the shape of a check RESETS its accumulated state (status, consecutive failures, open alert). A check whose path or expected status just changed has not observed the new check failing, so carrying failures forward would alert about a condition that was never measured. A write that resolves to the shape already stored changes nothing and leaves that state alone \u2014 but the inputs are defaults, not a partial edit, so omitting intervalSeconds on a check set to 30 resolves to 60 and IS a change. Read a check with get_uptime_check; never re-save one in order to look at it.",
4881
5074
  "",
4882
- "`path` is a path, not a URL: the host comes from the service's primary domain at probe time, so moving the service to a new domain moves the check with it.",
5075
+ "`path` is a path, not a URL: the host comes from the service's primary domain at probe time, so moving the service to a new domain moves the check with it. On a service with SEVERAL domains that is a choice, and this tool does not make it \u2014 nominate the hostname to watch with update_domain({ domain_id, isPrimary: true }), or the platform falls back to the OLDEST domain, which after a rename is usually the alias that redirects. Check which one is live with get_uptime_check's lastTargetUrl before trusting an expectedStatus. When ONE named hostname has to be asserted whatever the platform picks \u2014 a canonical host and a 301 alias cannot share an expectedStatus \u2014 give that hostname its own analytics site and check it with set_site_uptime_check.",
4883
5076
  "",
4884
5077
  "Method is GET or HEAD only. A probe fires unattended every interval forever, so it has to be safe to repeat \u2014 a check that could POST would be a scheduled writer against the team's own API.",
4885
5078
  "",
@@ -4908,9 +5101,11 @@ defineTool({
4908
5101
  `/api/services/${teamId}/${serviceId}/uptime-check`,
4909
5102
  body
4910
5103
  );
5104
+ const check = shape(response.check);
5105
+ const probed = check["lastCheckedAt"] !== null && check["lastCheckedAt"] !== void 0;
4911
5106
  return respond({
4912
- summary: `Uptime check saved. It will start reporting within a minute or two.`,
4913
- data: { check: shape(response.check) }
5107
+ summary: probed ? `Uptime check saved \u2014 the shape was already stored, so nothing was reset: still ${String(check["status"])}, last probed ${String(check["lastCheckedAt"])}.` : 'Uptime check saved, and not probed yet \u2014 "unknown" with no lastCheckedAt is the state of a check that has not run once, not a broken one. It reports within a minute or two: read it with get_uptime_check, never by saving it again.',
5108
+ data: { check }
4914
5109
  });
4915
5110
  }
4916
5111
  });