mosaic-headless 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -32,6 +32,7 @@ import argparse
32
32
  import json
33
33
  import os
34
34
  import sys
35
+ import urllib.request
35
36
  import urllib.parse
36
37
  import uuid
37
38
 
@@ -148,6 +149,45 @@ def build_page_document(client, cfg, master_id, template_id, tree, surface):
148
149
  return len(records)
149
150
 
150
151
 
152
+ def cache_check(url, _fresh_bytes=None):
153
+ """Fetch the URL twice - once as a visitor, once cache-busted - and compare.
154
+
155
+ Every check in this repo cache-busts, for good reason: a verifier that reads a
156
+ stale copy reports the previous build. But that means nothing here had ever
157
+ looked at the page an actual VISITOR receives. A full-page cache in front of
158
+ WordPress - Varnish on Cloudways, any CDN - keeps serving the old document after
159
+ a perfectly successful build, and the whole toolchain reports green while the
160
+ site shows yesterday's page. Measured here: 141,134 bytes to a visitor while the
161
+ build had just produced 148,457.
162
+
163
+ Both fetches happen here, at the same moment, because comparing against a size
164
+ measured earlier in the build folds in every byte that changed in between. Two
165
+ fetches of the same live page still differ by a few hundred bytes - nonces and
166
+ ids are regenerated per request - so the tolerance is proportional: natural
167
+ variance measured at 0.25%, a stale document at 5%.
168
+
169
+ This cannot purge the cache; that needs credentials this tool has no business
170
+ holding. Silence is the one thing it must not do.
171
+ """
172
+ def get(u):
173
+ req = urllib.request.Request(u, headers={"User-Agent": "Mozilla/5.0"})
174
+ with urllib.request.urlopen(req, timeout=60) as r:
175
+ return r.read(), (r.headers.get("Age") or r.headers.get("X-Cache") or "")
176
+
177
+ try:
178
+ sep = "&" if "?" in url else "?"
179
+ fresh, _ = get("%s%s_v=%d" % (url, sep, uuid.uuid4().int % 10 ** 9))
180
+ plain, age = get(url)
181
+ except Exception as exc: # noqa: BLE001
182
+ return "could not compare the cached page: %s" % exc
183
+
184
+ if abs(len(plain) - len(fresh)) <= max(512, len(fresh) // 100):
185
+ return ""
186
+ return ("STALE CACHE: a visitor gets %d bytes, a fresh fetch gives %d%s"
187
+ " - purge the page cache or the site keeps serving the old document"
188
+ % (len(plain), len(fresh), (" (Age %s)" % age) if age else ""))
189
+
190
+
151
191
  def main():
152
192
  ap = argparse.ArgumentParser()
153
193
  ap.add_argument("--config", required=True)
@@ -168,6 +208,9 @@ def main():
168
208
  ok = len(body) >= MIN_HEALTHY_BYTES
169
209
  print(" %-16s %-7s %6d bytes %3d nodes %s" % (
170
210
  page["slug"], "OK" if ok else "BROKEN", len(body), n, url))
211
+ stale = cache_check(url, len(body))
212
+ if stale:
213
+ print(" %-16s %s" % ("", stale))
171
214
 
172
215
 
173
216
  if __name__ == "__main__":
@@ -52,7 +52,8 @@ import time
52
52
 
53
53
  # When to read. Dense through the sequence, then two well past its end - the last
54
54
  # two are what turn "it looked right" into "it finished".
55
- SAMPLES_MS = [120, 350, 700, 1100, 1500, 1750, 2050, 2400, 3200, 4500]
55
+ SAMPLES_MS = [120, 400, 800, 1300, 1800, 2300, 2700, 3000, 3300, 3700, 4200,
56
+ 5200, 6500]
56
57
 
57
58
  # What to read at each sample. Anything whose id starts with the veil prefix is
58
59
  # treated as part of the intro; everything else is content that must end up visible.
@@ -205,7 +206,7 @@ def main():
205
206
  ap.add_argument("--page")
206
207
  ap.add_argument("--veil-prefix", default="mk-boot",
207
208
  help="ids under this prefix are the intro, not the content")
208
- ap.add_argument("--deadline-ms", type=int, default=3200,
209
+ ap.add_argument("--deadline-ms", type=int, default=4200,
209
210
  help="by this point the intro must be over")
210
211
  ap.add_argument("--csv")
211
212
  ap.add_argument("--frames")