n-seo 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +132 -46
  2. package/bin/n-seo.mjs +14 -4
  3. package/docs/FAQ.md +14 -3
  4. package/docs/MCP.md +25 -9
  5. package/docs/PRD.md +6 -4
  6. package/docs/RELEASING.md +21 -0
  7. package/docs/SCHEDULING.md +42 -5
  8. package/docs/SETUP-GOOGLE.md +1 -1
  9. package/ingest/__pycache__/analyze_metadata.cpython-312.pyc +0 -0
  10. package/ingest/__pycache__/google_auth.cpython-312.pyc +0 -0
  11. package/ingest/__pycache__/http_util.cpython-312.pyc +0 -0
  12. package/ingest/__pycache__/seo_config.cpython-312.pyc +0 -0
  13. package/ingest/analyze_ga4.py +2 -2
  14. package/ingest/analyze_gsc.py +2 -2
  15. package/ingest/analyze_metadata.py +2 -2
  16. package/ingest/analyze_trends.py +2 -2
  17. package/ingest/google_auth.py +53 -20
  18. package/ingest/pull_ga4.py +1 -1
  19. package/ingest/pull_gsc.py +1 -1
  20. package/ingest/pull_index_status.py +1 -1
  21. package/ingest/pull_timeseries.py +3 -3
  22. package/ingest/seo_config.py +19 -3
  23. package/ops/__pycache__/daily.cpython-312.pyc +0 -0
  24. package/ops/__pycache__/daily_diff.cpython-312.pyc +0 -0
  25. package/ops/__pycache__/demo_data.cpython-312.pyc +0 -0
  26. package/ops/__pycache__/export_static.cpython-312.pyc +0 -0
  27. package/ops/__pycache__/llm.cpython-312.pyc +0 -0
  28. package/ops/__pycache__/publish.cpython-312.pyc +0 -0
  29. package/ops/daily.py +2 -2
  30. package/ops/daily_diff.py +9 -8
  31. package/ops/demo_data.py +17 -17
  32. package/ops/doctor.py +21 -7
  33. package/ops/export_static.py +4 -4
  34. package/ops/hn_digest.py +1 -1
  35. package/ops/indexnow.py +3 -3
  36. package/ops/llm.py +23 -1
  37. package/ops/opportunity_scan.py +4 -4
  38. package/ops/py.mjs +40 -0
  39. package/ops/reddit_digest.py +1 -1
  40. package/ops/templates/n-seo-daily-task.xml +65 -0
  41. package/package.json +16 -9
  42. package/probes/__pycache__/site_probe.cpython-312.pyc +0 -0
  43. package/probes/site_probe.py +1 -1
  44. package/src/actions.ts +1 -1
  45. package/src/config.ts +1 -1
  46. package/src/mcp.ts +2 -2
  47. package/src/settings.tsx +3 -3
  48. package/src/views.tsx +9 -9
  49. package/ingest/__pycache__/analyze_ga4.cpython-313.pyc +0 -0
  50. package/ingest/__pycache__/analyze_gsc.cpython-313.pyc +0 -0
  51. package/ingest/__pycache__/analyze_metadata.cpython-313.pyc +0 -0
  52. package/ingest/__pycache__/analyze_trends.cpython-313.pyc +0 -0
  53. package/ingest/__pycache__/google_auth.cpython-313.pyc +0 -0
  54. package/ingest/__pycache__/http_util.cpython-313.pyc +0 -0
  55. package/ingest/__pycache__/pull_ga4.cpython-313.pyc +0 -0
  56. package/ingest/__pycache__/pull_gsc.cpython-313.pyc +0 -0
  57. package/ingest/__pycache__/pull_index_status.cpython-313.pyc +0 -0
  58. package/ingest/__pycache__/pull_timeseries.cpython-313.pyc +0 -0
  59. package/ingest/__pycache__/seo_config.cpython-313.pyc +0 -0
  60. package/ops/__pycache__/daily.cpython-313.pyc +0 -0
  61. package/ops/__pycache__/daily_diff.cpython-313.pyc +0 -0
  62. package/ops/__pycache__/demo_data.cpython-313.pyc +0 -0
  63. package/ops/__pycache__/doctor.cpython-313.pyc +0 -0
  64. package/ops/__pycache__/export_static.cpython-313.pyc +0 -0
  65. package/ops/__pycache__/hn_digest.cpython-313.pyc +0 -0
  66. package/ops/__pycache__/indexnow.cpython-313.pyc +0 -0
  67. package/ops/__pycache__/llm.cpython-313.pyc +0 -0
  68. package/ops/__pycache__/opportunity_scan.cpython-313.pyc +0 -0
  69. package/ops/__pycache__/publish.cpython-313.pyc +0 -0
  70. package/ops/__pycache__/reddit_digest.cpython-313.pyc +0 -0
  71. package/probes/__pycache__/site_probe.cpython-313.pyc +0 -0
@@ -3,10 +3,10 @@
3
3
  service-account-key (default, recommended)
4
4
  A service-account JSON key on disk (config google.serviceAccountKey,
5
5
  or $GOOGLE_APPLICATION_CREDENTIALS). We build the OAuth JWT ourselves
6
- and sign it with the `openssl` CLI, so there is no gcloud and no pip
7
- dependency. Add the service account's email to your Search Console
8
- property (Full user) and your GA4 property (Viewer) — that is all the
9
- access it gets.
6
+ and sign it with node's crypto module, so there is no gcloud, no pip
7
+ dependency and no openssl binary. Add the service account's email to
8
+ your Search Console property (Full user) and your GA4 property
9
+ (Viewer) — that is all the access it gets.
10
10
 
11
11
  gcloud-impersonate
12
12
  `gcloud auth print-access-token --impersonate-service-account=<sa>`.
@@ -33,9 +33,10 @@ inspection, analytics.readonly for GA4.
33
33
  import base64
34
34
  import json
35
35
  import os
36
+ import shutil
36
37
  import subprocess
37
- import tempfile
38
38
  import time
39
+ from pathlib import Path
39
40
 
40
41
  import seo_config
41
42
  from http_util import curl_json
@@ -58,6 +59,50 @@ def _b64url(b: bytes) -> str:
58
59
  return base64.urlsafe_b64encode(b).rstrip(b"=").decode()
59
60
 
60
61
 
62
+ # RS256 without an openssl binary. Node is already required (the dashboard
63
+ # runs on it) and its crypto module produces byte-identical signatures, so
64
+ # signing through node drops an external dependency on every platform — and
65
+ # makes Windows work, where openssl is not installed by default.
66
+ #
67
+ # Key and payload go in over stdin, never on the command line: argv is
68
+ # readable by any other process on the machine.
69
+ _NODE_SIGN_JS = (
70
+ "let s='';"
71
+ "process.stdin.setEncoding('utf8')"
72
+ ".on('data', d => s += d)"
73
+ ".on('end', () => {"
74
+ " const { key, data } = JSON.parse(s);"
75
+ " const sig = require('node:crypto')"
76
+ " .sign('sha256', Buffer.from(data, 'base64'), key);"
77
+ " process.stdout.write(sig.toString('base64'));"
78
+ "});"
79
+ )
80
+
81
+
82
+ def node_bin() -> str:
83
+ """The node executable. $NODE overrides, mirroring $PYTHON for the CLI."""
84
+ return os.environ.get("NODE") or "node"
85
+
86
+
87
+ def _sign_rs256(private_key_pem: str, data: bytes) -> bytes:
88
+ """RSASSA-PKCS1-v1_5 over SHA-256, the `RS256` a Google JWT needs."""
89
+ node = node_bin()
90
+ if not shutil.which(node):
91
+ raise RuntimeError(
92
+ f"{node} is not on PATH, and the service-account JWT is signed with it. "
93
+ "Install Node 20+ (it is required for the dashboard anyway), or set "
94
+ "$NODE to its path.")
95
+ payload = json.dumps({"key": private_key_pem,
96
+ "data": base64.b64encode(data).decode()})
97
+ p = subprocess.run([node, "-e", _NODE_SIGN_JS],
98
+ input=payload, capture_output=True, text=True)
99
+ if p.returncode != 0 or not p.stdout.strip():
100
+ raise RuntimeError(
101
+ "could not sign the service-account JWT — the key file's "
102
+ f"private_key may be malformed: {(p.stderr or '').strip()[:200]}")
103
+ return base64.b64decode(p.stdout.strip())
104
+
105
+
61
106
  def key_path() -> str | None:
62
107
  """The service-account key to sign with.
63
108
 
@@ -82,7 +127,7 @@ def _sa_key_token(scope: str) -> str:
82
127
  raise RuntimeError(
83
128
  "google.auth is service-account-key but no key file was found at "
84
129
  f"{kp or '(unset)'} — see docs/SETUP-GOOGLE.md")
85
- key = json.loads(open(kp).read())
130
+ key = json.loads(Path(kp).read_text(encoding="utf-8"))
86
131
  # Backdate slightly: Google rejects a JWT issued in its future, so a
87
132
  # machine whose clock runs a few seconds fast otherwise fails to
88
133
  # authenticate at all, with an error that names nothing useful.
@@ -93,19 +138,7 @@ def _sa_key_token(scope: str) -> str:
93
138
  "iat": iat, "exp": iat + 3600,
94
139
  }).encode())
95
140
  signing_input = f"{header}.{claims}".encode()
96
- # openssl needs the private key in a file; keep it 0600 and short-lived.
97
- fd, tmp = tempfile.mkstemp(prefix="n-seo-", suffix=".pem")
98
- try:
99
- os.fchmod(fd, 0o600)
100
- with os.fdopen(fd, "w") as f:
101
- f.write(key["private_key"])
102
- sig = subprocess.run(["openssl", "dgst", "-sha256", "-sign", tmp],
103
- input=signing_input, capture_output=True, check=True).stdout
104
- finally:
105
- try:
106
- os.unlink(tmp)
107
- except OSError:
108
- pass
141
+ sig = _sign_rs256(key["private_key"], signing_input)
109
142
  assertion = signing_input.decode() + "." + _b64url(sig)
110
143
  resp = curl_json([
111
144
  "-X", "POST", "-d", f"grant_type=urn:ietf:params:oauth:grant-type:jwt-bearer&assertion={assertion}",
@@ -222,7 +255,7 @@ def service_account_email() -> str | None:
222
255
  kp = key_path()
223
256
  if kp and os.path.exists(kp):
224
257
  try:
225
- return json.loads(open(kp).read()).get("client_email")
258
+ return json.loads(Path(kp).read_text(encoding="utf-8")).get("client_email")
226
259
  except (OSError, json.JSONDecodeError):
227
260
  return None
228
261
  return None
@@ -96,7 +96,7 @@ def main():
96
96
  continue
97
97
  (site_dir / f"{name}.json").write_text(json.dumps(
98
98
  {"site": host, "property": f"properties/{prop}",
99
- "pulled": date.today().isoformat(), **resp}))
99
+ "pulled": date.today().isoformat(), **resp}), encoding="utf-8")
100
100
  print(f"{host:28s} {name:8s} {resp.get('rowCount', 0)} rows")
101
101
 
102
102
  print(f"\nSaved under {out_root}")
@@ -100,7 +100,7 @@ def main():
100
100
  "rowCount": len(result["rows"]),
101
101
  "rows": result["rows"],
102
102
  }
103
- (site_dir / f"{name}{suffix}.json").write_text(json.dumps(payload))
103
+ (site_dir / f"{name}{suffix}.json").write_text(json.dumps(payload), encoding="utf-8")
104
104
  print(f"{slug:28s} {name}{suffix:5s} {len(result['rows'])} rows")
105
105
 
106
106
  print(f"\nWindows: {start_full} and {start_recent} -> {end}\nSaved under {out_root}")
@@ -170,7 +170,7 @@ def main() -> int:
170
170
  f"({len(never_crawled)} never crawled)")
171
171
 
172
172
  OUT.parent.mkdir(parents=True, exist_ok=True)
173
- OUT.write_text(json.dumps(out, indent=2))
173
+ OUT.write_text(json.dumps(out, indent=2), encoding="utf-8")
174
174
  print(f"saved {OUT} — {total_problems} URLs needing attention")
175
175
  return 0
176
176
 
@@ -87,7 +87,7 @@ def main():
87
87
  if failed:
88
88
  continue
89
89
  (OUT / f"gsc-{slug}.json").write_text(json.dumps(
90
- {"site": site, "startDate": start, "endDate": end, "rows": rows}))
90
+ {"site": site, "startDate": start, "endDate": end, "rows": rows}), encoding="utf-8")
91
91
  print(f"gsc {slug:28s} {len(rows)} date x page rows")
92
92
 
93
93
  ga4_props = seo_config.ga4_properties()
@@ -104,7 +104,7 @@ def main():
104
104
  "page": row["dimensionValues"][1]["value"],
105
105
  "sessions": float(row["metricValues"][0]["value"])}
106
106
  for row in batch]
107
- (OUT / f"ga4-{host}.json").write_text(json.dumps({"site": host, "rows": rows}))
107
+ (OUT / f"ga4-{host}.json").write_text(json.dumps({"site": host, "rows": rows}), encoding="utf-8")
108
108
  print(f"ga4 {host:28s} {len(rows)} date x page rows")
109
109
 
110
110
  batch, ok = ga4_series(tok, prop, ["date", "sessionSource", "sessionMedium"],
@@ -118,7 +118,7 @@ def main():
118
118
  "sessions": float(row["metricValues"][0]["value"])}
119
119
  for row in batch]
120
120
  (OUT / f"ga4-sources-{host}.json").write_text(
121
- json.dumps({"site": host, "rows": rows}))
121
+ json.dumps({"site": host, "rows": rows}), encoding="utf-8")
122
122
  print(f"ga4 {host:28s} {len(rows)} date x source rows")
123
123
 
124
124
  if not gsc_props and not ga4_props:
@@ -9,8 +9,24 @@ and export. Stdlib only.
9
9
  import json
10
10
  import os
11
11
  import subprocess
12
+ import sys
12
13
  from pathlib import Path
13
14
 
15
+ # Print UTF-8 whatever the platform thinks the console encoding is.
16
+ #
17
+ # Windows picks the ANSI code page (cp1252 on a US install) for a piped
18
+ # stdout, so the em dashes and middle dots this tool prints come out as
19
+ # mojibake or raise UnicodeEncodeError mid-run. Every entry point imports
20
+ # this module, so fixing it once here covers all of them. File I/O is
21
+ # handled separately: every read_text/write_text/open call passes
22
+ # encoding="utf-8" explicitly, and tests/py/test_portability.py enforces it.
23
+ for _stream in (sys.stdout, sys.stderr):
24
+ try:
25
+ if getattr(_stream, "encoding", "").lower().replace("-", "") != "utf8":
26
+ _stream.reconfigure(encoding="utf-8")
27
+ except (AttributeError, ValueError, OSError):
28
+ pass # a captured or replaced stream; nothing to fix
29
+
14
30
  # The engine checkout: code, engine docs, public assets.
15
31
  ROOT = Path(__file__).resolve().parent.parent
16
32
  # The instance: one user's config, queue, content and data. Defaults to the
@@ -50,7 +66,7 @@ def load(force: bool = False) -> dict:
50
66
  if _cache is not None and not force:
51
67
  return _cache
52
68
  path = CONFIG_PATH if CONFIG_PATH.exists() else EXAMPLE_PATH
53
- raw = json.loads(path.read_text())
69
+ raw = json.loads(path.read_text(encoding="utf-8"))
54
70
  modules = {k: {"enabled": False, **MODULE_DEFAULTS.get(k, {})} for k in MODULE_KEYS}
55
71
  for k, v in (raw.get("modules") or {}).items():
56
72
  modules[k] = {"enabled": False, **MODULE_DEFAULTS.get(k, {}), **(v or {})}
@@ -96,7 +112,7 @@ def load(force: bool = False) -> dict:
96
112
  def engine_info() -> dict:
97
113
  """Mirrors src/config.ts engineInfo(): what engine is running, for which instance."""
98
114
  try:
99
- version = json.loads((ROOT / "package.json").read_text()).get("version", "0.0.0")
115
+ version = json.loads((ROOT / "package.json").read_text(encoding="utf-8")).get("version", "0.0.0")
100
116
  except (OSError, json.JSONDecodeError):
101
117
  version = "0.0.0"
102
118
  try:
@@ -204,7 +220,7 @@ def env(key: str, default: str = "") -> str:
204
220
  if os.environ.get(key):
205
221
  return os.environ[key]
206
222
  try:
207
- for line in (INSTANCE / ".env").read_text().splitlines():
223
+ for line in (INSTANCE / ".env").read_text(encoding="utf-8").splitlines():
208
224
  line = line.strip()
209
225
  if line.startswith(f"{key}="):
210
226
  return line.split("=", 1)[1].strip().strip('"').strip("'")
package/ops/daily.py CHANGED
@@ -57,7 +57,7 @@ STEPS = [
57
57
  def log(line: str):
58
58
  print(line, flush=True)
59
59
  LOG.parent.mkdir(parents=True, exist_ok=True)
60
- with LOG.open("a") as f:
60
+ with LOG.open("a", encoding="utf-8") as f:
61
61
  f.write(line.rstrip("\n") + "\n")
62
62
 
63
63
 
@@ -230,7 +230,7 @@ def main():
230
230
  "ts": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%MZ"),
231
231
  "failures": "; ".join(failures),
232
232
  "steps": results,
233
- }, indent=1))
233
+ }, indent=1), encoding="utf-8")
234
234
 
235
235
  write_last_run()
236
236
  if run_hooks_flag and hooks["afterRun"]:
package/ops/daily_diff.py CHANGED
@@ -31,7 +31,7 @@ PROBE_KEYS = [
31
31
 
32
32
  def probes():
33
33
  files = sorted((DATA / "probes").glob("probe-*.json"))
34
- return [json.loads(f.read_text()) for f in files[-2:]]
34
+ return [json.loads(f.read_text(encoding="utf-8")) for f in files[-2:]]
35
35
 
36
36
 
37
37
  def watch_groups():
@@ -76,7 +76,7 @@ def main():
76
76
  if not f.exists():
77
77
  continue
78
78
  win = "90d" if f90.exists() else "16mo"
79
- by_url = {r["keys"][0]: r for r in json.loads(f.read_text())["rows"]}
79
+ by_url = {r["keys"][0]: r for r in json.loads(f.read_text(encoding="utf-8"))["rows"]}
80
80
  for url in urls:
81
81
  r = by_url.get(url) or by_url.get(url.rstrip("/")) or by_url.get(url + "/")
82
82
  label = url.split("//", 1)[-1]
@@ -90,7 +90,7 @@ def main():
90
90
  conv = seo_config.load()["conversions"]
91
91
  if conv:
92
92
  try:
93
- fr = json.loads((DATA / "ga4" / conv["site"] / "funnel.json").read_text())
93
+ fr = json.loads((DATA / "ga4" / conv["site"] / "funnel.json").read_text(encoding="utf-8"))
94
94
  rows = fr.get("rows", [])
95
95
  if rows:
96
96
  counts = {}
@@ -113,7 +113,7 @@ def main():
113
113
  # subdomain's own traffic as a referral from the apex
114
114
  others = {h.removeprefix("www.") for h in hosts if h != host}
115
115
  xref = []
116
- for r in json.loads(p.read_text()).get("rows", []):
116
+ for r in json.loads(p.read_text(encoding="utf-8")).get("rows", []):
117
117
  name = r["dimensionValues"][0]["value"]
118
118
  if name.lower().removeprefix("www.") in others:
119
119
  xref.append(f"{name}={float(r['metricValues'][0]['value']):.0f}")
@@ -123,7 +123,8 @@ def main():
123
123
  log = INSTANCE / "docs" / "daily-log.md"
124
124
  log.parent.mkdir(parents=True, exist_ok=True)
125
125
  if not log.exists():
126
- log.write_text("# Daily ops log\n\nAppended by ops/daily.py — newest entries last.\n")
126
+ log.write_text("# Daily ops log\n\nAppended by ops/daily.py — newest entries last.\n",
127
+ encoding="utf-8")
127
128
  today = date.today().isoformat()
128
129
  entry = [f"\n## {today}\n"]
129
130
  entry += [f"- **ALERT:** {a}" for a in alerts]
@@ -132,15 +133,15 @@ def main():
132
133
 
133
134
  # Re-running on the same day must replace today's entry, not stack a
134
135
  # second one under the same heading.
135
- text = log.read_text()
136
+ text = log.read_text(encoding="utf-8")
136
137
  head = f"\n## {today}\n"
137
138
  start = text.find(head)
138
139
  if start == -1:
139
- log.write_text(text.rstrip("\n") + "\n" + body)
140
+ log.write_text(text.rstrip("\n") + "\n" + body, encoding="utf-8")
140
141
  else:
141
142
  nxt = text.find("\n## ", start + len(head))
142
143
  tail = text[nxt:] if nxt != -1 else ""
143
- log.write_text(text[:start].rstrip("\n") + "\n" + body + tail.lstrip("\n"))
144
+ log.write_text(text[:start].rstrip("\n") + "\n" + body + tail.lstrip("\n"), encoding="utf-8")
144
145
  for a in alerts:
145
146
  print(f"ALERT: {a}")
146
147
  print(f"daily-log updated ({len(alerts)} alerts, {len(lines)} lines)")
package/ops/demo_data.py CHANGED
@@ -151,7 +151,7 @@ def write_gsc(prop, slug, sites_in_prop, index_of):
151
151
  ("queries", ["query"], agg(by_q)),
152
152
  ("pages", ["page"], agg(by_p))):
153
153
  (d / f"{name}{suffix}.json").write_text(json.dumps(
154
- {**meta, "dimensions": dims, "rowCount": len(rows_), "rows": rows_}))
154
+ {**meta, "dimensions": dims, "rowCount": len(rows_), "rows": rows_}), encoding="utf-8")
155
155
 
156
156
  dataset(full_rows, 1.0, "", END - timedelta(days=FULL_DAYS))
157
157
  dataset(full_rows, 0.22, "_90d", END - timedelta(days=RECENT_DAYS))
@@ -168,7 +168,7 @@ def write_gsc(prop, slug, sites_in_prop, index_of):
168
168
  "ctr": clicks / imps if imps else 0, "position": round(rng.uniform(9, 14), 1)})
169
169
  (d / "dates.json").write_text(json.dumps(
170
170
  {"site": prop, "dimensions": ["date"], "startDate": day0.isoformat(), "endDate": END.isoformat(),
171
- "rowCount": len(dates), "rows": dates}))
171
+ "rowCount": len(dates), "rows": dates}), encoding="utf-8")
172
172
  return full_rows
173
173
 
174
174
 
@@ -199,14 +199,14 @@ def write_ga4(site, idx, conv, gsc_rows):
199
199
  total = sum(m[0] for _, m in daily)
200
200
  meta = {"site": host, "property": f"properties/{site.get('ga4Property') or '000000000'}",
201
201
  "pulled": TODAY.isoformat()}
202
- (d / "daily.json").write_text(json.dumps({**meta, **ga4_report(["date"], ["sessions", "totalUsers"], daily)}))
202
+ (d / "daily.json").write_text(json.dumps({**meta, **ga4_report(["date"], ["sessions", "totalUsers"], daily)}), encoding="utf-8")
203
203
 
204
204
  mix = [(src, med, w) for src, med, w, _ai in SOURCE_MIX]
205
205
  others = [h for h in seo_config.hosts() if h != host]
206
206
  if others:
207
207
  mix.append((others[0], "referral", 0.015))
208
208
  sources = [([src, med], [round(total * w), round(total * w * 0.8)]) for src, med, w in mix]
209
- (d / "sources.json").write_text(json.dumps({**meta, **ga4_report(["sessionSource", "sessionMedium"], ["sessions", "totalUsers"], sources)}))
209
+ (d / "sources.json").write_text(json.dumps({**meta, **ga4_report(["sessionSource", "sessionMedium"], ["sessions", "totalUsers"], sources)}), encoding="utf-8")
210
210
 
211
211
  pages = sorted({r[1] for r in gsc_rows if r[1].split("/")[2] == site["gscHost"]})
212
212
  paths = [p.split(site["gscHost"], 1)[1].rstrip("/") or "/" for p in pages] or ["/", "/docs", "/pricing"]
@@ -219,7 +219,7 @@ def write_ga4(site, idx, conv, gsc_rows):
219
219
  eng = 0.18
220
220
  landing.append(([p], [sessions, round(eng, 4)]))
221
221
  landing.sort(key=lambda r: -r[1][0])
222
- (d / "landing.json").write_text(json.dumps({**meta, **ga4_report(["landingPage"], ["sessions", "engagementRate"], landing)}))
222
+ (d / "landing.json").write_text(json.dumps({**meta, **ga4_report(["landingPage"], ["sessions", "engagementRate"], landing)}), encoding="utf-8")
223
223
 
224
224
  if conv and conv.get("site") == host:
225
225
  rows = []
@@ -227,7 +227,7 @@ def write_ga4(site, idx, conv, gsc_rows):
227
227
  day = (day0 + timedelta(days=i)).strftime("%Y%m%d")
228
228
  for ev in conv.get("events", [])[:2]:
229
229
  rows.append(([day, ev, rng.choice(["web", "docs", "(not set)"])], [rng.randint(0, 4)]))
230
- (d / "funnel.json").write_text(json.dumps({**meta, **ga4_report(["date", "eventName", conv.get("sourceDimension") or "customEvent:source_app"], ["eventCount"], rows)}))
230
+ (d / "funnel.json").write_text(json.dumps({**meta, **ga4_report(["date", "eventName", conv.get("sourceDimension") or "customEvent:source_app"], ["eventCount"], rows)}), encoding="utf-8")
231
231
  return paths, total
232
232
 
233
233
 
@@ -250,7 +250,7 @@ def write_timeseries(prop, slug, sites_in_prop, gsc_rows, ga4_paths):
250
250
  rows.append({"keys": [day.isoformat(), p], "clicks": round(dc), "impressions": round(di),
251
251
  "ctr": (dc / di) if di else 0, "position": round(rng.uniform(4, 20), 1)})
252
252
  (d / f"gsc-{slug}.json").write_text(json.dumps(
253
- {"site": prop, "startDate": day0.isoformat(), "endDate": END.isoformat(), "rows": rows}))
253
+ {"site": prop, "startDate": day0.isoformat(), "endDate": END.isoformat(), "rows": rows}), encoding="utf-8")
254
254
  for s in sites_in_prop:
255
255
  paths, total = ga4_paths.get(s["host"], ([], 0))
256
256
  if not s.get("ga4Property"):
@@ -263,7 +263,7 @@ def write_timeseries(prop, slug, sites_in_prop, gsc_rows, ga4_paths):
263
263
  v = weekly(day, (total / 90) * (0.35 if j == 0 else 0.08), day0=ga0)
264
264
  if round(v):
265
265
  rows.append({"date": day.strftime("%Y%m%d"), "page": p, "sessions": round(v)})
266
- (d / f"ga4-{s['host']}.json").write_text(json.dumps({"site": s["host"], "rows": rows}))
266
+ (d / f"ga4-{s['host']}.json").write_text(json.dumps({"site": s["host"], "rows": rows}), encoding="utf-8")
267
267
 
268
268
  # date x source/medium. AI assistants grow over the window and the
269
269
  # rest hold roughly flat, so the demo actually shows the thing the
@@ -278,7 +278,7 @@ def write_timeseries(prop, slug, sites_in_prop, gsc_rows, ga4_paths):
278
278
  rows.append({"date": day.strftime("%Y%m%d"), "source": src,
279
279
  "medium": med, "sessions": round(v)})
280
280
  (d / f"ga4-sources-{s['host']}.json").write_text(
281
- json.dumps({"site": s["host"], "rows": rows}))
281
+ json.dumps({"site": s["host"], "rows": rows}), encoding="utf-8")
282
282
 
283
283
 
284
284
  def write_probe(sites):
@@ -302,7 +302,7 @@ def write_probe(sites):
302
302
  "soft_404": {"status": 404, "real_404": True},
303
303
  })
304
304
  now = datetime.now(timezone.utc)
305
- (d / f"probe-{now:%Y%m%d-%H%M%S}.json").write_text(json.dumps({"probed_at": now.isoformat(), "sites": out}, indent=2))
305
+ (d / f"probe-{now:%Y%m%d-%H%M%S}.json").write_text(json.dumps({"probed_at": now.isoformat(), "sites": out}, indent=2), encoding="utf-8")
306
306
 
307
307
 
308
308
  def write_metadata_audit(sites, gsc_by_host):
@@ -333,7 +333,7 @@ def write_metadata_audit(sites, gsc_by_host):
333
333
  "imps": imps, "clicks": sum(q["clicks"] for q in qs), "issues": issues,
334
334
  "top_queries": qs[:5], "missed_clicks_window": missed})
335
335
  audit["sites"][s["host"]] = findings
336
- (DATA / "metadata-audit.json").write_text(json.dumps(audit, indent=1))
336
+ (DATA / "metadata-audit.json").write_text(json.dumps(audit, indent=1), encoding="utf-8")
337
337
 
338
338
 
339
339
  def write_index_status(sites, gsc_by_host):
@@ -365,7 +365,7 @@ def write_index_status(sites, gsc_by_host):
365
365
  "pending": False, "errors": 0, "warnings": 0}]},
366
366
  "problems": problems,
367
367
  }
368
- (DATA / "index-status.json").write_text(json.dumps(out, indent=2))
368
+ (DATA / "index-status.json").write_text(json.dumps(out, indent=2), encoding="utf-8")
369
369
 
370
370
 
371
371
  def write_trends(props, sites, gsc_rows_by_prop):
@@ -399,7 +399,7 @@ def write_trends(props, sites, gsc_rows_by_prop):
399
399
 
400
400
  monthly = {}
401
401
  dates_file = DATA / "gsc" / slug / "dates.json"
402
- for r in json.loads(dates_file.read_text())["rows"]:
402
+ for r in json.loads(dates_file.read_text(encoding="utf-8"))["rows"]:
403
403
  m = r["keys"][0][:7]
404
404
  cur = monthly.setdefault(m, {"clicks": 0, "imps": 0})
405
405
  cur["clicks"] += r["clicks"]
@@ -417,7 +417,7 @@ def write_trends(props, sites, gsc_rows_by_prop):
417
417
  total[ym] = t
418
418
  ai[ym] = round(t * (0.01 + (12 - k) * 0.004))
419
419
  out["ai_referrals"][s["host"]] = {"ai": ai, "total": total}
420
- (DATA / f"trends-{TODAY.isoformat()}.json").write_text(json.dumps(out, indent=1))
420
+ (DATA / f"trends-{TODAY.isoformat()}.json").write_text(json.dumps(out, indent=1), encoding="utf-8")
421
421
 
422
422
 
423
423
  def write_proposals(sites, gsc_by_host):
@@ -450,7 +450,7 @@ def write_proposals(sites, gsc_by_host):
450
450
  "evidence": "DEMO DATA — 12 days since the change; CTR up from 1.9% to 2.6% but the 28-day window has not closed."}],
451
451
  "inference_ran": True,
452
452
  }
453
- (DATA / "opportunity-proposals.json").write_text(json.dumps(out, indent=1))
453
+ (DATA / "opportunity-proposals.json").write_text(json.dumps(out, indent=1), encoding="utf-8")
454
454
 
455
455
 
456
456
  def write_run_files():
@@ -459,8 +459,8 @@ def write_run_files():
459
459
  "ts": ts.strftime("%Y-%m-%dT%H:%MZ"), "failures": "",
460
460
  "steps": [{"name": n, "ok": True, "seconds": s} for n, s in
461
461
  (("probe", 6.2), ("gsc", 41.0), ("ga4", 9.8), ("timeseries", 22.4), ("metadata-audit", 14.1),
462
- ("index-status", 38.5), ("opportunity-scan", 17.0), ("daily-diff", 0.3))]}, indent=1))
463
- with (DATA / "daily-ops.log").open("a") as f:
462
+ ("index-status", 38.5), ("opportunity-scan", 17.0), ("daily-diff", 0.3))]}, indent=1), encoding="utf-8")
463
+ with (DATA / "daily-ops.log").open("a", encoding="utf-8") as f:
464
464
  f.write(f"=== daily run {ts:%Y-%m-%d %H:%M} (demo data) ===\n")
465
465
  f.write("--- probe\n 2 sites probed\n--- gsc\n 2 properties pulled\n--- daily-diff\n daily-log updated (0 alerts)\n")
466
466
  f.write("=== done (0 failures) ===\n")
package/ops/doctor.py CHANGED
@@ -64,10 +64,19 @@ def check_config():
64
64
 
65
65
  def check_tools():
66
66
  print("tools")
67
- for tool, why in (("curl", "all HTTP goes through curl"), ("openssl", "signs the service-account JWT")):
68
- report("OK" if shutil.which(tool) else "FAIL", tool, "" if shutil.which(tool) else f"install it — {why}")
67
+ report("OK" if shutil.which("curl") else "FAIL", "curl",
68
+ "" if shutil.which("curl") else "install it — all HTTP goes through curl "
69
+ "(Windows 10+ ships it in System32)")
70
+ # node signs the service-account JWT, so it is required even for someone
71
+ # who never opens the dashboard.
72
+ import google_auth
73
+ node = google_auth.node_bin()
74
+ report("OK" if shutil.which(node) else "FAIL", f"node ({node})",
75
+ "" if shutil.which(node) else "install Node 20+ — it runs the dashboard "
76
+ "and signs the service-account JWT; $NODE overrides the path")
69
77
  v = sys.version_info
70
- report("OK" if v >= (3, 10) else "FAIL", f"python {v.major}.{v.minor}",
78
+ report("OK" if v >= (3, 10) else "FAIL",
79
+ f"python {v.major}.{v.minor} ({Path(sys.executable).name})",
71
80
  "" if v >= (3, 10) else "3.10+ required")
72
81
  nm = seo_config.ROOT / "node_modules"
73
82
  report("OK" if nm.exists() else "WARN", "node_modules", "" if nm.exists() else "run: npm install")
@@ -86,7 +95,7 @@ def check_auth(cfg):
86
95
  f"{kp or '(unset)'} — set google.serviceAccountKey or $GOOGLE_APPLICATION_CREDENTIALS; see docs/SETUP-GOOGLE.md")
87
96
  return None
88
97
  try:
89
- key = json.loads(Path(kp).read_text())
98
+ key = json.loads(Path(kp).read_text(encoding="utf-8"))
90
99
  except (OSError, json.JSONDecodeError) as exc:
91
100
  report("FAIL", "key file unreadable", str(exc))
92
101
  return None
@@ -197,7 +206,6 @@ def check_modules(cfg):
197
206
  report("OK", "enabled: " + (", ".join(on) or "(none beyond defaults)"))
198
207
  llm = cfg["modules"].get("llm", {})
199
208
  if llm.get("enabled"):
200
- import shlex
201
209
  sys.path.insert(0, str(Path(__file__).resolve().parent))
202
210
  import llm as llm_mod
203
211
  http = llm_mod.http_config()
@@ -211,11 +219,17 @@ def check_modules(cfg):
211
219
  f"check provider/model and that {llm['http'].get('apiKeyEnv') or 'apiKeyEnv'} "
212
220
  "is set in the environment or .env")
213
221
  for key in ("command", "fastCommand"):
214
- cmd = shlex.split(str(llm.get(key) or ""))
222
+ raw = str(llm.get(key) or "").strip()
223
+ cmd = llm_mod.split_command(raw)
215
224
  if not cmd:
225
+ if raw:
226
+ # Configured but unparseable — almost always an unbalanced
227
+ # quote. Say so, or the next line reads as "not configured".
228
+ report("FAIL", f"llm.{key} could not be parsed: {raw[:60]}",
229
+ "check for an unbalanced quote")
216
230
  # Only complain about a missing command when there is no http
217
231
  # block at all; a broken one already reported itself.
218
- if key == "command" and not http and not llm.get("http"):
232
+ elif key == "command" and not http and not llm.get("http"):
219
233
  report("FAIL", "llm has neither http nor command configured")
220
234
  continue
221
235
  report("OK" if shutil.which(cmd[0]) else "FAIL", f"llm.{key}: {cmd[0]}",
@@ -103,11 +103,11 @@ def main():
103
103
  html = html.replace("</body>", staleness + "</body>", 1)
104
104
  dest = out / "index.html" if route == "/" else out / route.lstrip("/") / "index.html"
105
105
  dest.parent.mkdir(parents=True, exist_ok=True)
106
- dest.write_text(html)
106
+ dest.write_text(html, encoding="utf-8")
107
107
 
108
- (out / "styles.css").write_text(fetch("/styles.css"))
109
- (out / "favicon.svg").write_text(fetch("/favicon.svg"))
110
- (out / "robots.txt").write_text("User-agent: *\nDisallow: /\n")
108
+ (out / "styles.css").write_text(fetch("/styles.css"), encoding="utf-8")
109
+ (out / "favicon.svg").write_text(fetch("/favicon.svg"), encoding="utf-8")
110
+ (out / "robots.txt").write_text("User-agent: *\nDisallow: /\n", encoding="utf-8")
111
111
  except BaseException:
112
112
  # Never leave a half-built staging directory behind; the next
113
113
  # run would otherwise start from someone else's leftovers.
package/ops/hn_digest.py CHANGED
@@ -157,7 +157,7 @@ def main():
157
157
  seo_config.DATA.mkdir(parents=True, exist_ok=True)
158
158
  (seo_config.DATA / "hn-digest.json").write_text(json.dumps(
159
159
  {"generated": datetime.now(timezone.utc).isoformat(timespec="minutes"),
160
- "stats": stats, "picks": picks}, indent=1))
160
+ "stats": stats, "picks": picks}, indent=1), encoding="utf-8")
161
161
  for p in picks:
162
162
  print(f"- {p['title']} ({p['comments']}c/{p['points']}p) {p['url']} [{p['why']}]")
163
163
  if not picks:
package/ops/indexnow.py CHANGED
@@ -33,12 +33,12 @@ def key_file() -> Path:
33
33
  def cmd_init() -> int:
34
34
  kf = key_file()
35
35
  if kf.exists():
36
- key = kf.read_text().strip()
36
+ key = kf.read_text(encoding="utf-8").strip()
37
37
  print(f"key file already exists: {kf}")
38
38
  else:
39
39
  key = secrets.token_hex(16)
40
40
  kf.parent.mkdir(parents=True, exist_ok=True)
41
- kf.write_text(key + "\n")
41
+ kf.write_text(key + "\n", encoding="utf-8")
42
42
  print(f"wrote {kf}")
43
43
  print(f"\nkey: {key}\n\nServe this key as plain text at, for every site:")
44
44
  for s in seo_config.sites():
@@ -52,7 +52,7 @@ def cmd_ping(urls: list[str]) -> int:
52
52
  if not kf.exists():
53
53
  print(f"no key file at {kf} — run: python3 ops/indexnow.py init")
54
54
  return 1
55
- key = kf.read_text().strip()
55
+ key = kf.read_text(encoding="utf-8").strip()
56
56
  by_host: dict[str, list[str]] = {}
57
57
  for u in urls:
58
58
  host = urlsplit(u).netloc
package/ops/llm.py CHANGED
@@ -28,6 +28,7 @@ path runs. Nothing that comes back is applied automatically: callers turn
28
28
  replies into briefings or proposals a human accepts or ignores.
29
29
  """
30
30
  import json
31
+ import os
31
32
  import shlex
32
33
  import shutil
33
34
  import subprocess
@@ -42,6 +43,27 @@ ANTHROPIC_VERSION = "2023-06-01"
42
43
  MAX_TOKENS = 4096
43
44
 
44
45
 
46
+ def split_command(cmd: str) -> list[str]:
47
+ """Split a configured command string into argv, per platform.
48
+
49
+ POSIX rules treat a backslash as an escape, so on Windows they quietly
50
+ turn `C:\\tools\\claude.exe` into `C:toolsclaude.exe`. There, split with
51
+ the non-POSIX rules cmd.exe uses and strip the quotes shlex leaves on.
52
+
53
+ Returns [] for anything unparseable. This string comes from the Settings
54
+ form, so a stray quote is a typo a user can make, and it must not take
55
+ the daily run down with a ValueError from inside shlex — callers already
56
+ treat [] as "no command configured". doctor tells them which it was.
57
+ """
58
+ try:
59
+ if os.name != "nt":
60
+ return shlex.split(cmd)
61
+ parts = shlex.split(cmd, posix=False)
62
+ except ValueError:
63
+ return []
64
+ return [p[1:-1] if len(p) > 1 and p[0] == p[-1] == '"' else p for p in parts]
65
+
66
+
45
67
  def _enabled() -> dict | None:
46
68
  m = seo_config.module("llm")
47
69
  return m if m.get("enabled") else None
@@ -71,7 +93,7 @@ def command(fast: bool = False) -> list[str] | None:
71
93
  if not m:
72
94
  return None
73
95
  cmd = (m.get("fastCommand") if fast else None) or m.get("command") or ""
74
- parts = shlex.split(str(cmd))
96
+ parts = split_command(str(cmd))
75
97
  if not parts or not shutil.which(parts[0]):
76
98
  return None
77
99
  return parts