n-seo 0.1.1 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +104 -41
- package/bin/n-seo.mjs +14 -4
- package/docs/FAQ.md +14 -3
- package/docs/MCP.md +25 -9
- package/docs/PRD.md +6 -4
- package/docs/SCHEDULING.md +42 -5
- package/docs/SETUP-GOOGLE.md +1 -1
- package/ingest/__pycache__/analyze_metadata.cpython-312.pyc +0 -0
- package/ingest/__pycache__/google_auth.cpython-312.pyc +0 -0
- package/ingest/__pycache__/http_util.cpython-312.pyc +0 -0
- package/ingest/__pycache__/seo_config.cpython-312.pyc +0 -0
- package/ingest/analyze_ga4.py +2 -2
- package/ingest/analyze_gsc.py +2 -2
- package/ingest/analyze_metadata.py +2 -2
- package/ingest/analyze_trends.py +2 -2
- package/ingest/google_auth.py +53 -20
- package/ingest/pull_ga4.py +1 -1
- package/ingest/pull_gsc.py +1 -1
- package/ingest/pull_index_status.py +1 -1
- package/ingest/pull_timeseries.py +3 -3
- package/ingest/seo_config.py +19 -3
- package/ops/__pycache__/daily.cpython-312.pyc +0 -0
- package/ops/__pycache__/daily_diff.cpython-312.pyc +0 -0
- package/ops/__pycache__/demo_data.cpython-312.pyc +0 -0
- package/ops/__pycache__/export_static.cpython-312.pyc +0 -0
- package/ops/__pycache__/llm.cpython-312.pyc +0 -0
- package/ops/__pycache__/publish.cpython-312.pyc +0 -0
- package/ops/daily.py +2 -2
- package/ops/daily_diff.py +9 -8
- package/ops/demo_data.py +17 -17
- package/ops/doctor.py +21 -7
- package/ops/export_static.py +4 -4
- package/ops/hn_digest.py +1 -1
- package/ops/indexnow.py +3 -3
- package/ops/llm.py +23 -1
- package/ops/opportunity_scan.py +4 -4
- package/ops/py.mjs +40 -0
- package/ops/reddit_digest.py +1 -1
- package/ops/templates/n-seo-daily-task.xml +65 -0
- package/package.json +15 -8
- package/probes/__pycache__/site_probe.cpython-312.pyc +0 -0
- package/probes/site_probe.py +1 -1
- package/src/actions.ts +1 -1
- package/src/config.ts +1 -1
- package/src/mcp.ts +2 -2
- package/src/settings.tsx +3 -3
- package/src/views.tsx +9 -9
package/ingest/pull_gsc.py
CHANGED
|
@@ -100,7 +100,7 @@ def main():
|
|
|
100
100
|
"rowCount": len(result["rows"]),
|
|
101
101
|
"rows": result["rows"],
|
|
102
102
|
}
|
|
103
|
-
(site_dir / f"{name}{suffix}.json").write_text(json.dumps(payload))
|
|
103
|
+
(site_dir / f"{name}{suffix}.json").write_text(json.dumps(payload), encoding="utf-8")
|
|
104
104
|
print(f"{slug:28s} {name}{suffix:5s} {len(result['rows'])} rows")
|
|
105
105
|
|
|
106
106
|
print(f"\nWindows: {start_full} and {start_recent} -> {end}\nSaved under {out_root}")
|
|
@@ -170,7 +170,7 @@ def main() -> int:
|
|
|
170
170
|
f"({len(never_crawled)} never crawled)")
|
|
171
171
|
|
|
172
172
|
OUT.parent.mkdir(parents=True, exist_ok=True)
|
|
173
|
-
OUT.write_text(json.dumps(out, indent=2))
|
|
173
|
+
OUT.write_text(json.dumps(out, indent=2), encoding="utf-8")
|
|
174
174
|
print(f"saved {OUT} — {total_problems} URLs needing attention")
|
|
175
175
|
return 0
|
|
176
176
|
|
|
@@ -87,7 +87,7 @@ def main():
|
|
|
87
87
|
if failed:
|
|
88
88
|
continue
|
|
89
89
|
(OUT / f"gsc-{slug}.json").write_text(json.dumps(
|
|
90
|
-
{"site": site, "startDate": start, "endDate": end, "rows": rows}))
|
|
90
|
+
{"site": site, "startDate": start, "endDate": end, "rows": rows}), encoding="utf-8")
|
|
91
91
|
print(f"gsc {slug:28s} {len(rows)} date x page rows")
|
|
92
92
|
|
|
93
93
|
ga4_props = seo_config.ga4_properties()
|
|
@@ -104,7 +104,7 @@ def main():
|
|
|
104
104
|
"page": row["dimensionValues"][1]["value"],
|
|
105
105
|
"sessions": float(row["metricValues"][0]["value"])}
|
|
106
106
|
for row in batch]
|
|
107
|
-
(OUT / f"ga4-{host}.json").write_text(json.dumps({"site": host, "rows": rows}))
|
|
107
|
+
(OUT / f"ga4-{host}.json").write_text(json.dumps({"site": host, "rows": rows}), encoding="utf-8")
|
|
108
108
|
print(f"ga4 {host:28s} {len(rows)} date x page rows")
|
|
109
109
|
|
|
110
110
|
batch, ok = ga4_series(tok, prop, ["date", "sessionSource", "sessionMedium"],
|
|
@@ -118,7 +118,7 @@ def main():
|
|
|
118
118
|
"sessions": float(row["metricValues"][0]["value"])}
|
|
119
119
|
for row in batch]
|
|
120
120
|
(OUT / f"ga4-sources-{host}.json").write_text(
|
|
121
|
-
json.dumps({"site": host, "rows": rows}))
|
|
121
|
+
json.dumps({"site": host, "rows": rows}), encoding="utf-8")
|
|
122
122
|
print(f"ga4 {host:28s} {len(rows)} date x source rows")
|
|
123
123
|
|
|
124
124
|
if not gsc_props and not ga4_props:
|
package/ingest/seo_config.py
CHANGED
|
@@ -9,8 +9,24 @@ and export. Stdlib only.
|
|
|
9
9
|
import json
|
|
10
10
|
import os
|
|
11
11
|
import subprocess
|
|
12
|
+
import sys
|
|
12
13
|
from pathlib import Path
|
|
13
14
|
|
|
15
|
+
# Print UTF-8 whatever the platform thinks the console encoding is.
|
|
16
|
+
#
|
|
17
|
+
# Windows picks the ANSI code page (cp1252 on a US install) for a piped
|
|
18
|
+
# stdout, so the em dashes and middle dots this tool prints come out as
|
|
19
|
+
# mojibake or raise UnicodeEncodeError mid-run. Every entry point imports
|
|
20
|
+
# this module, so fixing it once here covers all of them. File I/O is
|
|
21
|
+
# handled separately: every read_text/write_text/open call passes
|
|
22
|
+
# encoding="utf-8" explicitly, and tests/py/test_portability.py enforces it.
|
|
23
|
+
for _stream in (sys.stdout, sys.stderr):
|
|
24
|
+
try:
|
|
25
|
+
if getattr(_stream, "encoding", "").lower().replace("-", "") != "utf8":
|
|
26
|
+
_stream.reconfigure(encoding="utf-8")
|
|
27
|
+
except (AttributeError, ValueError, OSError):
|
|
28
|
+
pass # a captured or replaced stream; nothing to fix
|
|
29
|
+
|
|
14
30
|
# The engine checkout: code, engine docs, public assets.
|
|
15
31
|
ROOT = Path(__file__).resolve().parent.parent
|
|
16
32
|
# The instance: one user's config, queue, content and data. Defaults to the
|
|
@@ -50,7 +66,7 @@ def load(force: bool = False) -> dict:
|
|
|
50
66
|
if _cache is not None and not force:
|
|
51
67
|
return _cache
|
|
52
68
|
path = CONFIG_PATH if CONFIG_PATH.exists() else EXAMPLE_PATH
|
|
53
|
-
raw = json.loads(path.read_text())
|
|
69
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
54
70
|
modules = {k: {"enabled": False, **MODULE_DEFAULTS.get(k, {})} for k in MODULE_KEYS}
|
|
55
71
|
for k, v in (raw.get("modules") or {}).items():
|
|
56
72
|
modules[k] = {"enabled": False, **MODULE_DEFAULTS.get(k, {}), **(v or {})}
|
|
@@ -96,7 +112,7 @@ def load(force: bool = False) -> dict:
|
|
|
96
112
|
def engine_info() -> dict:
|
|
97
113
|
"""Mirrors src/config.ts engineInfo(): what engine is running, for which instance."""
|
|
98
114
|
try:
|
|
99
|
-
version = json.loads((ROOT / "package.json").read_text()).get("version", "0.0.0")
|
|
115
|
+
version = json.loads((ROOT / "package.json").read_text(encoding="utf-8")).get("version", "0.0.0")
|
|
100
116
|
except (OSError, json.JSONDecodeError):
|
|
101
117
|
version = "0.0.0"
|
|
102
118
|
try:
|
|
@@ -204,7 +220,7 @@ def env(key: str, default: str = "") -> str:
|
|
|
204
220
|
if os.environ.get(key):
|
|
205
221
|
return os.environ[key]
|
|
206
222
|
try:
|
|
207
|
-
for line in (INSTANCE / ".env").read_text().splitlines():
|
|
223
|
+
for line in (INSTANCE / ".env").read_text(encoding="utf-8").splitlines():
|
|
208
224
|
line = line.strip()
|
|
209
225
|
if line.startswith(f"{key}="):
|
|
210
226
|
return line.split("=", 1)[1].strip().strip('"').strip("'")
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
package/ops/daily.py
CHANGED
|
@@ -57,7 +57,7 @@ STEPS = [
|
|
|
57
57
|
def log(line: str):
|
|
58
58
|
print(line, flush=True)
|
|
59
59
|
LOG.parent.mkdir(parents=True, exist_ok=True)
|
|
60
|
-
with LOG.open("a") as f:
|
|
60
|
+
with LOG.open("a", encoding="utf-8") as f:
|
|
61
61
|
f.write(line.rstrip("\n") + "\n")
|
|
62
62
|
|
|
63
63
|
|
|
@@ -230,7 +230,7 @@ def main():
|
|
|
230
230
|
"ts": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%MZ"),
|
|
231
231
|
"failures": "; ".join(failures),
|
|
232
232
|
"steps": results,
|
|
233
|
-
}, indent=1))
|
|
233
|
+
}, indent=1), encoding="utf-8")
|
|
234
234
|
|
|
235
235
|
write_last_run()
|
|
236
236
|
if run_hooks_flag and hooks["afterRun"]:
|
package/ops/daily_diff.py
CHANGED
|
@@ -31,7 +31,7 @@ PROBE_KEYS = [
|
|
|
31
31
|
|
|
32
32
|
def probes():
|
|
33
33
|
files = sorted((DATA / "probes").glob("probe-*.json"))
|
|
34
|
-
return [json.loads(f.read_text()) for f in files[-2:]]
|
|
34
|
+
return [json.loads(f.read_text(encoding="utf-8")) for f in files[-2:]]
|
|
35
35
|
|
|
36
36
|
|
|
37
37
|
def watch_groups():
|
|
@@ -76,7 +76,7 @@ def main():
|
|
|
76
76
|
if not f.exists():
|
|
77
77
|
continue
|
|
78
78
|
win = "90d" if f90.exists() else "16mo"
|
|
79
|
-
by_url = {r["keys"][0]: r for r in json.loads(f.read_text())["rows"]}
|
|
79
|
+
by_url = {r["keys"][0]: r for r in json.loads(f.read_text(encoding="utf-8"))["rows"]}
|
|
80
80
|
for url in urls:
|
|
81
81
|
r = by_url.get(url) or by_url.get(url.rstrip("/")) or by_url.get(url + "/")
|
|
82
82
|
label = url.split("//", 1)[-1]
|
|
@@ -90,7 +90,7 @@ def main():
|
|
|
90
90
|
conv = seo_config.load()["conversions"]
|
|
91
91
|
if conv:
|
|
92
92
|
try:
|
|
93
|
-
fr = json.loads((DATA / "ga4" / conv["site"] / "funnel.json").read_text())
|
|
93
|
+
fr = json.loads((DATA / "ga4" / conv["site"] / "funnel.json").read_text(encoding="utf-8"))
|
|
94
94
|
rows = fr.get("rows", [])
|
|
95
95
|
if rows:
|
|
96
96
|
counts = {}
|
|
@@ -113,7 +113,7 @@ def main():
|
|
|
113
113
|
# subdomain's own traffic as a referral from the apex
|
|
114
114
|
others = {h.removeprefix("www.") for h in hosts if h != host}
|
|
115
115
|
xref = []
|
|
116
|
-
for r in json.loads(p.read_text()).get("rows", []):
|
|
116
|
+
for r in json.loads(p.read_text(encoding="utf-8")).get("rows", []):
|
|
117
117
|
name = r["dimensionValues"][0]["value"]
|
|
118
118
|
if name.lower().removeprefix("www.") in others:
|
|
119
119
|
xref.append(f"{name}={float(r['metricValues'][0]['value']):.0f}")
|
|
@@ -123,7 +123,8 @@ def main():
|
|
|
123
123
|
log = INSTANCE / "docs" / "daily-log.md"
|
|
124
124
|
log.parent.mkdir(parents=True, exist_ok=True)
|
|
125
125
|
if not log.exists():
|
|
126
|
-
log.write_text("# Daily ops log\n\nAppended by ops/daily.py — newest entries last.\n"
|
|
126
|
+
log.write_text("# Daily ops log\n\nAppended by ops/daily.py — newest entries last.\n",
|
|
127
|
+
encoding="utf-8")
|
|
127
128
|
today = date.today().isoformat()
|
|
128
129
|
entry = [f"\n## {today}\n"]
|
|
129
130
|
entry += [f"- **ALERT:** {a}" for a in alerts]
|
|
@@ -132,15 +133,15 @@ def main():
|
|
|
132
133
|
|
|
133
134
|
# Re-running on the same day must replace today's entry, not stack a
|
|
134
135
|
# second one under the same heading.
|
|
135
|
-
text = log.read_text()
|
|
136
|
+
text = log.read_text(encoding="utf-8")
|
|
136
137
|
head = f"\n## {today}\n"
|
|
137
138
|
start = text.find(head)
|
|
138
139
|
if start == -1:
|
|
139
|
-
log.write_text(text.rstrip("\n") + "\n" + body)
|
|
140
|
+
log.write_text(text.rstrip("\n") + "\n" + body, encoding="utf-8")
|
|
140
141
|
else:
|
|
141
142
|
nxt = text.find("\n## ", start + len(head))
|
|
142
143
|
tail = text[nxt:] if nxt != -1 else ""
|
|
143
|
-
log.write_text(text[:start].rstrip("\n") + "\n" + body + tail.lstrip("\n"))
|
|
144
|
+
log.write_text(text[:start].rstrip("\n") + "\n" + body + tail.lstrip("\n"), encoding="utf-8")
|
|
144
145
|
for a in alerts:
|
|
145
146
|
print(f"ALERT: {a}")
|
|
146
147
|
print(f"daily-log updated ({len(alerts)} alerts, {len(lines)} lines)")
|
package/ops/demo_data.py
CHANGED
|
@@ -151,7 +151,7 @@ def write_gsc(prop, slug, sites_in_prop, index_of):
|
|
|
151
151
|
("queries", ["query"], agg(by_q)),
|
|
152
152
|
("pages", ["page"], agg(by_p))):
|
|
153
153
|
(d / f"{name}{suffix}.json").write_text(json.dumps(
|
|
154
|
-
{**meta, "dimensions": dims, "rowCount": len(rows_), "rows": rows_}))
|
|
154
|
+
{**meta, "dimensions": dims, "rowCount": len(rows_), "rows": rows_}), encoding="utf-8")
|
|
155
155
|
|
|
156
156
|
dataset(full_rows, 1.0, "", END - timedelta(days=FULL_DAYS))
|
|
157
157
|
dataset(full_rows, 0.22, "_90d", END - timedelta(days=RECENT_DAYS))
|
|
@@ -168,7 +168,7 @@ def write_gsc(prop, slug, sites_in_prop, index_of):
|
|
|
168
168
|
"ctr": clicks / imps if imps else 0, "position": round(rng.uniform(9, 14), 1)})
|
|
169
169
|
(d / "dates.json").write_text(json.dumps(
|
|
170
170
|
{"site": prop, "dimensions": ["date"], "startDate": day0.isoformat(), "endDate": END.isoformat(),
|
|
171
|
-
"rowCount": len(dates), "rows": dates}))
|
|
171
|
+
"rowCount": len(dates), "rows": dates}), encoding="utf-8")
|
|
172
172
|
return full_rows
|
|
173
173
|
|
|
174
174
|
|
|
@@ -199,14 +199,14 @@ def write_ga4(site, idx, conv, gsc_rows):
|
|
|
199
199
|
total = sum(m[0] for _, m in daily)
|
|
200
200
|
meta = {"site": host, "property": f"properties/{site.get('ga4Property') or '000000000'}",
|
|
201
201
|
"pulled": TODAY.isoformat()}
|
|
202
|
-
(d / "daily.json").write_text(json.dumps({**meta, **ga4_report(["date"], ["sessions", "totalUsers"], daily)}))
|
|
202
|
+
(d / "daily.json").write_text(json.dumps({**meta, **ga4_report(["date"], ["sessions", "totalUsers"], daily)}), encoding="utf-8")
|
|
203
203
|
|
|
204
204
|
mix = [(src, med, w) for src, med, w, _ai in SOURCE_MIX]
|
|
205
205
|
others = [h for h in seo_config.hosts() if h != host]
|
|
206
206
|
if others:
|
|
207
207
|
mix.append((others[0], "referral", 0.015))
|
|
208
208
|
sources = [([src, med], [round(total * w), round(total * w * 0.8)]) for src, med, w in mix]
|
|
209
|
-
(d / "sources.json").write_text(json.dumps({**meta, **ga4_report(["sessionSource", "sessionMedium"], ["sessions", "totalUsers"], sources)}))
|
|
209
|
+
(d / "sources.json").write_text(json.dumps({**meta, **ga4_report(["sessionSource", "sessionMedium"], ["sessions", "totalUsers"], sources)}), encoding="utf-8")
|
|
210
210
|
|
|
211
211
|
pages = sorted({r[1] for r in gsc_rows if r[1].split("/")[2] == site["gscHost"]})
|
|
212
212
|
paths = [p.split(site["gscHost"], 1)[1].rstrip("/") or "/" for p in pages] or ["/", "/docs", "/pricing"]
|
|
@@ -219,7 +219,7 @@ def write_ga4(site, idx, conv, gsc_rows):
|
|
|
219
219
|
eng = 0.18
|
|
220
220
|
landing.append(([p], [sessions, round(eng, 4)]))
|
|
221
221
|
landing.sort(key=lambda r: -r[1][0])
|
|
222
|
-
(d / "landing.json").write_text(json.dumps({**meta, **ga4_report(["landingPage"], ["sessions", "engagementRate"], landing)}))
|
|
222
|
+
(d / "landing.json").write_text(json.dumps({**meta, **ga4_report(["landingPage"], ["sessions", "engagementRate"], landing)}), encoding="utf-8")
|
|
223
223
|
|
|
224
224
|
if conv and conv.get("site") == host:
|
|
225
225
|
rows = []
|
|
@@ -227,7 +227,7 @@ def write_ga4(site, idx, conv, gsc_rows):
|
|
|
227
227
|
day = (day0 + timedelta(days=i)).strftime("%Y%m%d")
|
|
228
228
|
for ev in conv.get("events", [])[:2]:
|
|
229
229
|
rows.append(([day, ev, rng.choice(["web", "docs", "(not set)"])], [rng.randint(0, 4)]))
|
|
230
|
-
(d / "funnel.json").write_text(json.dumps({**meta, **ga4_report(["date", "eventName", conv.get("sourceDimension") or "customEvent:source_app"], ["eventCount"], rows)}))
|
|
230
|
+
(d / "funnel.json").write_text(json.dumps({**meta, **ga4_report(["date", "eventName", conv.get("sourceDimension") or "customEvent:source_app"], ["eventCount"], rows)}), encoding="utf-8")
|
|
231
231
|
return paths, total
|
|
232
232
|
|
|
233
233
|
|
|
@@ -250,7 +250,7 @@ def write_timeseries(prop, slug, sites_in_prop, gsc_rows, ga4_paths):
|
|
|
250
250
|
rows.append({"keys": [day.isoformat(), p], "clicks": round(dc), "impressions": round(di),
|
|
251
251
|
"ctr": (dc / di) if di else 0, "position": round(rng.uniform(4, 20), 1)})
|
|
252
252
|
(d / f"gsc-{slug}.json").write_text(json.dumps(
|
|
253
|
-
{"site": prop, "startDate": day0.isoformat(), "endDate": END.isoformat(), "rows": rows}))
|
|
253
|
+
{"site": prop, "startDate": day0.isoformat(), "endDate": END.isoformat(), "rows": rows}), encoding="utf-8")
|
|
254
254
|
for s in sites_in_prop:
|
|
255
255
|
paths, total = ga4_paths.get(s["host"], ([], 0))
|
|
256
256
|
if not s.get("ga4Property"):
|
|
@@ -263,7 +263,7 @@ def write_timeseries(prop, slug, sites_in_prop, gsc_rows, ga4_paths):
|
|
|
263
263
|
v = weekly(day, (total / 90) * (0.35 if j == 0 else 0.08), day0=ga0)
|
|
264
264
|
if round(v):
|
|
265
265
|
rows.append({"date": day.strftime("%Y%m%d"), "page": p, "sessions": round(v)})
|
|
266
|
-
(d / f"ga4-{s['host']}.json").write_text(json.dumps({"site": s["host"], "rows": rows}))
|
|
266
|
+
(d / f"ga4-{s['host']}.json").write_text(json.dumps({"site": s["host"], "rows": rows}), encoding="utf-8")
|
|
267
267
|
|
|
268
268
|
# date x source/medium. AI assistants grow over the window and the
|
|
269
269
|
# rest hold roughly flat, so the demo actually shows the thing the
|
|
@@ -278,7 +278,7 @@ def write_timeseries(prop, slug, sites_in_prop, gsc_rows, ga4_paths):
|
|
|
278
278
|
rows.append({"date": day.strftime("%Y%m%d"), "source": src,
|
|
279
279
|
"medium": med, "sessions": round(v)})
|
|
280
280
|
(d / f"ga4-sources-{s['host']}.json").write_text(
|
|
281
|
-
json.dumps({"site": s["host"], "rows": rows}))
|
|
281
|
+
json.dumps({"site": s["host"], "rows": rows}), encoding="utf-8")
|
|
282
282
|
|
|
283
283
|
|
|
284
284
|
def write_probe(sites):
|
|
@@ -302,7 +302,7 @@ def write_probe(sites):
|
|
|
302
302
|
"soft_404": {"status": 404, "real_404": True},
|
|
303
303
|
})
|
|
304
304
|
now = datetime.now(timezone.utc)
|
|
305
|
-
(d / f"probe-{now:%Y%m%d-%H%M%S}.json").write_text(json.dumps({"probed_at": now.isoformat(), "sites": out}, indent=2))
|
|
305
|
+
(d / f"probe-{now:%Y%m%d-%H%M%S}.json").write_text(json.dumps({"probed_at": now.isoformat(), "sites": out}, indent=2), encoding="utf-8")
|
|
306
306
|
|
|
307
307
|
|
|
308
308
|
def write_metadata_audit(sites, gsc_by_host):
|
|
@@ -333,7 +333,7 @@ def write_metadata_audit(sites, gsc_by_host):
|
|
|
333
333
|
"imps": imps, "clicks": sum(q["clicks"] for q in qs), "issues": issues,
|
|
334
334
|
"top_queries": qs[:5], "missed_clicks_window": missed})
|
|
335
335
|
audit["sites"][s["host"]] = findings
|
|
336
|
-
(DATA / "metadata-audit.json").write_text(json.dumps(audit, indent=1))
|
|
336
|
+
(DATA / "metadata-audit.json").write_text(json.dumps(audit, indent=1), encoding="utf-8")
|
|
337
337
|
|
|
338
338
|
|
|
339
339
|
def write_index_status(sites, gsc_by_host):
|
|
@@ -365,7 +365,7 @@ def write_index_status(sites, gsc_by_host):
|
|
|
365
365
|
"pending": False, "errors": 0, "warnings": 0}]},
|
|
366
366
|
"problems": problems,
|
|
367
367
|
}
|
|
368
|
-
(DATA / "index-status.json").write_text(json.dumps(out, indent=2))
|
|
368
|
+
(DATA / "index-status.json").write_text(json.dumps(out, indent=2), encoding="utf-8")
|
|
369
369
|
|
|
370
370
|
|
|
371
371
|
def write_trends(props, sites, gsc_rows_by_prop):
|
|
@@ -399,7 +399,7 @@ def write_trends(props, sites, gsc_rows_by_prop):
|
|
|
399
399
|
|
|
400
400
|
monthly = {}
|
|
401
401
|
dates_file = DATA / "gsc" / slug / "dates.json"
|
|
402
|
-
for r in json.loads(dates_file.read_text())["rows"]:
|
|
402
|
+
for r in json.loads(dates_file.read_text(encoding="utf-8"))["rows"]:
|
|
403
403
|
m = r["keys"][0][:7]
|
|
404
404
|
cur = monthly.setdefault(m, {"clicks": 0, "imps": 0})
|
|
405
405
|
cur["clicks"] += r["clicks"]
|
|
@@ -417,7 +417,7 @@ def write_trends(props, sites, gsc_rows_by_prop):
|
|
|
417
417
|
total[ym] = t
|
|
418
418
|
ai[ym] = round(t * (0.01 + (12 - k) * 0.004))
|
|
419
419
|
out["ai_referrals"][s["host"]] = {"ai": ai, "total": total}
|
|
420
|
-
(DATA / f"trends-{TODAY.isoformat()}.json").write_text(json.dumps(out, indent=1))
|
|
420
|
+
(DATA / f"trends-{TODAY.isoformat()}.json").write_text(json.dumps(out, indent=1), encoding="utf-8")
|
|
421
421
|
|
|
422
422
|
|
|
423
423
|
def write_proposals(sites, gsc_by_host):
|
|
@@ -450,7 +450,7 @@ def write_proposals(sites, gsc_by_host):
|
|
|
450
450
|
"evidence": "DEMO DATA — 12 days since the change; CTR up from 1.9% to 2.6% but the 28-day window has not closed."}],
|
|
451
451
|
"inference_ran": True,
|
|
452
452
|
}
|
|
453
|
-
(DATA / "opportunity-proposals.json").write_text(json.dumps(out, indent=1))
|
|
453
|
+
(DATA / "opportunity-proposals.json").write_text(json.dumps(out, indent=1), encoding="utf-8")
|
|
454
454
|
|
|
455
455
|
|
|
456
456
|
def write_run_files():
|
|
@@ -459,8 +459,8 @@ def write_run_files():
|
|
|
459
459
|
"ts": ts.strftime("%Y-%m-%dT%H:%MZ"), "failures": "",
|
|
460
460
|
"steps": [{"name": n, "ok": True, "seconds": s} for n, s in
|
|
461
461
|
(("probe", 6.2), ("gsc", 41.0), ("ga4", 9.8), ("timeseries", 22.4), ("metadata-audit", 14.1),
|
|
462
|
-
("index-status", 38.5), ("opportunity-scan", 17.0), ("daily-diff", 0.3))]}, indent=1))
|
|
463
|
-
with (DATA / "daily-ops.log").open("a") as f:
|
|
462
|
+
("index-status", 38.5), ("opportunity-scan", 17.0), ("daily-diff", 0.3))]}, indent=1), encoding="utf-8")
|
|
463
|
+
with (DATA / "daily-ops.log").open("a", encoding="utf-8") as f:
|
|
464
464
|
f.write(f"=== daily run {ts:%Y-%m-%d %H:%M} (demo data) ===\n")
|
|
465
465
|
f.write("--- probe\n 2 sites probed\n--- gsc\n 2 properties pulled\n--- daily-diff\n daily-log updated (0 alerts)\n")
|
|
466
466
|
f.write("=== done (0 failures) ===\n")
|
package/ops/doctor.py
CHANGED
|
@@ -64,10 +64,19 @@ def check_config():
|
|
|
64
64
|
|
|
65
65
|
def check_tools():
|
|
66
66
|
print("tools")
|
|
67
|
-
|
|
68
|
-
|
|
67
|
+
report("OK" if shutil.which("curl") else "FAIL", "curl",
|
|
68
|
+
"" if shutil.which("curl") else "install it — all HTTP goes through curl "
|
|
69
|
+
"(Windows 10+ ships it in System32)")
|
|
70
|
+
# node signs the service-account JWT, so it is required even for someone
|
|
71
|
+
# who never opens the dashboard.
|
|
72
|
+
import google_auth
|
|
73
|
+
node = google_auth.node_bin()
|
|
74
|
+
report("OK" if shutil.which(node) else "FAIL", f"node ({node})",
|
|
75
|
+
"" if shutil.which(node) else "install Node 20+ — it runs the dashboard "
|
|
76
|
+
"and signs the service-account JWT; $NODE overrides the path")
|
|
69
77
|
v = sys.version_info
|
|
70
|
-
report("OK" if v >= (3, 10) else "FAIL",
|
|
78
|
+
report("OK" if v >= (3, 10) else "FAIL",
|
|
79
|
+
f"python {v.major}.{v.minor} ({Path(sys.executable).name})",
|
|
71
80
|
"" if v >= (3, 10) else "3.10+ required")
|
|
72
81
|
nm = seo_config.ROOT / "node_modules"
|
|
73
82
|
report("OK" if nm.exists() else "WARN", "node_modules", "" if nm.exists() else "run: npm install")
|
|
@@ -86,7 +95,7 @@ def check_auth(cfg):
|
|
|
86
95
|
f"{kp or '(unset)'} — set google.serviceAccountKey or $GOOGLE_APPLICATION_CREDENTIALS; see docs/SETUP-GOOGLE.md")
|
|
87
96
|
return None
|
|
88
97
|
try:
|
|
89
|
-
key = json.loads(Path(kp).read_text())
|
|
98
|
+
key = json.loads(Path(kp).read_text(encoding="utf-8"))
|
|
90
99
|
except (OSError, json.JSONDecodeError) as exc:
|
|
91
100
|
report("FAIL", "key file unreadable", str(exc))
|
|
92
101
|
return None
|
|
@@ -197,7 +206,6 @@ def check_modules(cfg):
|
|
|
197
206
|
report("OK", "enabled: " + (", ".join(on) or "(none beyond defaults)"))
|
|
198
207
|
llm = cfg["modules"].get("llm", {})
|
|
199
208
|
if llm.get("enabled"):
|
|
200
|
-
import shlex
|
|
201
209
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
202
210
|
import llm as llm_mod
|
|
203
211
|
http = llm_mod.http_config()
|
|
@@ -211,11 +219,17 @@ def check_modules(cfg):
|
|
|
211
219
|
f"check provider/model and that {llm['http'].get('apiKeyEnv') or 'apiKeyEnv'} "
|
|
212
220
|
"is set in the environment or .env")
|
|
213
221
|
for key in ("command", "fastCommand"):
|
|
214
|
-
|
|
222
|
+
raw = str(llm.get(key) or "").strip()
|
|
223
|
+
cmd = llm_mod.split_command(raw)
|
|
215
224
|
if not cmd:
|
|
225
|
+
if raw:
|
|
226
|
+
# Configured but unparseable — almost always an unbalanced
|
|
227
|
+
# quote. Say so, or the next line reads as "not configured".
|
|
228
|
+
report("FAIL", f"llm.{key} could not be parsed: {raw[:60]}",
|
|
229
|
+
"check for an unbalanced quote")
|
|
216
230
|
# Only complain about a missing command when there is no http
|
|
217
231
|
# block at all; a broken one already reported itself.
|
|
218
|
-
|
|
232
|
+
elif key == "command" and not http and not llm.get("http"):
|
|
219
233
|
report("FAIL", "llm has neither http nor command configured")
|
|
220
234
|
continue
|
|
221
235
|
report("OK" if shutil.which(cmd[0]) else "FAIL", f"llm.{key}: {cmd[0]}",
|
package/ops/export_static.py
CHANGED
|
@@ -103,11 +103,11 @@ def main():
|
|
|
103
103
|
html = html.replace("</body>", staleness + "</body>", 1)
|
|
104
104
|
dest = out / "index.html" if route == "/" else out / route.lstrip("/") / "index.html"
|
|
105
105
|
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
106
|
-
dest.write_text(html)
|
|
106
|
+
dest.write_text(html, encoding="utf-8")
|
|
107
107
|
|
|
108
|
-
(out / "styles.css").write_text(fetch("/styles.css"))
|
|
109
|
-
(out / "favicon.svg").write_text(fetch("/favicon.svg"))
|
|
110
|
-
(out / "robots.txt").write_text("User-agent: *\nDisallow: /\n")
|
|
108
|
+
(out / "styles.css").write_text(fetch("/styles.css"), encoding="utf-8")
|
|
109
|
+
(out / "favicon.svg").write_text(fetch("/favicon.svg"), encoding="utf-8")
|
|
110
|
+
(out / "robots.txt").write_text("User-agent: *\nDisallow: /\n", encoding="utf-8")
|
|
111
111
|
except BaseException:
|
|
112
112
|
# Never leave a half-built staging directory behind; the next
|
|
113
113
|
# run would otherwise start from someone else's leftovers.
|
package/ops/hn_digest.py
CHANGED
|
@@ -157,7 +157,7 @@ def main():
|
|
|
157
157
|
seo_config.DATA.mkdir(parents=True, exist_ok=True)
|
|
158
158
|
(seo_config.DATA / "hn-digest.json").write_text(json.dumps(
|
|
159
159
|
{"generated": datetime.now(timezone.utc).isoformat(timespec="minutes"),
|
|
160
|
-
"stats": stats, "picks": picks}, indent=1))
|
|
160
|
+
"stats": stats, "picks": picks}, indent=1), encoding="utf-8")
|
|
161
161
|
for p in picks:
|
|
162
162
|
print(f"- {p['title']} ({p['comments']}c/{p['points']}p) {p['url']} [{p['why']}]")
|
|
163
163
|
if not picks:
|
package/ops/indexnow.py
CHANGED
|
@@ -33,12 +33,12 @@ def key_file() -> Path:
|
|
|
33
33
|
def cmd_init() -> int:
|
|
34
34
|
kf = key_file()
|
|
35
35
|
if kf.exists():
|
|
36
|
-
key = kf.read_text().strip()
|
|
36
|
+
key = kf.read_text(encoding="utf-8").strip()
|
|
37
37
|
print(f"key file already exists: {kf}")
|
|
38
38
|
else:
|
|
39
39
|
key = secrets.token_hex(16)
|
|
40
40
|
kf.parent.mkdir(parents=True, exist_ok=True)
|
|
41
|
-
kf.write_text(key + "\n")
|
|
41
|
+
kf.write_text(key + "\n", encoding="utf-8")
|
|
42
42
|
print(f"wrote {kf}")
|
|
43
43
|
print(f"\nkey: {key}\n\nServe this key as plain text at, for every site:")
|
|
44
44
|
for s in seo_config.sites():
|
|
@@ -52,7 +52,7 @@ def cmd_ping(urls: list[str]) -> int:
|
|
|
52
52
|
if not kf.exists():
|
|
53
53
|
print(f"no key file at {kf} — run: python3 ops/indexnow.py init")
|
|
54
54
|
return 1
|
|
55
|
-
key = kf.read_text().strip()
|
|
55
|
+
key = kf.read_text(encoding="utf-8").strip()
|
|
56
56
|
by_host: dict[str, list[str]] = {}
|
|
57
57
|
for u in urls:
|
|
58
58
|
host = urlsplit(u).netloc
|
package/ops/llm.py
CHANGED
|
@@ -28,6 +28,7 @@ path runs. Nothing that comes back is applied automatically: callers turn
|
|
|
28
28
|
replies into briefings or proposals a human accepts or ignores.
|
|
29
29
|
"""
|
|
30
30
|
import json
|
|
31
|
+
import os
|
|
31
32
|
import shlex
|
|
32
33
|
import shutil
|
|
33
34
|
import subprocess
|
|
@@ -42,6 +43,27 @@ ANTHROPIC_VERSION = "2023-06-01"
|
|
|
42
43
|
MAX_TOKENS = 4096
|
|
43
44
|
|
|
44
45
|
|
|
46
|
+
def split_command(cmd: str) -> list[str]:
|
|
47
|
+
"""Split a configured command string into argv, per platform.
|
|
48
|
+
|
|
49
|
+
POSIX rules treat a backslash as an escape, so on Windows they quietly
|
|
50
|
+
turn `C:\\tools\\claude.exe` into `C:toolsclaude.exe`. There, split with
|
|
51
|
+
the non-POSIX rules cmd.exe uses and strip the quotes shlex leaves on.
|
|
52
|
+
|
|
53
|
+
Returns [] for anything unparseable. This string comes from the Settings
|
|
54
|
+
form, so a stray quote is a typo a user can make, and it must not take
|
|
55
|
+
the daily run down with a ValueError from inside shlex — callers already
|
|
56
|
+
treat [] as "no command configured". doctor tells them which it was.
|
|
57
|
+
"""
|
|
58
|
+
try:
|
|
59
|
+
if os.name != "nt":
|
|
60
|
+
return shlex.split(cmd)
|
|
61
|
+
parts = shlex.split(cmd, posix=False)
|
|
62
|
+
except ValueError:
|
|
63
|
+
return []
|
|
64
|
+
return [p[1:-1] if len(p) > 1 and p[0] == p[-1] == '"' else p for p in parts]
|
|
65
|
+
|
|
66
|
+
|
|
45
67
|
def _enabled() -> dict | None:
|
|
46
68
|
m = seo_config.module("llm")
|
|
47
69
|
return m if m.get("enabled") else None
|
|
@@ -71,7 +93,7 @@ def command(fast: bool = False) -> list[str] | None:
|
|
|
71
93
|
if not m:
|
|
72
94
|
return None
|
|
73
95
|
cmd = (m.get("fastCommand") if fast else None) or m.get("command") or ""
|
|
74
|
-
parts =
|
|
96
|
+
parts = split_command(str(cmd))
|
|
75
97
|
if not parts or not shutil.which(parts[0]):
|
|
76
98
|
return None
|
|
77
99
|
return parts
|
package/ops/opportunity_scan.py
CHANGED
|
@@ -36,7 +36,7 @@ RISE_MIN_RATIO = 2.5 # and have grown at least this much vs prior 84d
|
|
|
36
36
|
|
|
37
37
|
def latest_trends():
|
|
38
38
|
files = sorted(DATA.glob("trends-*.json"))
|
|
39
|
-
return json.loads(files[-1].read_text()) if files else None
|
|
39
|
+
return json.loads(files[-1].read_text(encoding="utf-8")) if files else None
|
|
40
40
|
|
|
41
41
|
|
|
42
42
|
def queue():
|
|
@@ -51,7 +51,7 @@ def queue():
|
|
|
51
51
|
except json.JSONDecodeError:
|
|
52
52
|
pass
|
|
53
53
|
try:
|
|
54
|
-
return json.loads((seo_config.INSTANCE / "config" / "backlog.json").read_text()).get("actions", [])
|
|
54
|
+
return json.loads((seo_config.INSTANCE / "config" / "backlog.json").read_text(encoding="utf-8")).get("actions", [])
|
|
55
55
|
except (OSError, json.JSONDecodeError):
|
|
56
56
|
return []
|
|
57
57
|
|
|
@@ -84,7 +84,7 @@ def _host_resolver(prop):
|
|
|
84
84
|
best = {}
|
|
85
85
|
qp = seo_config.DATA / "gsc" / seo_config.gsc_data_slug(prop) / "query_page_90d.json"
|
|
86
86
|
try:
|
|
87
|
-
for r in json.loads(qp.read_text()).get("rows", []):
|
|
87
|
+
for r in json.loads(qp.read_text(encoding="utf-8")).get("rows", []):
|
|
88
88
|
q, page = r["keys"][0], r["keys"][1]
|
|
89
89
|
h = page.split("/")[2] if page.count("/") >= 2 else ""
|
|
90
90
|
if h in by_gsc_host and r["impressions"] > best.get(q, (0, ""))[0]:
|
|
@@ -175,7 +175,7 @@ def main():
|
|
|
175
175
|
print("llm module off — candidates recorded without proposals")
|
|
176
176
|
|
|
177
177
|
DATA.mkdir(parents=True, exist_ok=True)
|
|
178
|
-
(DATA / "opportunity-proposals.json").write_text(json.dumps(out, indent=1))
|
|
178
|
+
(DATA / "opportunity-proposals.json").write_text(json.dumps(out, indent=1), encoding="utf-8")
|
|
179
179
|
print(f"candidates: {len(candidates)} | proposals: {len(out['proposals'])} | "
|
|
180
180
|
f"verdicts: {len(out['verdicts'])} | inference: {out['inference_ran']}")
|
|
181
181
|
return 0
|
package/ops/py.mjs
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/* Run a Python script with whichever interpreter this machine actually has.
|
|
3
|
+
*
|
|
4
|
+
* The npm scripts used to hardcode `python3`, which does not exist on
|
|
5
|
+
* Windows: Python installs there as `python`, or only behind the `py`
|
|
6
|
+
* launcher. Worse, Windows ships an App Execution Alias at `python3.exe`
|
|
7
|
+
* that opens the Microsoft Store instead of running anything, so a hardcoded
|
|
8
|
+
* `python3` fails in a way that looks like Python is missing when it is not.
|
|
9
|
+
* Probing for one that answers `--version` is the only reliable answer.
|
|
10
|
+
*
|
|
11
|
+
* node ops/py.mjs ops/doctor.py --offline
|
|
12
|
+
*
|
|
13
|
+
* $PYTHON overrides, which is also how the n-seo CLI behaves.
|
|
14
|
+
*/
|
|
15
|
+
import { spawn, spawnSync } from "node:child_process";
|
|
16
|
+
import { pathToFileURL } from "node:url";
|
|
17
|
+
|
|
18
|
+
const CANDIDATES = process.platform === "win32"
|
|
19
|
+
? ["python", "python3", "py"]
|
|
20
|
+
: ["python3", "python"];
|
|
21
|
+
|
|
22
|
+
export function pythonBin() {
|
|
23
|
+
if (process.env.PYTHON) return process.env.PYTHON;
|
|
24
|
+
for (const c of CANDIDATES) {
|
|
25
|
+
const probe = spawnSync(c, ["--version"], { stdio: "ignore" });
|
|
26
|
+
if (!probe.error && probe.status === 0) return c;
|
|
27
|
+
}
|
|
28
|
+
return CANDIDATES[0]; // let the real command fail with a useful message
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// Only act as a launcher when invoked directly, not when imported.
|
|
32
|
+
if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) {
|
|
33
|
+
const child = spawn(pythonBin(), process.argv.slice(2), { stdio: "inherit" });
|
|
34
|
+
child.on("error", (err) => {
|
|
35
|
+
console.error(`could not start ${pythonBin()}: ${err.message}`);
|
|
36
|
+
console.error("install Python 3.10+, or set $PYTHON to its path");
|
|
37
|
+
process.exit(127);
|
|
38
|
+
});
|
|
39
|
+
child.on("exit", (code, signal) => process.exit(signal ? 1 : code ?? 1));
|
|
40
|
+
}
|
package/ops/reddit_digest.py
CHANGED
|
@@ -151,7 +151,7 @@ def main():
|
|
|
151
151
|
seo_config.DATA.mkdir(parents=True, exist_ok=True)
|
|
152
152
|
(seo_config.DATA / "reddit-digest.json").write_text(json.dumps(
|
|
153
153
|
{"generated": datetime.now(timezone.utc).isoformat(timespec="minutes"),
|
|
154
|
-
"user": user or None, "auth": bool(oauth_token()), "picks": picks}, indent=1))
|
|
154
|
+
"user": user or None, "auth": bool(oauth_token()), "picks": picks}, indent=1), encoding="utf-8")
|
|
155
155
|
for p in picks:
|
|
156
156
|
print(f"- r/{p['sub']}: {p['title']} ({p['comments']}c/{p['score']}pts, {p['age_days']}d) {p['url']}")
|
|
157
157
|
if not picks:
|