n-seo 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +132 -46
- package/bin/n-seo.mjs +14 -4
- package/docs/FAQ.md +14 -3
- package/docs/MCP.md +25 -9
- package/docs/PRD.md +6 -4
- package/docs/RELEASING.md +21 -0
- package/docs/SCHEDULING.md +42 -5
- package/docs/SETUP-GOOGLE.md +1 -1
- package/ingest/__pycache__/analyze_metadata.cpython-312.pyc +0 -0
- package/ingest/__pycache__/google_auth.cpython-312.pyc +0 -0
- package/ingest/__pycache__/http_util.cpython-312.pyc +0 -0
- package/ingest/__pycache__/seo_config.cpython-312.pyc +0 -0
- package/ingest/analyze_ga4.py +2 -2
- package/ingest/analyze_gsc.py +2 -2
- package/ingest/analyze_metadata.py +2 -2
- package/ingest/analyze_trends.py +2 -2
- package/ingest/google_auth.py +53 -20
- package/ingest/pull_ga4.py +1 -1
- package/ingest/pull_gsc.py +1 -1
- package/ingest/pull_index_status.py +1 -1
- package/ingest/pull_timeseries.py +3 -3
- package/ingest/seo_config.py +19 -3
- package/ops/__pycache__/daily.cpython-312.pyc +0 -0
- package/ops/__pycache__/daily_diff.cpython-312.pyc +0 -0
- package/ops/__pycache__/demo_data.cpython-312.pyc +0 -0
- package/ops/__pycache__/export_static.cpython-312.pyc +0 -0
- package/ops/__pycache__/llm.cpython-312.pyc +0 -0
- package/ops/__pycache__/publish.cpython-312.pyc +0 -0
- package/ops/daily.py +2 -2
- package/ops/daily_diff.py +9 -8
- package/ops/demo_data.py +17 -17
- package/ops/doctor.py +21 -7
- package/ops/export_static.py +4 -4
- package/ops/hn_digest.py +1 -1
- package/ops/indexnow.py +3 -3
- package/ops/llm.py +23 -1
- package/ops/opportunity_scan.py +4 -4
- package/ops/py.mjs +40 -0
- package/ops/reddit_digest.py +1 -1
- package/ops/templates/n-seo-daily-task.xml +65 -0
- package/package.json +16 -9
- package/probes/__pycache__/site_probe.cpython-312.pyc +0 -0
- package/probes/site_probe.py +1 -1
- package/src/actions.ts +1 -1
- package/src/config.ts +1 -1
- package/src/mcp.ts +2 -2
- package/src/settings.tsx +3 -3
- package/src/views.tsx +9 -9
- package/ingest/__pycache__/analyze_ga4.cpython-313.pyc +0 -0
- package/ingest/__pycache__/analyze_gsc.cpython-313.pyc +0 -0
- package/ingest/__pycache__/analyze_metadata.cpython-313.pyc +0 -0
- package/ingest/__pycache__/analyze_trends.cpython-313.pyc +0 -0
- package/ingest/__pycache__/google_auth.cpython-313.pyc +0 -0
- package/ingest/__pycache__/http_util.cpython-313.pyc +0 -0
- package/ingest/__pycache__/pull_ga4.cpython-313.pyc +0 -0
- package/ingest/__pycache__/pull_gsc.cpython-313.pyc +0 -0
- package/ingest/__pycache__/pull_index_status.cpython-313.pyc +0 -0
- package/ingest/__pycache__/pull_timeseries.cpython-313.pyc +0 -0
- package/ingest/__pycache__/seo_config.cpython-313.pyc +0 -0
- package/ops/__pycache__/daily.cpython-313.pyc +0 -0
- package/ops/__pycache__/daily_diff.cpython-313.pyc +0 -0
- package/ops/__pycache__/demo_data.cpython-313.pyc +0 -0
- package/ops/__pycache__/doctor.cpython-313.pyc +0 -0
- package/ops/__pycache__/export_static.cpython-313.pyc +0 -0
- package/ops/__pycache__/hn_digest.cpython-313.pyc +0 -0
- package/ops/__pycache__/indexnow.cpython-313.pyc +0 -0
- package/ops/__pycache__/llm.cpython-313.pyc +0 -0
- package/ops/__pycache__/opportunity_scan.cpython-313.pyc +0 -0
- package/ops/__pycache__/publish.cpython-313.pyc +0 -0
- package/ops/__pycache__/reddit_digest.cpython-313.pyc +0 -0
- package/probes/__pycache__/site_probe.cpython-313.pyc +0 -0
package/ingest/google_auth.py
CHANGED
|
@@ -3,10 +3,10 @@
|
|
|
3
3
|
service-account-key (default, recommended)
|
|
4
4
|
A service-account JSON key on disk (config google.serviceAccountKey,
|
|
5
5
|
or $GOOGLE_APPLICATION_CREDENTIALS). We build the OAuth JWT ourselves
|
|
6
|
-
and sign it with
|
|
7
|
-
dependency. Add the service account's email to
|
|
8
|
-
property (Full user) and your GA4 property
|
|
9
|
-
access it gets.
|
|
6
|
+
and sign it with node's crypto module, so there is no gcloud, no pip
|
|
7
|
+
dependency and no openssl binary. Add the service account's email to
|
|
8
|
+
your Search Console property (Full user) and your GA4 property
|
|
9
|
+
(Viewer) — that is all the access it gets.
|
|
10
10
|
|
|
11
11
|
gcloud-impersonate
|
|
12
12
|
`gcloud auth print-access-token --impersonate-service-account=<sa>`.
|
|
@@ -33,9 +33,10 @@ inspection, analytics.readonly for GA4.
|
|
|
33
33
|
import base64
|
|
34
34
|
import json
|
|
35
35
|
import os
|
|
36
|
+
import shutil
|
|
36
37
|
import subprocess
|
|
37
|
-
import tempfile
|
|
38
38
|
import time
|
|
39
|
+
from pathlib import Path
|
|
39
40
|
|
|
40
41
|
import seo_config
|
|
41
42
|
from http_util import curl_json
|
|
@@ -58,6 +59,50 @@ def _b64url(b: bytes) -> str:
|
|
|
58
59
|
return base64.urlsafe_b64encode(b).rstrip(b"=").decode()
|
|
59
60
|
|
|
60
61
|
|
|
62
|
+
# RS256 without an openssl binary. Node is already required (the dashboard
|
|
63
|
+
# runs on it) and its crypto module produces byte-identical signatures, so
|
|
64
|
+
# signing through node drops an external dependency on every platform — and
|
|
65
|
+
# makes Windows work, where openssl is not installed by default.
|
|
66
|
+
#
|
|
67
|
+
# Key and payload go in over stdin, never on the command line: argv is
|
|
68
|
+
# readable by any other process on the machine.
|
|
69
|
+
_NODE_SIGN_JS = (
|
|
70
|
+
"let s='';"
|
|
71
|
+
"process.stdin.setEncoding('utf8')"
|
|
72
|
+
".on('data', d => s += d)"
|
|
73
|
+
".on('end', () => {"
|
|
74
|
+
" const { key, data } = JSON.parse(s);"
|
|
75
|
+
" const sig = require('node:crypto')"
|
|
76
|
+
" .sign('sha256', Buffer.from(data, 'base64'), key);"
|
|
77
|
+
" process.stdout.write(sig.toString('base64'));"
|
|
78
|
+
"});"
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def node_bin() -> str:
|
|
83
|
+
"""The node executable. $NODE overrides, mirroring $PYTHON for the CLI."""
|
|
84
|
+
return os.environ.get("NODE") or "node"
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _sign_rs256(private_key_pem: str, data: bytes) -> bytes:
|
|
88
|
+
"""RSASSA-PKCS1-v1_5 over SHA-256, the `RS256` a Google JWT needs."""
|
|
89
|
+
node = node_bin()
|
|
90
|
+
if not shutil.which(node):
|
|
91
|
+
raise RuntimeError(
|
|
92
|
+
f"{node} is not on PATH, and the service-account JWT is signed with it. "
|
|
93
|
+
"Install Node 20+ (it is required for the dashboard anyway), or set "
|
|
94
|
+
"$NODE to its path.")
|
|
95
|
+
payload = json.dumps({"key": private_key_pem,
|
|
96
|
+
"data": base64.b64encode(data).decode()})
|
|
97
|
+
p = subprocess.run([node, "-e", _NODE_SIGN_JS],
|
|
98
|
+
input=payload, capture_output=True, text=True)
|
|
99
|
+
if p.returncode != 0 or not p.stdout.strip():
|
|
100
|
+
raise RuntimeError(
|
|
101
|
+
"could not sign the service-account JWT — the key file's "
|
|
102
|
+
f"private_key may be malformed: {(p.stderr or '').strip()[:200]}")
|
|
103
|
+
return base64.b64decode(p.stdout.strip())
|
|
104
|
+
|
|
105
|
+
|
|
61
106
|
def key_path() -> str | None:
|
|
62
107
|
"""The service-account key to sign with.
|
|
63
108
|
|
|
@@ -82,7 +127,7 @@ def _sa_key_token(scope: str) -> str:
|
|
|
82
127
|
raise RuntimeError(
|
|
83
128
|
"google.auth is service-account-key but no key file was found at "
|
|
84
129
|
f"{kp or '(unset)'} — see docs/SETUP-GOOGLE.md")
|
|
85
|
-
key = json.loads(
|
|
130
|
+
key = json.loads(Path(kp).read_text(encoding="utf-8"))
|
|
86
131
|
# Backdate slightly: Google rejects a JWT issued in its future, so a
|
|
87
132
|
# machine whose clock runs a few seconds fast otherwise fails to
|
|
88
133
|
# authenticate at all, with an error that names nothing useful.
|
|
@@ -93,19 +138,7 @@ def _sa_key_token(scope: str) -> str:
|
|
|
93
138
|
"iat": iat, "exp": iat + 3600,
|
|
94
139
|
}).encode())
|
|
95
140
|
signing_input = f"{header}.{claims}".encode()
|
|
96
|
-
|
|
97
|
-
fd, tmp = tempfile.mkstemp(prefix="n-seo-", suffix=".pem")
|
|
98
|
-
try:
|
|
99
|
-
os.fchmod(fd, 0o600)
|
|
100
|
-
with os.fdopen(fd, "w") as f:
|
|
101
|
-
f.write(key["private_key"])
|
|
102
|
-
sig = subprocess.run(["openssl", "dgst", "-sha256", "-sign", tmp],
|
|
103
|
-
input=signing_input, capture_output=True, check=True).stdout
|
|
104
|
-
finally:
|
|
105
|
-
try:
|
|
106
|
-
os.unlink(tmp)
|
|
107
|
-
except OSError:
|
|
108
|
-
pass
|
|
141
|
+
sig = _sign_rs256(key["private_key"], signing_input)
|
|
109
142
|
assertion = signing_input.decode() + "." + _b64url(sig)
|
|
110
143
|
resp = curl_json([
|
|
111
144
|
"-X", "POST", "-d", f"grant_type=urn:ietf:params:oauth:grant-type:jwt-bearer&assertion={assertion}",
|
|
@@ -222,7 +255,7 @@ def service_account_email() -> str | None:
|
|
|
222
255
|
kp = key_path()
|
|
223
256
|
if kp and os.path.exists(kp):
|
|
224
257
|
try:
|
|
225
|
-
return json.loads(
|
|
258
|
+
return json.loads(Path(kp).read_text(encoding="utf-8")).get("client_email")
|
|
226
259
|
except (OSError, json.JSONDecodeError):
|
|
227
260
|
return None
|
|
228
261
|
return None
|
package/ingest/pull_ga4.py
CHANGED
|
@@ -96,7 +96,7 @@ def main():
|
|
|
96
96
|
continue
|
|
97
97
|
(site_dir / f"{name}.json").write_text(json.dumps(
|
|
98
98
|
{"site": host, "property": f"properties/{prop}",
|
|
99
|
-
"pulled": date.today().isoformat(), **resp}))
|
|
99
|
+
"pulled": date.today().isoformat(), **resp}), encoding="utf-8")
|
|
100
100
|
print(f"{host:28s} {name:8s} {resp.get('rowCount', 0)} rows")
|
|
101
101
|
|
|
102
102
|
print(f"\nSaved under {out_root}")
|
package/ingest/pull_gsc.py
CHANGED
|
@@ -100,7 +100,7 @@ def main():
|
|
|
100
100
|
"rowCount": len(result["rows"]),
|
|
101
101
|
"rows": result["rows"],
|
|
102
102
|
}
|
|
103
|
-
(site_dir / f"{name}{suffix}.json").write_text(json.dumps(payload))
|
|
103
|
+
(site_dir / f"{name}{suffix}.json").write_text(json.dumps(payload), encoding="utf-8")
|
|
104
104
|
print(f"{slug:28s} {name}{suffix:5s} {len(result['rows'])} rows")
|
|
105
105
|
|
|
106
106
|
print(f"\nWindows: {start_full} and {start_recent} -> {end}\nSaved under {out_root}")
|
|
@@ -170,7 +170,7 @@ def main() -> int:
|
|
|
170
170
|
f"({len(never_crawled)} never crawled)")
|
|
171
171
|
|
|
172
172
|
OUT.parent.mkdir(parents=True, exist_ok=True)
|
|
173
|
-
OUT.write_text(json.dumps(out, indent=2))
|
|
173
|
+
OUT.write_text(json.dumps(out, indent=2), encoding="utf-8")
|
|
174
174
|
print(f"saved {OUT} — {total_problems} URLs needing attention")
|
|
175
175
|
return 0
|
|
176
176
|
|
|
@@ -87,7 +87,7 @@ def main():
|
|
|
87
87
|
if failed:
|
|
88
88
|
continue
|
|
89
89
|
(OUT / f"gsc-{slug}.json").write_text(json.dumps(
|
|
90
|
-
{"site": site, "startDate": start, "endDate": end, "rows": rows}))
|
|
90
|
+
{"site": site, "startDate": start, "endDate": end, "rows": rows}), encoding="utf-8")
|
|
91
91
|
print(f"gsc {slug:28s} {len(rows)} date x page rows")
|
|
92
92
|
|
|
93
93
|
ga4_props = seo_config.ga4_properties()
|
|
@@ -104,7 +104,7 @@ def main():
|
|
|
104
104
|
"page": row["dimensionValues"][1]["value"],
|
|
105
105
|
"sessions": float(row["metricValues"][0]["value"])}
|
|
106
106
|
for row in batch]
|
|
107
|
-
(OUT / f"ga4-{host}.json").write_text(json.dumps({"site": host, "rows": rows}))
|
|
107
|
+
(OUT / f"ga4-{host}.json").write_text(json.dumps({"site": host, "rows": rows}), encoding="utf-8")
|
|
108
108
|
print(f"ga4 {host:28s} {len(rows)} date x page rows")
|
|
109
109
|
|
|
110
110
|
batch, ok = ga4_series(tok, prop, ["date", "sessionSource", "sessionMedium"],
|
|
@@ -118,7 +118,7 @@ def main():
|
|
|
118
118
|
"sessions": float(row["metricValues"][0]["value"])}
|
|
119
119
|
for row in batch]
|
|
120
120
|
(OUT / f"ga4-sources-{host}.json").write_text(
|
|
121
|
-
json.dumps({"site": host, "rows": rows}))
|
|
121
|
+
json.dumps({"site": host, "rows": rows}), encoding="utf-8")
|
|
122
122
|
print(f"ga4 {host:28s} {len(rows)} date x source rows")
|
|
123
123
|
|
|
124
124
|
if not gsc_props and not ga4_props:
|
package/ingest/seo_config.py
CHANGED
|
@@ -9,8 +9,24 @@ and export. Stdlib only.
|
|
|
9
9
|
import json
|
|
10
10
|
import os
|
|
11
11
|
import subprocess
|
|
12
|
+
import sys
|
|
12
13
|
from pathlib import Path
|
|
13
14
|
|
|
15
|
+
# Print UTF-8 whatever the platform thinks the console encoding is.
|
|
16
|
+
#
|
|
17
|
+
# Windows picks the ANSI code page (cp1252 on a US install) for a piped
|
|
18
|
+
# stdout, so the em dashes and middle dots this tool prints come out as
|
|
19
|
+
# mojibake or raise UnicodeEncodeError mid-run. Every entry point imports
|
|
20
|
+
# this module, so fixing it once here covers all of them. File I/O is
|
|
21
|
+
# handled separately: every read_text/write_text/open call passes
|
|
22
|
+
# encoding="utf-8" explicitly, and tests/py/test_portability.py enforces it.
|
|
23
|
+
for _stream in (sys.stdout, sys.stderr):
|
|
24
|
+
try:
|
|
25
|
+
if getattr(_stream, "encoding", "").lower().replace("-", "") != "utf8":
|
|
26
|
+
_stream.reconfigure(encoding="utf-8")
|
|
27
|
+
except (AttributeError, ValueError, OSError):
|
|
28
|
+
pass # a captured or replaced stream; nothing to fix
|
|
29
|
+
|
|
14
30
|
# The engine checkout: code, engine docs, public assets.
|
|
15
31
|
ROOT = Path(__file__).resolve().parent.parent
|
|
16
32
|
# The instance: one user's config, queue, content and data. Defaults to the
|
|
@@ -50,7 +66,7 @@ def load(force: bool = False) -> dict:
|
|
|
50
66
|
if _cache is not None and not force:
|
|
51
67
|
return _cache
|
|
52
68
|
path = CONFIG_PATH if CONFIG_PATH.exists() else EXAMPLE_PATH
|
|
53
|
-
raw = json.loads(path.read_text())
|
|
69
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
54
70
|
modules = {k: {"enabled": False, **MODULE_DEFAULTS.get(k, {})} for k in MODULE_KEYS}
|
|
55
71
|
for k, v in (raw.get("modules") or {}).items():
|
|
56
72
|
modules[k] = {"enabled": False, **MODULE_DEFAULTS.get(k, {}), **(v or {})}
|
|
@@ -96,7 +112,7 @@ def load(force: bool = False) -> dict:
|
|
|
96
112
|
def engine_info() -> dict:
|
|
97
113
|
"""Mirrors src/config.ts engineInfo(): what engine is running, for which instance."""
|
|
98
114
|
try:
|
|
99
|
-
version = json.loads((ROOT / "package.json").read_text()).get("version", "0.0.0")
|
|
115
|
+
version = json.loads((ROOT / "package.json").read_text(encoding="utf-8")).get("version", "0.0.0")
|
|
100
116
|
except (OSError, json.JSONDecodeError):
|
|
101
117
|
version = "0.0.0"
|
|
102
118
|
try:
|
|
@@ -204,7 +220,7 @@ def env(key: str, default: str = "") -> str:
|
|
|
204
220
|
if os.environ.get(key):
|
|
205
221
|
return os.environ[key]
|
|
206
222
|
try:
|
|
207
|
-
for line in (INSTANCE / ".env").read_text().splitlines():
|
|
223
|
+
for line in (INSTANCE / ".env").read_text(encoding="utf-8").splitlines():
|
|
208
224
|
line = line.strip()
|
|
209
225
|
if line.startswith(f"{key}="):
|
|
210
226
|
return line.split("=", 1)[1].strip().strip('"').strip("'")
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
package/ops/daily.py
CHANGED
|
@@ -57,7 +57,7 @@ STEPS = [
|
|
|
57
57
|
def log(line: str):
|
|
58
58
|
print(line, flush=True)
|
|
59
59
|
LOG.parent.mkdir(parents=True, exist_ok=True)
|
|
60
|
-
with LOG.open("a") as f:
|
|
60
|
+
with LOG.open("a", encoding="utf-8") as f:
|
|
61
61
|
f.write(line.rstrip("\n") + "\n")
|
|
62
62
|
|
|
63
63
|
|
|
@@ -230,7 +230,7 @@ def main():
|
|
|
230
230
|
"ts": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%MZ"),
|
|
231
231
|
"failures": "; ".join(failures),
|
|
232
232
|
"steps": results,
|
|
233
|
-
}, indent=1))
|
|
233
|
+
}, indent=1), encoding="utf-8")
|
|
234
234
|
|
|
235
235
|
write_last_run()
|
|
236
236
|
if run_hooks_flag and hooks["afterRun"]:
|
package/ops/daily_diff.py
CHANGED
|
@@ -31,7 +31,7 @@ PROBE_KEYS = [
|
|
|
31
31
|
|
|
32
32
|
def probes():
|
|
33
33
|
files = sorted((DATA / "probes").glob("probe-*.json"))
|
|
34
|
-
return [json.loads(f.read_text()) for f in files[-2:]]
|
|
34
|
+
return [json.loads(f.read_text(encoding="utf-8")) for f in files[-2:]]
|
|
35
35
|
|
|
36
36
|
|
|
37
37
|
def watch_groups():
|
|
@@ -76,7 +76,7 @@ def main():
|
|
|
76
76
|
if not f.exists():
|
|
77
77
|
continue
|
|
78
78
|
win = "90d" if f90.exists() else "16mo"
|
|
79
|
-
by_url = {r["keys"][0]: r for r in json.loads(f.read_text())["rows"]}
|
|
79
|
+
by_url = {r["keys"][0]: r for r in json.loads(f.read_text(encoding="utf-8"))["rows"]}
|
|
80
80
|
for url in urls:
|
|
81
81
|
r = by_url.get(url) or by_url.get(url.rstrip("/")) or by_url.get(url + "/")
|
|
82
82
|
label = url.split("//", 1)[-1]
|
|
@@ -90,7 +90,7 @@ def main():
|
|
|
90
90
|
conv = seo_config.load()["conversions"]
|
|
91
91
|
if conv:
|
|
92
92
|
try:
|
|
93
|
-
fr = json.loads((DATA / "ga4" / conv["site"] / "funnel.json").read_text())
|
|
93
|
+
fr = json.loads((DATA / "ga4" / conv["site"] / "funnel.json").read_text(encoding="utf-8"))
|
|
94
94
|
rows = fr.get("rows", [])
|
|
95
95
|
if rows:
|
|
96
96
|
counts = {}
|
|
@@ -113,7 +113,7 @@ def main():
|
|
|
113
113
|
# subdomain's own traffic as a referral from the apex
|
|
114
114
|
others = {h.removeprefix("www.") for h in hosts if h != host}
|
|
115
115
|
xref = []
|
|
116
|
-
for r in json.loads(p.read_text()).get("rows", []):
|
|
116
|
+
for r in json.loads(p.read_text(encoding="utf-8")).get("rows", []):
|
|
117
117
|
name = r["dimensionValues"][0]["value"]
|
|
118
118
|
if name.lower().removeprefix("www.") in others:
|
|
119
119
|
xref.append(f"{name}={float(r['metricValues'][0]['value']):.0f}")
|
|
@@ -123,7 +123,8 @@ def main():
|
|
|
123
123
|
log = INSTANCE / "docs" / "daily-log.md"
|
|
124
124
|
log.parent.mkdir(parents=True, exist_ok=True)
|
|
125
125
|
if not log.exists():
|
|
126
|
-
log.write_text("# Daily ops log\n\nAppended by ops/daily.py — newest entries last.\n"
|
|
126
|
+
log.write_text("# Daily ops log\n\nAppended by ops/daily.py — newest entries last.\n",
|
|
127
|
+
encoding="utf-8")
|
|
127
128
|
today = date.today().isoformat()
|
|
128
129
|
entry = [f"\n## {today}\n"]
|
|
129
130
|
entry += [f"- **ALERT:** {a}" for a in alerts]
|
|
@@ -132,15 +133,15 @@ def main():
|
|
|
132
133
|
|
|
133
134
|
# Re-running on the same day must replace today's entry, not stack a
|
|
134
135
|
# second one under the same heading.
|
|
135
|
-
text = log.read_text()
|
|
136
|
+
text = log.read_text(encoding="utf-8")
|
|
136
137
|
head = f"\n## {today}\n"
|
|
137
138
|
start = text.find(head)
|
|
138
139
|
if start == -1:
|
|
139
|
-
log.write_text(text.rstrip("\n") + "\n" + body)
|
|
140
|
+
log.write_text(text.rstrip("\n") + "\n" + body, encoding="utf-8")
|
|
140
141
|
else:
|
|
141
142
|
nxt = text.find("\n## ", start + len(head))
|
|
142
143
|
tail = text[nxt:] if nxt != -1 else ""
|
|
143
|
-
log.write_text(text[:start].rstrip("\n") + "\n" + body + tail.lstrip("\n"))
|
|
144
|
+
log.write_text(text[:start].rstrip("\n") + "\n" + body + tail.lstrip("\n"), encoding="utf-8")
|
|
144
145
|
for a in alerts:
|
|
145
146
|
print(f"ALERT: {a}")
|
|
146
147
|
print(f"daily-log updated ({len(alerts)} alerts, {len(lines)} lines)")
|
package/ops/demo_data.py
CHANGED
|
@@ -151,7 +151,7 @@ def write_gsc(prop, slug, sites_in_prop, index_of):
|
|
|
151
151
|
("queries", ["query"], agg(by_q)),
|
|
152
152
|
("pages", ["page"], agg(by_p))):
|
|
153
153
|
(d / f"{name}{suffix}.json").write_text(json.dumps(
|
|
154
|
-
{**meta, "dimensions": dims, "rowCount": len(rows_), "rows": rows_}))
|
|
154
|
+
{**meta, "dimensions": dims, "rowCount": len(rows_), "rows": rows_}), encoding="utf-8")
|
|
155
155
|
|
|
156
156
|
dataset(full_rows, 1.0, "", END - timedelta(days=FULL_DAYS))
|
|
157
157
|
dataset(full_rows, 0.22, "_90d", END - timedelta(days=RECENT_DAYS))
|
|
@@ -168,7 +168,7 @@ def write_gsc(prop, slug, sites_in_prop, index_of):
|
|
|
168
168
|
"ctr": clicks / imps if imps else 0, "position": round(rng.uniform(9, 14), 1)})
|
|
169
169
|
(d / "dates.json").write_text(json.dumps(
|
|
170
170
|
{"site": prop, "dimensions": ["date"], "startDate": day0.isoformat(), "endDate": END.isoformat(),
|
|
171
|
-
"rowCount": len(dates), "rows": dates}))
|
|
171
|
+
"rowCount": len(dates), "rows": dates}), encoding="utf-8")
|
|
172
172
|
return full_rows
|
|
173
173
|
|
|
174
174
|
|
|
@@ -199,14 +199,14 @@ def write_ga4(site, idx, conv, gsc_rows):
|
|
|
199
199
|
total = sum(m[0] for _, m in daily)
|
|
200
200
|
meta = {"site": host, "property": f"properties/{site.get('ga4Property') or '000000000'}",
|
|
201
201
|
"pulled": TODAY.isoformat()}
|
|
202
|
-
(d / "daily.json").write_text(json.dumps({**meta, **ga4_report(["date"], ["sessions", "totalUsers"], daily)}))
|
|
202
|
+
(d / "daily.json").write_text(json.dumps({**meta, **ga4_report(["date"], ["sessions", "totalUsers"], daily)}), encoding="utf-8")
|
|
203
203
|
|
|
204
204
|
mix = [(src, med, w) for src, med, w, _ai in SOURCE_MIX]
|
|
205
205
|
others = [h for h in seo_config.hosts() if h != host]
|
|
206
206
|
if others:
|
|
207
207
|
mix.append((others[0], "referral", 0.015))
|
|
208
208
|
sources = [([src, med], [round(total * w), round(total * w * 0.8)]) for src, med, w in mix]
|
|
209
|
-
(d / "sources.json").write_text(json.dumps({**meta, **ga4_report(["sessionSource", "sessionMedium"], ["sessions", "totalUsers"], sources)}))
|
|
209
|
+
(d / "sources.json").write_text(json.dumps({**meta, **ga4_report(["sessionSource", "sessionMedium"], ["sessions", "totalUsers"], sources)}), encoding="utf-8")
|
|
210
210
|
|
|
211
211
|
pages = sorted({r[1] for r in gsc_rows if r[1].split("/")[2] == site["gscHost"]})
|
|
212
212
|
paths = [p.split(site["gscHost"], 1)[1].rstrip("/") or "/" for p in pages] or ["/", "/docs", "/pricing"]
|
|
@@ -219,7 +219,7 @@ def write_ga4(site, idx, conv, gsc_rows):
|
|
|
219
219
|
eng = 0.18
|
|
220
220
|
landing.append(([p], [sessions, round(eng, 4)]))
|
|
221
221
|
landing.sort(key=lambda r: -r[1][0])
|
|
222
|
-
(d / "landing.json").write_text(json.dumps({**meta, **ga4_report(["landingPage"], ["sessions", "engagementRate"], landing)}))
|
|
222
|
+
(d / "landing.json").write_text(json.dumps({**meta, **ga4_report(["landingPage"], ["sessions", "engagementRate"], landing)}), encoding="utf-8")
|
|
223
223
|
|
|
224
224
|
if conv and conv.get("site") == host:
|
|
225
225
|
rows = []
|
|
@@ -227,7 +227,7 @@ def write_ga4(site, idx, conv, gsc_rows):
|
|
|
227
227
|
day = (day0 + timedelta(days=i)).strftime("%Y%m%d")
|
|
228
228
|
for ev in conv.get("events", [])[:2]:
|
|
229
229
|
rows.append(([day, ev, rng.choice(["web", "docs", "(not set)"])], [rng.randint(0, 4)]))
|
|
230
|
-
(d / "funnel.json").write_text(json.dumps({**meta, **ga4_report(["date", "eventName", conv.get("sourceDimension") or "customEvent:source_app"], ["eventCount"], rows)}))
|
|
230
|
+
(d / "funnel.json").write_text(json.dumps({**meta, **ga4_report(["date", "eventName", conv.get("sourceDimension") or "customEvent:source_app"], ["eventCount"], rows)}), encoding="utf-8")
|
|
231
231
|
return paths, total
|
|
232
232
|
|
|
233
233
|
|
|
@@ -250,7 +250,7 @@ def write_timeseries(prop, slug, sites_in_prop, gsc_rows, ga4_paths):
|
|
|
250
250
|
rows.append({"keys": [day.isoformat(), p], "clicks": round(dc), "impressions": round(di),
|
|
251
251
|
"ctr": (dc / di) if di else 0, "position": round(rng.uniform(4, 20), 1)})
|
|
252
252
|
(d / f"gsc-{slug}.json").write_text(json.dumps(
|
|
253
|
-
{"site": prop, "startDate": day0.isoformat(), "endDate": END.isoformat(), "rows": rows}))
|
|
253
|
+
{"site": prop, "startDate": day0.isoformat(), "endDate": END.isoformat(), "rows": rows}), encoding="utf-8")
|
|
254
254
|
for s in sites_in_prop:
|
|
255
255
|
paths, total = ga4_paths.get(s["host"], ([], 0))
|
|
256
256
|
if not s.get("ga4Property"):
|
|
@@ -263,7 +263,7 @@ def write_timeseries(prop, slug, sites_in_prop, gsc_rows, ga4_paths):
|
|
|
263
263
|
v = weekly(day, (total / 90) * (0.35 if j == 0 else 0.08), day0=ga0)
|
|
264
264
|
if round(v):
|
|
265
265
|
rows.append({"date": day.strftime("%Y%m%d"), "page": p, "sessions": round(v)})
|
|
266
|
-
(d / f"ga4-{s['host']}.json").write_text(json.dumps({"site": s["host"], "rows": rows}))
|
|
266
|
+
(d / f"ga4-{s['host']}.json").write_text(json.dumps({"site": s["host"], "rows": rows}), encoding="utf-8")
|
|
267
267
|
|
|
268
268
|
# date x source/medium. AI assistants grow over the window and the
|
|
269
269
|
# rest hold roughly flat, so the demo actually shows the thing the
|
|
@@ -278,7 +278,7 @@ def write_timeseries(prop, slug, sites_in_prop, gsc_rows, ga4_paths):
|
|
|
278
278
|
rows.append({"date": day.strftime("%Y%m%d"), "source": src,
|
|
279
279
|
"medium": med, "sessions": round(v)})
|
|
280
280
|
(d / f"ga4-sources-{s['host']}.json").write_text(
|
|
281
|
-
json.dumps({"site": s["host"], "rows": rows}))
|
|
281
|
+
json.dumps({"site": s["host"], "rows": rows}), encoding="utf-8")
|
|
282
282
|
|
|
283
283
|
|
|
284
284
|
def write_probe(sites):
|
|
@@ -302,7 +302,7 @@ def write_probe(sites):
|
|
|
302
302
|
"soft_404": {"status": 404, "real_404": True},
|
|
303
303
|
})
|
|
304
304
|
now = datetime.now(timezone.utc)
|
|
305
|
-
(d / f"probe-{now:%Y%m%d-%H%M%S}.json").write_text(json.dumps({"probed_at": now.isoformat(), "sites": out}, indent=2))
|
|
305
|
+
(d / f"probe-{now:%Y%m%d-%H%M%S}.json").write_text(json.dumps({"probed_at": now.isoformat(), "sites": out}, indent=2), encoding="utf-8")
|
|
306
306
|
|
|
307
307
|
|
|
308
308
|
def write_metadata_audit(sites, gsc_by_host):
|
|
@@ -333,7 +333,7 @@ def write_metadata_audit(sites, gsc_by_host):
|
|
|
333
333
|
"imps": imps, "clicks": sum(q["clicks"] for q in qs), "issues": issues,
|
|
334
334
|
"top_queries": qs[:5], "missed_clicks_window": missed})
|
|
335
335
|
audit["sites"][s["host"]] = findings
|
|
336
|
-
(DATA / "metadata-audit.json").write_text(json.dumps(audit, indent=1))
|
|
336
|
+
(DATA / "metadata-audit.json").write_text(json.dumps(audit, indent=1), encoding="utf-8")
|
|
337
337
|
|
|
338
338
|
|
|
339
339
|
def write_index_status(sites, gsc_by_host):
|
|
@@ -365,7 +365,7 @@ def write_index_status(sites, gsc_by_host):
|
|
|
365
365
|
"pending": False, "errors": 0, "warnings": 0}]},
|
|
366
366
|
"problems": problems,
|
|
367
367
|
}
|
|
368
|
-
(DATA / "index-status.json").write_text(json.dumps(out, indent=2))
|
|
368
|
+
(DATA / "index-status.json").write_text(json.dumps(out, indent=2), encoding="utf-8")
|
|
369
369
|
|
|
370
370
|
|
|
371
371
|
def write_trends(props, sites, gsc_rows_by_prop):
|
|
@@ -399,7 +399,7 @@ def write_trends(props, sites, gsc_rows_by_prop):
|
|
|
399
399
|
|
|
400
400
|
monthly = {}
|
|
401
401
|
dates_file = DATA / "gsc" / slug / "dates.json"
|
|
402
|
-
for r in json.loads(dates_file.read_text())["rows"]:
|
|
402
|
+
for r in json.loads(dates_file.read_text(encoding="utf-8"))["rows"]:
|
|
403
403
|
m = r["keys"][0][:7]
|
|
404
404
|
cur = monthly.setdefault(m, {"clicks": 0, "imps": 0})
|
|
405
405
|
cur["clicks"] += r["clicks"]
|
|
@@ -417,7 +417,7 @@ def write_trends(props, sites, gsc_rows_by_prop):
|
|
|
417
417
|
total[ym] = t
|
|
418
418
|
ai[ym] = round(t * (0.01 + (12 - k) * 0.004))
|
|
419
419
|
out["ai_referrals"][s["host"]] = {"ai": ai, "total": total}
|
|
420
|
-
(DATA / f"trends-{TODAY.isoformat()}.json").write_text(json.dumps(out, indent=1))
|
|
420
|
+
(DATA / f"trends-{TODAY.isoformat()}.json").write_text(json.dumps(out, indent=1), encoding="utf-8")
|
|
421
421
|
|
|
422
422
|
|
|
423
423
|
def write_proposals(sites, gsc_by_host):
|
|
@@ -450,7 +450,7 @@ def write_proposals(sites, gsc_by_host):
|
|
|
450
450
|
"evidence": "DEMO DATA — 12 days since the change; CTR up from 1.9% to 2.6% but the 28-day window has not closed."}],
|
|
451
451
|
"inference_ran": True,
|
|
452
452
|
}
|
|
453
|
-
(DATA / "opportunity-proposals.json").write_text(json.dumps(out, indent=1))
|
|
453
|
+
(DATA / "opportunity-proposals.json").write_text(json.dumps(out, indent=1), encoding="utf-8")
|
|
454
454
|
|
|
455
455
|
|
|
456
456
|
def write_run_files():
|
|
@@ -459,8 +459,8 @@ def write_run_files():
|
|
|
459
459
|
"ts": ts.strftime("%Y-%m-%dT%H:%MZ"), "failures": "",
|
|
460
460
|
"steps": [{"name": n, "ok": True, "seconds": s} for n, s in
|
|
461
461
|
(("probe", 6.2), ("gsc", 41.0), ("ga4", 9.8), ("timeseries", 22.4), ("metadata-audit", 14.1),
|
|
462
|
-
("index-status", 38.5), ("opportunity-scan", 17.0), ("daily-diff", 0.3))]}, indent=1))
|
|
463
|
-
with (DATA / "daily-ops.log").open("a") as f:
|
|
462
|
+
("index-status", 38.5), ("opportunity-scan", 17.0), ("daily-diff", 0.3))]}, indent=1), encoding="utf-8")
|
|
463
|
+
with (DATA / "daily-ops.log").open("a", encoding="utf-8") as f:
|
|
464
464
|
f.write(f"=== daily run {ts:%Y-%m-%d %H:%M} (demo data) ===\n")
|
|
465
465
|
f.write("--- probe\n 2 sites probed\n--- gsc\n 2 properties pulled\n--- daily-diff\n daily-log updated (0 alerts)\n")
|
|
466
466
|
f.write("=== done (0 failures) ===\n")
|
package/ops/doctor.py
CHANGED
|
@@ -64,10 +64,19 @@ def check_config():
|
|
|
64
64
|
|
|
65
65
|
def check_tools():
|
|
66
66
|
print("tools")
|
|
67
|
-
|
|
68
|
-
|
|
67
|
+
report("OK" if shutil.which("curl") else "FAIL", "curl",
|
|
68
|
+
"" if shutil.which("curl") else "install it — all HTTP goes through curl "
|
|
69
|
+
"(Windows 10+ ships it in System32)")
|
|
70
|
+
# node signs the service-account JWT, so it is required even for someone
|
|
71
|
+
# who never opens the dashboard.
|
|
72
|
+
import google_auth
|
|
73
|
+
node = google_auth.node_bin()
|
|
74
|
+
report("OK" if shutil.which(node) else "FAIL", f"node ({node})",
|
|
75
|
+
"" if shutil.which(node) else "install Node 20+ — it runs the dashboard "
|
|
76
|
+
"and signs the service-account JWT; $NODE overrides the path")
|
|
69
77
|
v = sys.version_info
|
|
70
|
-
report("OK" if v >= (3, 10) else "FAIL",
|
|
78
|
+
report("OK" if v >= (3, 10) else "FAIL",
|
|
79
|
+
f"python {v.major}.{v.minor} ({Path(sys.executable).name})",
|
|
71
80
|
"" if v >= (3, 10) else "3.10+ required")
|
|
72
81
|
nm = seo_config.ROOT / "node_modules"
|
|
73
82
|
report("OK" if nm.exists() else "WARN", "node_modules", "" if nm.exists() else "run: npm install")
|
|
@@ -86,7 +95,7 @@ def check_auth(cfg):
|
|
|
86
95
|
f"{kp or '(unset)'} — set google.serviceAccountKey or $GOOGLE_APPLICATION_CREDENTIALS; see docs/SETUP-GOOGLE.md")
|
|
87
96
|
return None
|
|
88
97
|
try:
|
|
89
|
-
key = json.loads(Path(kp).read_text())
|
|
98
|
+
key = json.loads(Path(kp).read_text(encoding="utf-8"))
|
|
90
99
|
except (OSError, json.JSONDecodeError) as exc:
|
|
91
100
|
report("FAIL", "key file unreadable", str(exc))
|
|
92
101
|
return None
|
|
@@ -197,7 +206,6 @@ def check_modules(cfg):
|
|
|
197
206
|
report("OK", "enabled: " + (", ".join(on) or "(none beyond defaults)"))
|
|
198
207
|
llm = cfg["modules"].get("llm", {})
|
|
199
208
|
if llm.get("enabled"):
|
|
200
|
-
import shlex
|
|
201
209
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
202
210
|
import llm as llm_mod
|
|
203
211
|
http = llm_mod.http_config()
|
|
@@ -211,11 +219,17 @@ def check_modules(cfg):
|
|
|
211
219
|
f"check provider/model and that {llm['http'].get('apiKeyEnv') or 'apiKeyEnv'} "
|
|
212
220
|
"is set in the environment or .env")
|
|
213
221
|
for key in ("command", "fastCommand"):
|
|
214
|
-
|
|
222
|
+
raw = str(llm.get(key) or "").strip()
|
|
223
|
+
cmd = llm_mod.split_command(raw)
|
|
215
224
|
if not cmd:
|
|
225
|
+
if raw:
|
|
226
|
+
# Configured but unparseable — almost always an unbalanced
|
|
227
|
+
# quote. Say so, or the next line reads as "not configured".
|
|
228
|
+
report("FAIL", f"llm.{key} could not be parsed: {raw[:60]}",
|
|
229
|
+
"check for an unbalanced quote")
|
|
216
230
|
# Only complain about a missing command when there is no http
|
|
217
231
|
# block at all; a broken one already reported itself.
|
|
218
|
-
|
|
232
|
+
elif key == "command" and not http and not llm.get("http"):
|
|
219
233
|
report("FAIL", "llm has neither http nor command configured")
|
|
220
234
|
continue
|
|
221
235
|
report("OK" if shutil.which(cmd[0]) else "FAIL", f"llm.{key}: {cmd[0]}",
|
package/ops/export_static.py
CHANGED
|
@@ -103,11 +103,11 @@ def main():
|
|
|
103
103
|
html = html.replace("</body>", staleness + "</body>", 1)
|
|
104
104
|
dest = out / "index.html" if route == "/" else out / route.lstrip("/") / "index.html"
|
|
105
105
|
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
106
|
-
dest.write_text(html)
|
|
106
|
+
dest.write_text(html, encoding="utf-8")
|
|
107
107
|
|
|
108
|
-
(out / "styles.css").write_text(fetch("/styles.css"))
|
|
109
|
-
(out / "favicon.svg").write_text(fetch("/favicon.svg"))
|
|
110
|
-
(out / "robots.txt").write_text("User-agent: *\nDisallow: /\n")
|
|
108
|
+
(out / "styles.css").write_text(fetch("/styles.css"), encoding="utf-8")
|
|
109
|
+
(out / "favicon.svg").write_text(fetch("/favicon.svg"), encoding="utf-8")
|
|
110
|
+
(out / "robots.txt").write_text("User-agent: *\nDisallow: /\n", encoding="utf-8")
|
|
111
111
|
except BaseException:
|
|
112
112
|
# Never leave a half-built staging directory behind; the next
|
|
113
113
|
# run would otherwise start from someone else's leftovers.
|
package/ops/hn_digest.py
CHANGED
|
@@ -157,7 +157,7 @@ def main():
|
|
|
157
157
|
seo_config.DATA.mkdir(parents=True, exist_ok=True)
|
|
158
158
|
(seo_config.DATA / "hn-digest.json").write_text(json.dumps(
|
|
159
159
|
{"generated": datetime.now(timezone.utc).isoformat(timespec="minutes"),
|
|
160
|
-
"stats": stats, "picks": picks}, indent=1))
|
|
160
|
+
"stats": stats, "picks": picks}, indent=1), encoding="utf-8")
|
|
161
161
|
for p in picks:
|
|
162
162
|
print(f"- {p['title']} ({p['comments']}c/{p['points']}p) {p['url']} [{p['why']}]")
|
|
163
163
|
if not picks:
|
package/ops/indexnow.py
CHANGED
|
@@ -33,12 +33,12 @@ def key_file() -> Path:
|
|
|
33
33
|
def cmd_init() -> int:
|
|
34
34
|
kf = key_file()
|
|
35
35
|
if kf.exists():
|
|
36
|
-
key = kf.read_text().strip()
|
|
36
|
+
key = kf.read_text(encoding="utf-8").strip()
|
|
37
37
|
print(f"key file already exists: {kf}")
|
|
38
38
|
else:
|
|
39
39
|
key = secrets.token_hex(16)
|
|
40
40
|
kf.parent.mkdir(parents=True, exist_ok=True)
|
|
41
|
-
kf.write_text(key + "\n")
|
|
41
|
+
kf.write_text(key + "\n", encoding="utf-8")
|
|
42
42
|
print(f"wrote {kf}")
|
|
43
43
|
print(f"\nkey: {key}\n\nServe this key as plain text at, for every site:")
|
|
44
44
|
for s in seo_config.sites():
|
|
@@ -52,7 +52,7 @@ def cmd_ping(urls: list[str]) -> int:
|
|
|
52
52
|
if not kf.exists():
|
|
53
53
|
print(f"no key file at {kf} — run: python3 ops/indexnow.py init")
|
|
54
54
|
return 1
|
|
55
|
-
key = kf.read_text().strip()
|
|
55
|
+
key = kf.read_text(encoding="utf-8").strip()
|
|
56
56
|
by_host: dict[str, list[str]] = {}
|
|
57
57
|
for u in urls:
|
|
58
58
|
host = urlsplit(u).netloc
|
package/ops/llm.py
CHANGED
|
@@ -28,6 +28,7 @@ path runs. Nothing that comes back is applied automatically: callers turn
|
|
|
28
28
|
replies into briefings or proposals a human accepts or ignores.
|
|
29
29
|
"""
|
|
30
30
|
import json
|
|
31
|
+
import os
|
|
31
32
|
import shlex
|
|
32
33
|
import shutil
|
|
33
34
|
import subprocess
|
|
@@ -42,6 +43,27 @@ ANTHROPIC_VERSION = "2023-06-01"
|
|
|
42
43
|
MAX_TOKENS = 4096
|
|
43
44
|
|
|
44
45
|
|
|
46
|
+
def split_command(cmd: str) -> list[str]:
|
|
47
|
+
"""Split a configured command string into argv, per platform.
|
|
48
|
+
|
|
49
|
+
POSIX rules treat a backslash as an escape, so on Windows they quietly
|
|
50
|
+
turn `C:\\tools\\claude.exe` into `C:toolsclaude.exe`. There, split with
|
|
51
|
+
the non-POSIX rules cmd.exe uses and strip the quotes shlex leaves on.
|
|
52
|
+
|
|
53
|
+
Returns [] for anything unparseable. This string comes from the Settings
|
|
54
|
+
form, so a stray quote is a typo a user can make, and it must not take
|
|
55
|
+
the daily run down with a ValueError from inside shlex — callers already
|
|
56
|
+
treat [] as "no command configured". doctor tells them which it was.
|
|
57
|
+
"""
|
|
58
|
+
try:
|
|
59
|
+
if os.name != "nt":
|
|
60
|
+
return shlex.split(cmd)
|
|
61
|
+
parts = shlex.split(cmd, posix=False)
|
|
62
|
+
except ValueError:
|
|
63
|
+
return []
|
|
64
|
+
return [p[1:-1] if len(p) > 1 and p[0] == p[-1] == '"' else p for p in parts]
|
|
65
|
+
|
|
66
|
+
|
|
45
67
|
def _enabled() -> dict | None:
|
|
46
68
|
m = seo_config.module("llm")
|
|
47
69
|
return m if m.get("enabled") else None
|
|
@@ -71,7 +93,7 @@ def command(fast: bool = False) -> list[str] | None:
|
|
|
71
93
|
if not m:
|
|
72
94
|
return None
|
|
73
95
|
cmd = (m.get("fastCommand") if fast else None) or m.get("command") or ""
|
|
74
|
-
parts =
|
|
96
|
+
parts = split_command(str(cmd))
|
|
75
97
|
if not parts or not shutil.which(parts[0]):
|
|
76
98
|
return None
|
|
77
99
|
return parts
|