@topy-ai/maggie 0.7.20 → 0.7.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,6 +17,8 @@ from booking_capabilities import audit_matrix # noqa: E402
17
17
  NOW = lambda: datetime.now(timezone.utc).isoformat()
18
18
  MONEY = re.compile(r"(?:£|GBP\s*)\s*([0-9]+(?:[.,][0-9]{1,2})?)", re.I)
19
19
  DURATION = re.compile(r"(\d{2,3})\s*(?:min|mins|minutes?)", re.I)
20
+ SUPPLY_STATES = {"live", "withdrawn"}
21
+ DISPLAY_STATES = {"published", "hidden", "retired"}
20
22
 
21
23
  def slug(value):
22
24
  value = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-")
@@ -57,33 +59,37 @@ def flatten(value):
57
59
 
58
60
  def parse(source, provider):
59
61
  raw=fetch(source); parser=Extractor(); parser.feed(raw); records=[]
62
+ embedded_marker = provider == "fresha" and re.search(r'"services"\s*:\s*\[', raw)
60
63
  if provider == "fresha":
61
64
  records.extend(parse_fresha_embedded(raw, source))
62
- for obj in json_objects(parser.scripts):
63
- for item in flatten(obj):
64
- typ=item.get("@type", "")
65
- types=set(typ) if isinstance(typ, list) else {typ}
66
- if not types.intersection({"Service","Product","Offer","ListItem"}) and not any(k in item for k in ("duration","price","providerServiceId")): continue
67
- name=item.get("name") or item.get("title") or (item.get("item") or {}).get("name") if isinstance(item.get("item"),dict) else item.get("name")
68
- if not name or len(str(name)) < 3: continue
69
- offers=item.get("offers") if isinstance(item.get("offers"),list) else [item.get("offers", item)]
70
- variants=[]
71
- for offer in offers:
72
- if not isinstance(offer,dict): offer={}
73
- text=" ".join(str(offer.get(k,"")) for k in ("name","description","duration"))
74
- dm=DURATION.search(text); price=offer.get("price")
75
- if price is None:
76
- match=MONEY.search(text); price=match.group(1).replace(",","") if match else None
77
- currency=(offer.get("priceCurrency") or ("GBP" if price and ("£" in text or provider=="fresha") else None))
78
- if price is None and dm is None: continue
79
- amount=int(round(float(str(price).replace(",",""))*100)) if price is not None else None
80
- provider_variant_id = offer.get("sku") or offer.get("@id") or offer.get("id")
81
- variants.append({"id":str(provider_variant_id or f"{slug(name)}:{dm.group(1) if dm else 'default'}"),"idSource":"provider" if provider_variant_id else "derived","variantSource":"native" if provider_variant_id else "derived-offers","title":offer.get("name") or (f"{dm.group(1)} minutes" if dm else "Standard"),"durationMinutes":int(dm.group(1)) if dm else None,"price":{"amountMinor":amount,"currency":currency,"display":f"£{amount/100:.2f}" if amount is not None and currency=="GBP" else None}})
82
- url=item.get("url") or item.get("sameAs")
83
- if not variants and not url: continue
84
- pid=str(item.get("providerServiceId") or item.get("productID") or item.get("sku") or slug(name))
85
- records.append(make_record(provider, source, pid, str(name), str(item.get("description") or ""), variants, url, parser.links))
86
- if not records:
65
+ # A provider-specific authoritative catalogue is the only service source.
66
+ # Generic JSON-LD may include marketplace entities for another business.
67
+ if not embedded_marker:
68
+ for obj in json_objects(parser.scripts):
69
+ for item in flatten(obj):
70
+ typ=item.get("@type", "")
71
+ types=set(typ) if isinstance(typ, list) else {typ}
72
+ if not types.intersection({"Service","Product","Offer","ListItem"}) and not any(k in item for k in ("providerServiceId","productID","sku")): continue
73
+ name=item.get("name") or item.get("title") or (item.get("item") or {}).get("name") if isinstance(item.get("item"),dict) else item.get("name")
74
+ if not name or len(str(name)) < 3: continue
75
+ offers=item.get("offers") if isinstance(item.get("offers"),list) else [item.get("offers", item)]
76
+ variants=[]
77
+ for offer in offers:
78
+ if not isinstance(offer,dict): offer={}
79
+ text=" ".join(str(offer.get(k,"")) for k in ("name","description","duration"))
80
+ dm=DURATION.search(text); price=offer.get("price")
81
+ if price is None:
82
+ match=MONEY.search(text); price=match.group(1).replace(",","") if match else None
83
+ currency=(offer.get("priceCurrency") or ("GBP" if price and ("£" in text or provider=="fresha") else None))
84
+ if price is None and dm is None: continue
85
+ amount=int(round(float(str(price).replace(",",""))*100)) if price is not None else None
86
+ provider_variant_id = offer.get("sku") or offer.get("@id") or offer.get("id")
87
+ variants.append({"id":str(provider_variant_id or f"{slug(name)}:{dm.group(1) if dm else 'default'}"),"idSource":"provider" if provider_variant_id else "derived","variantSource":"native" if provider_variant_id else "derived-offers","title":offer.get("name") or (f"{dm.group(1)} minutes" if dm else "Standard"),"durationMinutes":int(dm.group(1)) if dm else None,"price":{"amountMinor":amount,"currency":currency,"display":f"£{amount/100:.2f}" if amount is not None and currency=="GBP" else None}})
88
+ url=item.get("url") or item.get("sameAs")
89
+ if not variants and not url: continue
90
+ pid=str(item.get("providerServiceId") or item.get("productID") or item.get("sku") or slug(name))
91
+ records.append(make_record(provider, source, pid, str(name), str(item.get("description") or ""), variants, url, parser.links))
92
+ if not records and not embedded_marker:
87
93
  for _, pieces in parser.headings:
88
94
  name=" ".join(pieces).strip()
89
95
  if len(name)>3: records.append(make_record(provider, source, slug(name), name, "", [], None, parser.links))
@@ -97,10 +103,10 @@ def parse(source, provider):
97
103
  def parse_fresha_embedded(raw, source):
98
104
  """Read the public service catalogue embedded in Fresha's Next data."""
99
105
  records=[]
100
- marker='"services":['; start=raw.find(marker)
101
- if start >= 0:
106
+ marker=re.search(r'"services"\s*:\s*\[', raw); start=marker.start() if marker else -1
107
+ if marker:
102
108
  try:
103
- groups=json.JSONDecoder().raw_decode(raw[start+len('"services":'):])[0]
109
+ groups=json.JSONDecoder().raw_decode(raw[marker.end()-1:])[0]
104
110
  except (ValueError, TypeError): groups=[]
105
111
  for group in groups if isinstance(groups,list) else []:
106
112
  level1=str(group.get("name") or "Uncategorised").strip()
@@ -182,7 +188,7 @@ def make_record(provider, source, pid, name, description, variants, booking, lin
182
188
  if not booking:
183
189
  for href,_ in links:
184
190
  if any(x in href.lower() for x in ("book","appointment","checkout")): booking=urljoin(source,href); break
185
- return {"id":f"{provider}:{pid}","provider":provider,"providerServiceId":pid,"slug":slug(name),"title":name,"description":description.strip(),"sourceUrl":source,"category":{"level1":"Uncategorised","level2":None},"variants":variants,"bookingUrl":booking,"paymentUrl":None,"status":"active","firstSeenAt":NOW(),"lastSeenAt":NOW()}
191
+ return {"id":f"{provider}:{pid}","provider":provider,"providerServiceId":pid,"slug":slug(name),"title":name,"description":description.strip(),"sourceUrl":source,"category":{"level1":"Uncategorised","level2":None},"variants":variants,"bookingUrl":booking,"paymentUrl":None,"status":"active","supplyState":"live","displayState":"published","firstSeenAt":NOW(),"lastSeenAt":NOW()}
186
192
 
187
193
  def root(args): return Path(args.project).resolve()
188
194
  def path(project): return project/".maggie"/"booking"/"services.json"
@@ -199,11 +205,15 @@ def validate(data):
199
205
  if not s.get(key): errors.append(f"{s.get('id','unknown')}: missing {key}")
200
206
  if s.get("id") in seen: errors.append(f"duplicate service id: {s['id']}")
201
207
  seen.add(s.get("id"));
208
+ supply=s.get("supplyState") or ("withdrawn" if s.get("status") in {"archived", "removed"} else "live")
209
+ display=s.get("displayState") or ("retired" if s.get("status") == "archived" else "published")
210
+ if supply not in SUPPLY_STATES: errors.append(f"{s.get('id')}: invalid supplyState")
211
+ if display not in DISPLAY_STATES: errors.append(f"{s.get('id')}: invalid displayState")
202
212
  if any(x.get("slug")==s.get("slug") for x in data.get("services",[]) if x is not s): errors.append(f"duplicate service slug: {s.get('slug')}")
203
213
  pages=s.get("pages",[])
204
214
  if not isinstance(pages,list): errors.append(f"{s.get('id')}: pages must be an array"); pages=[]
205
215
  if sum(p.get("role")=="canonical" for p in pages if isinstance(p,dict))>1: errors.append(f"{s.get('id')}: multiple canonical pages")
206
- if s.get("status")=="active" and not s.get("variants"): errors.append(f"{s.get('id')}: no variants")
216
+ if supply == "live" and not s.get("variants"): errors.append(f"{s.get('id')}: no variants")
207
217
  for v in s.get("variants",[]):
208
218
  if not v.get("durationMinutes") or not v.get("price",{}).get("currency"): errors.append(f"{s.get('id')}: incomplete variant {v.get('id')}")
209
219
  for key in ("bookingUrl","paymentUrl"):
@@ -312,33 +322,172 @@ def cmd_fact_audit(args):
312
322
  print(json.dumps({"status": "passed" if report["passed"] else "failed", "report": str(out), "reviewQueue": str(queue_path), "services": len(services), "variants": len(seen_variants), "sourceUrlsBackfilled": changed, "errors": errors[:20], "errorCount": len(errors), "publish": False}, indent=2, ensure_ascii=False))
313
323
  return 0 if report["passed"] else 1
314
324
 
325
+ def cmd_catalogue_check(args):
326
+ """Answer treatment-presence questions using the import/sync parser."""
327
+ try:
328
+ data = parse(args.source, args.provider)
329
+ except (OSError, ValueError, UnicodeError):
330
+ print(json.dumps({"status": "failed", "error": "source could not be parsed"}, indent=2))
331
+ return 1
332
+ services = data.get("services", [])
333
+ results = []
334
+ for query in args.treatment:
335
+ query_tokens = page_tokens(query)
336
+ matches = [
337
+ {"id": service.get("id"), "title": service.get("title"), "category": service.get("category", {}).get("level1")}
338
+ for service in services
339
+ if query_tokens and query_tokens <= page_tokens(service.get("title", ""))
340
+ ]
341
+ results.append({"treatment": query, "status": "found" if matches else "not-found", "matches": matches})
342
+ passed = all(item["status"] == "found" for item in results)
343
+ print(json.dumps({"status": "passed" if passed else "failed", "provider": args.provider, "serviceCount": len(services), "results": results}, indent=2, ensure_ascii=False))
344
+ return 0 if passed else 1
345
+
346
+ RETIREMENT_SCHEMA = "maggie-service-retirement-evidence.v1"
347
+ RETIREMENT_ENDINGS = {"pending", "redirect", "tombstone", "gone"}
348
+
349
+ def cmd_retirement_audit(args):
350
+ """Validate sanitized runtime evidence for withdrawn service URLs."""
351
+ project=root(args); catalogue_path=Path(args.catalogue); catalogue_path=catalogue_path if catalogue_path.is_absolute() else project/catalogue_path
352
+ evidence_path=Path(args.evidence); evidence_path=evidence_path if evidence_path.is_absolute() else project/evidence_path
353
+ try:
354
+ catalogue=json.loads(catalogue_path.read_text(encoding="utf-8")); evidence=json.loads(evidence_path.read_text(encoding="utf-8"))
355
+ except (OSError, ValueError):
356
+ print(json.dumps({"status":"failed","error":"catalogue or retirement evidence is invalid"},indent=2)); return 1
357
+ services={str(item.get("id")):item for item in catalogue.get("services",[]) if isinstance(item,dict) and item.get("id")}
358
+ rows=evidence.get("services") if isinstance(evidence,dict) else None
359
+ errors=[]; results=[]
360
+ if not isinstance(evidence,dict) or evidence.get("schemaVersion") != RETIREMENT_SCHEMA:
361
+ errors.append(f"evidence schemaVersion must be {RETIREMENT_SCHEMA}")
362
+ if not isinstance(rows,list):
363
+ errors.append("evidence.services must be a non-empty array")
364
+ rows=[]
365
+ if isinstance(rows,list) and not rows: errors.append("evidence.services must be a non-empty array")
366
+ seen=set()
367
+ for index,row in enumerate(rows):
368
+ item_errors=[]
369
+ if not isinstance(row,dict):
370
+ results.append({"index":index,"passed":False,"errors":["evidence row must be an object"]}); continue
371
+ service_id=str(row.get("serviceId") or "")
372
+ if not service_id or service_id in seen: item_errors.append("serviceId is missing or duplicated")
373
+ seen.add(service_id)
374
+ service=services.get(service_id)
375
+ if not service: item_errors.append("service is not in the catalogue")
376
+ supply=(service or {}).get("supplyState") or ("withdrawn" if (service or {}).get("status") in {"archived","removed"} else "live")
377
+ if supply != "withdrawn": item_errors.append("retirement evidence requires a withdrawn service")
378
+ ending=str(row.get("ending") or "")
379
+ if ending not in RETIREMENT_ENDINGS: item_errors.append("ending must be pending, redirect, tombstone, or gone")
380
+ expected_ending=((service or {}).get("retirement") or {}).get("ending")
381
+ if expected_ending and expected_ending != ending: item_errors.append("evidence ending does not match the catalogue decision")
382
+ status=row.get("httpStatus")
383
+ if not isinstance(status,int) or isinstance(status,bool): item_errors.append("httpStatus must be an integer")
384
+ if row.get("inSitemap") is not False: item_errors.append("withdrawn service must be absent from the sitemap")
385
+ if row.get("bookingSuppressed") is not True: item_errors.append("bookingSuppressed must be true")
386
+ if ending in {"pending","tombstone"}:
387
+ if status != 200: item_errors.append(f"{ending} must return HTTP 200")
388
+ if row.get("unavailableNotice") is not True: item_errors.append("unavailableNotice must be true")
389
+ if row.get("noindex") is not True: item_errors.append("noindex must be true")
390
+ elif ending == "redirect":
391
+ if status not in {301,308}: item_errors.append("redirect must return HTTP 301 or 308")
392
+ location=str(row.get("location") or "")
393
+ route=str(row.get("route") or "")
394
+ if not location or not location.startswith("/") or location == route: item_errors.append("redirect needs a different same-site location")
395
+ elif ending == "gone" and status != 410:
396
+ item_errors.append("gone must return HTTP 410")
397
+ results.append({"serviceId":service_id,"ending":ending,"passed":not item_errors,"errors":item_errors})
398
+ passed=not errors and bool(results) and all(item["passed"] for item in results)
399
+ report={"schemaVersion":"maggie-service-retirement-audit.v1","evidenceSchema":RETIREMENT_SCHEMA,"checked":len(results),"services":results,"errors":errors,"passed":passed}
400
+ output=Path(args.output).expanduser(); output=output if output.is_absolute() else project/output; output.parent.mkdir(parents=True,exist_ok=True); output.write_text(json.dumps(report,indent=2,ensure_ascii=False)+"\n",encoding="utf-8")
401
+ print(json.dumps({"status":"passed" if passed else "failed","report":str(output.resolve()),"checked":len(results),"errors":errors},indent=2)); return 0 if passed else 1
402
+
315
403
  def cmd_import(args):
316
404
  project=root(args); data=parse(args.source,args.provider); old=load(project) if path(project).exists() else {"services":[]}; previous={s["id"]:s for s in old.get("services",[])}
317
405
  for service in data.get("services",[]):
318
406
  if service["id"] in previous:
319
- service["pages"]=previous[service["id"]].get("pages",[]); service["additional"]={**previous[service["id"]].get("additional",{}),**service.get("additional",{})}
407
+ old_service=previous[service["id"]]
408
+ preserve_site_fields(service, old_service)
409
+ service["displayState"]=old_service.get("displayState") or ("retired" if old_service.get("status") == "archived" else "published")
410
+ service.setdefault("supplyState", "live"); service.setdefault("displayState", "published")
320
411
  errors=validate(data)
321
412
  if errors: print(json.dumps({"status":"failed","errors":errors},indent=2)); return 1
322
413
  save(project,data); print(json.dumps({"status":"imported","path":str(path(project)),"reviewPath":str(project/"docs"/"services.json"),"services":len(data["services"])},indent=2)); return 0
414
+
415
+ SERVICE_DIFF_FIELDS = ("provider", "providerServiceId", "slug", "title", "description", "category", "bookingUrl", "paymentUrl", "status", "supplyState", "displayState", "retirement")
416
+
417
+ def preserve_site_fields(service, old_service):
418
+ proposal=None
419
+ old_slug=str(old_service.get("slug") or "")
420
+ new_slug=str(service.get("slug") or "")
421
+ if old_slug and new_slug and old_slug != new_slug:
422
+ proposal={"status":"pending","from":old_slug,"to":new_slug,"reason":"provider name implies a different public slug"}
423
+ service["slug"]=old_slug
424
+ service["pages"] = old_service.get("pages", [])
425
+ service["additional"] = {**old_service.get("additional", {}), **service.get("additional", {})}
426
+ return proposal
427
+
428
+ def service_diff(before, after):
429
+ fields=[]
430
+ for field in SERVICE_DIFF_FIELDS:
431
+ old_value = before.get(field) if isinstance(before, dict) else None
432
+ new_value = after.get(field) if isinstance(after, dict) else None
433
+ if old_value != new_value:
434
+ fields.append({"field": field, "before": old_value, "after": new_value})
435
+ old_variants={str(item.get("id")): item for item in (before or {}).get("variants", []) if isinstance(item, dict) and item.get("id")}
436
+ new_variants={str(item.get("id")): item for item in (after or {}).get("variants", []) if isinstance(item, dict) and item.get("id")}
437
+ variants=[]
438
+ for variant_id in sorted(set(old_variants) | set(new_variants)):
439
+ old_variant=old_variants.get(variant_id); new_variant=new_variants.get(variant_id)
440
+ if old_variant is None:
441
+ variants.append({"id": variant_id, "change": "added", "before": None, "after": new_variant})
442
+ elif new_variant is None:
443
+ variants.append({"id": variant_id, "change": "removed", "before": old_variant, "after": None})
444
+ elif old_variant != new_variant:
445
+ variants.append({"id": variant_id, "change": "updated", "before": old_variant, "after": new_variant})
446
+ return {"fields": fields, "variantChanges": variants}
447
+
323
448
  def cmd_sync(args):
324
449
  project=root(args); old=load(project) if path(project).exists() else {"services":[]}; fresh=parse(args.source,args.provider); previous={s["id"]:s for s in old.get("services",[])}; current={s["id"]:s for s in fresh.get("services",[])}; changes=[]
325
- def comparable(service):
326
- value=dict(service)
327
- value.pop("firstSeenAt", None)
328
- value.pop("lastSeenAt", None)
329
- return value
330
450
  for sid, service in current.items():
331
451
  old_service = previous.get(sid)
452
+ slug_proposal=None
332
453
  if old_service:
333
- service["pages"] = old_service.get("pages", [])
334
- service["additional"] = {**old_service.get("additional", {}), **service.get("additional", {})}
335
- changes.append({"id":sid,"change":"added" if sid not in previous else ("updated" if json.dumps(comparable(service),sort_keys=True,ensure_ascii=False)!=json.dumps(comparable(old_service),sort_keys=True,ensure_ascii=False) else "unchanged")})
454
+ slug_proposal=preserve_site_fields(service, old_service)
455
+ # Provider sync owns supply; a person's display decision survives
456
+ # the run and is never reset by a newly fetched row.
457
+ service["displayState"] = old_service.get("displayState") or ("retired" if old_service.get("status") == "archived" else "published")
458
+ service["supplyState"] = "live"
459
+ diff=service_diff(old_service, service)
460
+ change="added" if sid not in previous else ("updated" if diff["fields"] or diff["variantChanges"] else "unchanged")
461
+ if slug_proposal:
462
+ diff["slugProposal"]=slug_proposal
463
+ diff["reviewRequired"]=True
464
+ change="updated"
465
+ changes.append({"id":sid,"change":change,**diff})
336
466
  for sid, service in previous.items():
337
- if sid not in current: archived={**service,"status":"archived","removedAt":NOW()}; current[sid]=archived; changes.append({"id":sid,"change":"removed"})
467
+ if sid not in current:
468
+ archived={**service,"status":"archived","supplyState":"withdrawn","displayState":service.get("displayState") or ("retired" if service.get("status") == "archived" else "published"),"retirement":service.get("retirement") or {"ending":"pending"},"removedAt":NOW()}; current[sid]=archived
469
+ changes.append({"id":sid,"change":"removed","fields":[{"field":"status","before":service.get("status"),"after":"archived"}],"variantChanges":[]})
338
470
  fresh["services"]=list(current.values()); save(project,fresh)
339
- run_id=datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
340
- report={"schemaVersion":"1.0","runId":run_id,"syncedAt":fresh["syncedAt"],"sourceUrl":args.source,"provider":args.provider,"changes":changes,"summary":{"total":len(changes),"added":sum(c["change"]=="added" for c in changes),"updated":sum(c["change"]=="updated" for c in changes),"removed":sum(c["change"]=="removed" for c in changes),"unchanged":sum(c["change"]=="unchanged" for c in changes)},"checkpoint":{"phase":"reconciled","processed":len(changes),"total":len(changes),"resumable":True,"status":"complete"}}
471
+ run_id=datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%S%fZ")
472
+ report={"schemaVersion":"maggie-service-sync-report.v1","runId":run_id,"syncedAt":fresh["syncedAt"],"sourceUrl":args.source,"provider":args.provider,"changes":changes,"summary":{"total":len(changes),"added":sum(c["change"]=="added" for c in changes),"updated":sum(c["change"]=="updated" for c in changes),"removed":sum(c["change"]=="removed" for c in changes),"unchanged":sum(c["change"]=="unchanged" for c in changes)},"checkpoint":{"phase":"reconciled","processed":len(changes),"total":len(changes),"resumable":True,"status":"complete"}}
341
473
  sync_dir=path(project).parent/"sync-runs"; sync_dir.mkdir(parents=True,exist_ok=True); (sync_dir/f"{run_id}.json").write_text(json.dumps(report,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); (path(project).parent/"sync-checkpoint.json").write_text(json.dumps(report["checkpoint"]|{"runId":run_id,"sourceUrl":args.source},indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); (path(project).parent/"last-sync.json").write_text(json.dumps(report,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); print(json.dumps(report,indent=2,ensure_ascii=False)); return 0 if not validate(fresh) else 1
474
+
475
+ def cmd_sync_report(args):
476
+ project=root(args); directory=path(project).parent/"sync-runs"
477
+ candidates=sorted(directory.glob("*.json")) if directory.exists() else []
478
+ if args.run_id:
479
+ candidates=[directory/f"{args.run_id}.json"]
480
+ if not candidates or not candidates[-1].exists():
481
+ print(json.dumps({"status":"failed","error":"sync report not found"},indent=2)); return 1
482
+ try:
483
+ report=json.loads(candidates[-1].read_text(encoding="utf-8"))
484
+ except (OSError, ValueError):
485
+ print(json.dumps({"status":"failed","error":"sync report is invalid"},indent=2)); return 1
486
+ if report.get("schemaVersion") != "maggie-service-sync-report.v1":
487
+ print(json.dumps({"status":"failed","error":"unsupported sync report schema"},indent=2)); return 1
488
+ if args.output:
489
+ output=Path(args.output).expanduser(); output=output if output.is_absolute() else project/output; output.parent.mkdir(parents=True,exist_ok=True); output.write_text(json.dumps(report,indent=2,ensure_ascii=False)+"\n",encoding="utf-8")
490
+ print(json.dumps(report,indent=2,ensure_ascii=False)); return 0
342
491
  def cmd_generate(args):
343
492
  print(json.dumps({"status":"blocked","error":"Public copy generation is agent-owned","required":"The AI agent must create and review the copy artifact, then the project renderer must consume it. This deterministic CLI never synthesizes public hero, section, CTA, title, or FAQ copy.","copyData":getattr(args, "copy_data", None)},indent=2)); return 1
344
493
  project=root(args); data=load(project); errors=validate(data)
@@ -1004,6 +1153,7 @@ def main():
1004
1153
  for name in ("import","sync","run"):
1005
1154
  q=sub.add_parser(name); q.add_argument("source"); q.add_argument("--project",default="."); q.add_argument("--provider",default="fresha",choices=["fresha"])
1006
1155
  if name == "run": q.add_argument("--copy-data", help="AI-authored service copy artifact; required before public page generation")
1156
+ q=sub.add_parser("sync-report"); q.add_argument("--project",default="."); q.add_argument("--run-id"); q.add_argument("--output")
1007
1157
  q=sub.add_parser("convert-page"); q.add_argument("page"); q.add_argument("--project",default=".")
1008
1158
  for name in ("polish","validate-polish"):
1009
1159
  q=sub.add_parser(name); q.add_argument("--service-id",required=True); q.add_argument("--page",required=True); q.add_argument("--project",default="."); q.add_argument("--rendered",help="rendered HTML file or URL for runtime evidence"); q.add_argument("--copy-data",required=True,help="AI-authored copy JSON with provenance and completed review statuses")
@@ -1013,6 +1163,8 @@ def main():
1013
1163
  q=sub.add_parser("category-editorial-review"); q.add_argument("--project",default="."); q.add_argument("--copy-data",default="docs/category-page-copy.json"); q.add_argument("--category",action="append",help="category to review; defaults to all seven launch categories"); q.add_argument("--reviewer"); q.add_argument("--evidence",action="append",default=[],help="review evidence path or URL; repeatable"); q.add_argument("--apply",action="store_true",help="write explicit editorial approval metadata"); q.add_argument("--confirm",action="store_true",help="confirm that the selected records were actually reviewed by a human")
1014
1164
  q=sub.add_parser("category-hash"); q.add_argument("--project",default="."); q.add_argument("--copy-data",required=True); q.add_argument("--apply",action="store_true",help="backfill deterministic provenance hashes in this generated artifact")
1015
1165
  q=sub.add_parser("fact-audit"); q.add_argument("--project",default="."); q.add_argument("--backfill-source",action="store_true",help="copy the catalogue sourceUrl into records that have no sourceUrl"); q.add_argument("--overrides",help="JSON of human-approved, evidenced descriptions for provider gaps"); q.add_argument("--apply-overrides",action="store_true",help="apply only approved fact overrides to the draft catalogue")
1166
+ q=sub.add_parser("catalogue-check"); q.add_argument("source"); q.add_argument("--project",default="."); q.add_argument("--provider",default="fresha",choices=["fresha"]); q.add_argument("--treatment",action="append",required=True,help="provider treatment name; repeat for multiple checks")
1167
+ q=sub.add_parser("retirement-audit"); q.add_argument("--project",default="."); q.add_argument("--catalogue",default=".maggie/booking/services.json"); q.add_argument("--evidence",required=True,help="sanitized runtime retirement evidence JSON"); q.add_argument("--output",default="docs/service-retirement-audit.json")
1016
1168
  q=sub.add_parser("capability-audit", help="validate provider variant declarations and fixture evidence"); q.add_argument("--project",default="."); q.add_argument("--provider"); q.add_argument("--catalogue",default=".maggie/booking/services.json"); q.add_argument("--capabilities-file",default=".maggie/booking/provider-capabilities.json"); q.add_argument("--fixture",help="sanitized provider fixture JSON"); q.add_argument("--output")
1017
1169
  q=sub.add_parser("category-context"); q.add_argument("--project",default="."); q.add_argument("--output")
1018
1170
  q=sub.add_parser("validate-copy"); q.add_argument("--project",default="."); q.add_argument("--service-id",required=True); q.add_argument("--copy-data",required=True,help="AI-authored service copy JSON")
@@ -1026,6 +1178,7 @@ def main():
1026
1178
  a=p.parse_args(); project=root(a)
1027
1179
  if a.command=="import": return cmd_import(a)
1028
1180
  if a.command=="sync": return cmd_sync(a)
1181
+ if a.command=="sync-report": return cmd_sync_report(a)
1029
1182
  if a.command=="run": return cmd_run(a)
1030
1183
  if a.command=="convert-page": return cmd_convert_page(a)
1031
1184
  if a.command in {"polish","validate-polish"}: return cmd_polish(a)
@@ -1035,6 +1188,8 @@ def main():
1035
1188
  if a.command=="category-editorial-review": return cmd_category_editorial_review(a)
1036
1189
  if a.command=="category-hash": return cmd_category_hash(a)
1037
1190
  if a.command=="fact-audit": return cmd_fact_audit(a)
1191
+ if a.command=="catalogue-check": return cmd_catalogue_check(a)
1192
+ if a.command=="retirement-audit": return cmd_retirement_audit(a)
1038
1193
  if a.command=="capability-audit": return cmd_capability_audit(a)
1039
1194
  if a.command=="category-context": return cmd_category_context(a)
1040
1195
  if a.command=="validate-copy": return cmd_validate_copy(a)
@@ -2,6 +2,7 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ from collections import Counter
5
6
  import re
6
7
  from typing import Any, Iterable
7
8
 
@@ -17,6 +18,10 @@ MARKETS = {"global", "uk", "us"}
17
18
  STATUSES = {"missing", "draft", "machine_translated", "needs_review", "approved", "published", "archived"}
18
19
  OPERATIONS = {"translate", "polish", "rewrite", "localise", "rebrand", "manual_edit"}
19
20
  PROTECTED_FIELDS = {"price", "currency", "rating", "provider", "providerId", "bookingUrl", "paymentUrl", "legalClaims", "healthClaims", "contentId", "slug", "canonicalUrl", "translationGroupId"}
21
+ QUALITY_OPERATIONS = {"polish", "rewrite"}
22
+ QUALITY_META_FIELDS = {"identity", "contentId", "contentType", "sourceRevision", "translationStatus", "isIndexable", "provenance", "sourceStringIds", "qualityEvidence", "humanReviewStatus"}
23
+ FACT_TOKEN_PATTERN = re.compile(r"https?://[^\s<>]+|[\w.+-]+@[\w.-]+\.[A-Za-z]{2,}|(?:£|\$|€|¥|₹)\s*\d[\d,.]*|\b\d[\d,.]*%|\b\d[\d,.]*\b", re.UNICODE)
24
+ PLACEHOLDER_PATTERN = re.compile(r"\{\{[^}]+\}\}|\$\{[^}]+\}|\b(?:TODO|TBD|lorem ipsum)\b", re.IGNORECASE)
20
25
 
21
26
 
22
27
  class TranslationIndex:
@@ -162,3 +167,135 @@ def protected_field_changes(source: Any, translation: Any) -> list[str]:
162
167
  if not isinstance(source, dict) or not isinstance(translation, dict):
163
168
  return []
164
169
  return [field for field in sorted(PROTECTED_FIELDS) if field in source and field in translation and source[field] != translation[field]]
170
+
171
+
172
+ def _copy_payload(value: Any) -> Any:
173
+ """Remove identity/approval metadata before comparing editorial structure."""
174
+ if isinstance(value, dict):
175
+ return {
176
+ key: _copy_payload(item)
177
+ for key, item in value.items()
178
+ if key not in QUALITY_META_FIELDS and key not in PROTECTED_FIELDS
179
+ }
180
+ if isinstance(value, list):
181
+ return [_copy_payload(item) for item in value]
182
+ return value
183
+
184
+
185
+ def _copy_text(value: Any) -> str:
186
+ if isinstance(value, dict):
187
+ return "\n".join(part for part in (_copy_text(item) for item in value.values()) if part)
188
+ if isinstance(value, list):
189
+ return "\n".join(part for part in (_copy_text(item) for item in value) if part)
190
+ return str(value).strip() if isinstance(value, str) else ""
191
+
192
+
193
+ def _copy_metrics(text: str) -> dict[str, int | bool]:
194
+ normalized = text.strip()
195
+ sentences = [item for item in re.split(r"[.!?。!?]+", normalized) if item.strip()]
196
+ words = re.findall(r"\w+(?:['’\-]\w+)*", normalized, re.UNICODE)
197
+ sentence_lengths = [len(re.findall(r"\w+", item, re.UNICODE)) for item in sentences]
198
+ return {
199
+ "characters": len(normalized),
200
+ "words": len(words),
201
+ "sentences": len(sentences),
202
+ "paragraphs": len([item for item in re.split(r"\n\s*\n", normalized) if item.strip()]),
203
+ "headings": len(re.findall(r"(?im)^\s{0,3}(?:#{1,6}\s+|<h[1-6]\b)", normalized)),
204
+ "listItems": len(re.findall(r"(?im)^\s*(?:[-*+]\s+|\d+[.)]\s+|<li\b)", normalized)),
205
+ "maxSentenceWords": max(sentence_lengths, default=0),
206
+ "placeholders": len(PLACEHOLDER_PATTERN.findall(normalized)),
207
+ "repeatedWhitespace": len(re.findall(r"[ \t]{2,}", normalized)),
208
+ }
209
+
210
+
211
+ def _fact_tokens(text: str) -> Counter[str]:
212
+ return Counter(token.casefold().rstrip(".,;:)") for token in FACT_TOKEN_PATTERN.findall(text))
213
+
214
+
215
+ def _structure_signature(value: Any) -> tuple[Any, ...]:
216
+ if isinstance(value, dict):
217
+ return ("object", tuple((key, _structure_signature(item)) for key, item in sorted(value.items())))
218
+ if isinstance(value, list):
219
+ return ("list", len(value), tuple(_structure_signature(item) for item in value))
220
+ if isinstance(value, str):
221
+ metrics = _copy_metrics(value)
222
+ return ("text", metrics["paragraphs"], metrics["headings"], metrics["listItems"], metrics["sentences"])
223
+ if value is None:
224
+ return ("null",)
225
+ return (type(value).__name__,)
226
+
227
+
228
+ def validate_operation_quality(source: Any, translation: Any, operation: str, review: Any = None) -> tuple[list[str], dict[str, Any]]:
229
+ """Run conservative, dependency-free checks for polish and rewrite jobs.
230
+
231
+ These checks cover observable copy structure and protected fact tokens. They
232
+ intentionally do not claim semantic equivalence; every applicable result
233
+ carries a mandatory human-review signal.
234
+ """
235
+ report: dict[str, Any] = {
236
+ "operation": operation,
237
+ "applicable": operation in QUALITY_OPERATIONS,
238
+ "method": "deterministic-copy-heuristics",
239
+ }
240
+ if operation not in QUALITY_OPERATIONS:
241
+ return [], report
242
+ source_copy = _copy_payload(source)
243
+ translation_copy = _copy_payload(translation)
244
+ source_text = _copy_text(source_copy)
245
+ translation_text = _copy_text(translation_copy)
246
+ source_metrics = _copy_metrics(source_text)
247
+ translation_metrics = _copy_metrics(translation_text)
248
+ source_facts = _fact_tokens(source_text)
249
+ translation_facts = _fact_tokens(translation_text)
250
+ changed_fact_count = sum((source_facts - translation_facts).values()) + sum((translation_facts - source_facts).values())
251
+ normalized_source = re.sub(r"\s+", " ", source_text).strip().casefold()
252
+ normalized_translation = re.sub(r"\s+", " ", translation_text).strip().casefold()
253
+ structure_changed = _structure_signature(source_copy) != _structure_signature(translation_copy)
254
+ clarity_signal = (
255
+ translation_metrics["maxSentenceWords"] < source_metrics["maxSentenceWords"]
256
+ or translation_metrics["sentences"] > source_metrics["sentences"]
257
+ or translation_metrics["repeatedWhitespace"] < source_metrics["repeatedWhitespace"]
258
+ or translation_metrics["placeholders"] < source_metrics["placeholders"]
259
+ )
260
+ declared_clarity = isinstance(translation, dict) and isinstance(translation.get("qualityEvidence"), dict) and translation["qualityEvidence"].get("clarityImproved") is True
261
+ human_review = review if isinstance(review, dict) else {}
262
+ review_approved = human_review.get("decision") == "approve" and bool(human_review.get("reviewer"))
263
+ errors: list[str] = []
264
+ if not source_text:
265
+ errors.append(f"{operation} quality requires source copy text")
266
+ if not translation_text:
267
+ errors.append(f"{operation} quality requires generated copy text")
268
+ if changed_fact_count:
269
+ errors.append(f"{operation} output changed {changed_fact_count} protected fact token(s)")
270
+ if operation == "polish":
271
+ if normalized_source and normalized_source == normalized_translation:
272
+ errors.append("polish output is unchanged")
273
+ if translation_metrics["placeholders"] > source_metrics["placeholders"]:
274
+ errors.append("polish output introduced unresolved placeholders")
275
+ if source_text and translation_text and not (clarity_signal or declared_clarity):
276
+ errors.append("polish output has no deterministic clarity-improvement signal")
277
+ if operation == "rewrite":
278
+ if normalized_source and normalized_source == normalized_translation:
279
+ errors.append("rewrite output is unchanged")
280
+ if source_text and translation_text and not structure_changed:
281
+ errors.append("rewrite output did not change copy structure")
282
+ report.update({
283
+ "source": source_metrics,
284
+ "output": translation_metrics,
285
+ "checks": {
286
+ "nonEmpty": bool(source_text and translation_text),
287
+ "changed": bool(normalized_source != normalized_translation),
288
+ "protectedFactsPreserved": changed_fact_count == 0,
289
+ "structureChanged": structure_changed,
290
+ "clarityImproved": bool(clarity_signal or declared_clarity) if operation == "polish" else None,
291
+ "clarityEvidence": "deterministic" if clarity_signal else "declared" if declared_clarity else "missing",
292
+ },
293
+ "humanReview": {
294
+ "required": True,
295
+ "status": "approved" if review_approved else "required",
296
+ "semanticEquivalence": "not-certified",
297
+ "reason": "heuristics do not replace human meaning review",
298
+ },
299
+ "status": "passed" if not errors else "failed",
300
+ })
301
+ return errors, report
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@topy-ai/maggie",
3
- "version": "0.7.20",
3
+ "version": "0.7.22",
4
4
  "description": "Install and manage Maggie Skills for AI coding agents",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -61,6 +61,15 @@ Supported operations are distinct: `translate`, `polish`, `rewrite`,
61
61
  source. A source revision change marks dependent translations stale. Fallback
62
62
  content is never indexable.
63
63
 
64
+ `polish` and `rewrite` validation includes deterministic output checks. Polish
65
+ must keep the source language, change the copy, preserve protected fact tokens,
66
+ and show a clarity signal (shorter maximum sentence, clearer sentence
67
+ segmentation, placeholder reduction, or explicit quality evidence). Rewrite
68
+ must change the copy and its observable structure, while preserving protected
69
+ fact tokens. These are conservative heuristics, not semantic equivalence
70
+ proof: every result contains `quality.humanReview.required: true` and marks
71
+ semantic equivalence as `not-certified` until a named reviewer approves it.
72
+
64
73
  The following are protected by default: price, currency, rating, provider
65
74
  facts, booking URL, legal/health claims, content ID, slug, canonical owner,
66
75
  and translation group. Changes require structured approval and evidence.
@@ -57,7 +57,7 @@ The stable model is deliberately small and provider-neutral:
57
57
  - **category**: `level1` and `level2`; a service may belong to more than one
58
58
  provider category through `additional.provider.categories`;
59
59
  - **service**: identity, slug, title, description, active/archived status,
60
- timestamps, and variants;
60
+ timestamps, supply/display state, and variants;
61
61
  - **variant**: identity, display name, duration, price, currency, discount or
62
62
  price range when the provider exposes them;
63
63
  - **booking actions**: booking URL, payment URL, and optional action metadata;
@@ -67,6 +67,44 @@ The stable model is deliberately small and provider-neutral:
67
67
  - **sync**: source, fetched time, content hash, parser version, and change
68
68
  status.
69
69
 
70
+ The portable lifecycle model keeps two owners separate:
71
+
72
+ - `supplyState`: provider sync state, either `live` or `withdrawn`;
73
+ - `displayState`: human publication decision, either `published`, `hidden`, or
74
+ `retired`.
75
+
76
+ Sync may change `supplyState` when the provider adds or removes a service, but
77
+ must preserve `displayState`. A withdrawn/published record is an explicit
78
+ interim state: keep the URL answerable, suppress booking and indexing, and
79
+ open a human retirement decision. The legacy `status` field may remain for
80
+ backwards compatibility, but it must not be used as both provider availability
81
+ and display policy.
82
+
83
+ The `slug` is site-owned. A provider rename may produce a `slugProposal` in a
84
+ sync report, but the stored slug remains unchanged until a person accepts the
85
+ proposal and the host writes the new slug plus its redirect in one transaction.
86
+
87
+ ## Withdrawal and retirement contract
88
+
89
+ Provider removal is not permission to delete a public route. The adapter keeps
90
+ the record with `supplyState: "withdrawn"` and the host chooses a reviewed
91
+ `retirement.ending`:
92
+
93
+ | Ending | Required host response | Required safety signals |
94
+ |---|---|---|
95
+ | `pending` | HTTP 200 interim page | unavailable notice, no booking action, `noindex`, absent from sitemap |
96
+ | `redirect` | HTTP 301 or 308 to a different same-site route | no booking action, absent from sitemap |
97
+ | `tombstone` | HTTP 200 unavailable page | unavailable notice, no booking action, `noindex`, absent from sitemap |
98
+ | `gone` | HTTP 410 | no booking action, absent from sitemap |
99
+
100
+ The host must preserve the old route long enough to apply its chosen ending,
101
+ and must not return a generic 404 as a substitute for the reviewed contract.
102
+ `maggie service retirement-audit` validates sanitized runtime evidence against
103
+ the catalogue and records only service IDs, ending decisions, statuses, and
104
+ safe pass/fail metadata. It does not fetch private routes, store response
105
+ bodies, or invent redirect targets. A person must review redirect destination,
106
+ copy, and accessibility before publication.
107
+
70
108
  The service page must render only canonical fields. `additional` is an
71
109
  explicit extension point for provider-specific or future fields and must be
72
110
  namespaced by concern (`provider`, `location`, `presentation`, `compliance`,
@@ -74,6 +112,21 @@ namespaced by concern (`provider`, `location`, `presentation`, `compliance`,
74
112
  nullable-safe, and must never override canonical fields. Unknown data stays in
75
113
  `additional`; it is not guessed into the stable model.
76
114
 
115
+ ## Source ownership and onboarding precedence
116
+
117
+ When a host onboarding flow collects a value directly from the merchant, that
118
+ explicit value is authoritative for the project. Provider research, imported
119
+ profiles, search results, and inferred metadata may fill an empty field only;
120
+ they must never replace a non-empty merchant entry. Persist the source of each
121
+ value when the host supports it, and show a conflict for human review when two
122
+ non-empty sources disagree. This rule applies to the website URL as well as
123
+ business name, address, phone, category, and booking-provider identity.
124
+
125
+ The shared service-booking parser does not implement a host's onboarding UI.
126
+ Adapters must enforce this precedence before passing project identity into
127
+ provider research or sync. A provider URL is evidence about the provider
128
+ catalogue, not permission to overwrite the merchant's website URL.
129
+
77
130
  Pages can also enter the model without a provider. A normal page conversion
78
131
  creates a `draft` service with `additional.conversion.sourcePage` and no
79
132
  invented price, duration, or booking URL. It becomes `active` only after the
@@ -104,6 +157,14 @@ silent conversion, idempotent add/update/archive synchronisation, source
104
157
  snapshots/checksums, import-run audit records and generated service pages with
105
158
  provider booking CTAs.
106
159
 
160
+ When a provider exposes an authoritative venue-owned catalogue, use that
161
+ adapter extraction for sync decisions and treatment-presence checks. Generic
162
+ JSON-LD, navigation links, and page-wide text can contain marketplace or other
163
+ business entities and are not proof that the connected venue still sells a
164
+ treatment. A read-only catalogue query should call the same parser as import
165
+ and sync so an operator can review a reproducible answer before applying a
166
+ withdrawal.
167
+
107
168
  It must not claim real-time availability, booking creation, cancellation sync
108
169
  or webhooks unless an official Fresha partner/API capability is available for
109
170
  that account. Public booking links remain the safe fallback. The paid Fresha