cctally 1.91.0 → 1.92.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -0
- package/README.md +4 -2
- package/bin/_cctally_cache.py +903 -74
- package/bin/_cctally_config.py +57 -0
- package/bin/_cctally_core.py +94 -14
- package/bin/_cctally_dashboard.py +217 -19
- package/bin/_cctally_dashboard_conversation.py +170 -20
- package/bin/_cctally_dashboard_envelope.py +2 -0
- package/bin/_cctally_db.py +481 -19
- package/bin/_cctally_doctor.py +18 -1
- package/bin/_cctally_journal.py +1156 -21
- package/bin/_cctally_journal_repair.py +6 -0
- package/bin/_cctally_parser.py +26 -0
- package/bin/_cctally_quota.py +171 -55
- package/bin/_cctally_record.py +13 -1
- package/bin/_cctally_rederive.py +4 -0
- package/bin/_cctally_statusline.py +6 -6
- package/bin/_cctally_store.py +1061 -40
- package/bin/_cctally_transcript.py +32 -2
- package/bin/_cctally_tui.py +54 -6
- package/bin/_lib_cache_report.py +8 -3
- package/bin/_lib_codex_conversation.py +851 -81
- package/bin/_lib_codex_conversation_query.py +2031 -96
- package/bin/_lib_codex_find_projection.py +517 -0
- package/bin/_lib_codex_harness_preamble.py +176 -0
- package/bin/_lib_codex_hooks.py +5 -3
- package/bin/_lib_codex_js_scan.py +254 -0
- package/bin/_lib_codex_landmarks.py +309 -0
- package/bin/_lib_codex_title_clean.py +116 -0
- package/bin/_lib_conversation_dispatch.py +168 -22
- package/bin/_lib_conversation_query.py +62 -2
- package/bin/_lib_conversation_watch.py +4 -2
- package/bin/_lib_doctor.py +64 -0
- package/bin/_lib_quota_alert_axes.py +31 -34
- package/bin/_lib_stats_damage.py +523 -0
- package/bin/_lib_stats_publish.py +243 -0
- package/bin/cctally +17 -3
- package/dashboard/static/assets/index-Dat-mza6.js +97 -0
- package/dashboard/static/assets/{index-Dwirao3Y.css → index-DnWdv8um.css} +1 -1
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +8 -1
- package/dashboard/static/assets/index-CILAoEja.js +0 -90
package/bin/_cctally_store.py
CHANGED
|
@@ -290,7 +290,11 @@ def would_block_prod_stats_cutover(path) -> bool:
|
|
|
290
290
|
# verification adds `quota_projection_ledger_state.last_full_pass_at`, and a
|
|
291
291
|
# legacy index that already ran the fixups at version 2 would skip the schema
|
|
292
292
|
# apply that adds it.
|
|
293
|
-
|
|
293
|
+
#
|
|
294
|
+
# 3 -> 4 (#460): scheduled boundary ownership adds
|
|
295
|
+
# `quota_projection_ledger_state.next_evaluation_by_root_json` through the same
|
|
296
|
+
# in-place legacy cutover seam.
|
|
297
|
+
_STATS_OPEN_FIXUPS_VERSION = 4
|
|
294
298
|
|
|
295
299
|
|
|
296
300
|
def stats_open_fixups_current(conn: sqlite3.Connection) -> bool:
|
|
@@ -784,6 +788,254 @@ def _stats_repair_marker(db_path) -> pathlib.Path:
|
|
|
784
788
|
return pathlib.Path(db_path).with_name("stats.db.repairing")
|
|
785
789
|
|
|
786
790
|
|
|
791
|
+
def _stats_publication_marker(db_path) -> pathlib.Path:
|
|
792
|
+
"""The durable publication marker for ``db_path`` (#496 S1 F1)."""
|
|
793
|
+
return pathlib.Path(str(db_path) + ".publication")
|
|
794
|
+
|
|
795
|
+
|
|
796
|
+
def _remove_stats_publication_marker(db_path) -> None:
|
|
797
|
+
import _cctally_journal
|
|
798
|
+
|
|
799
|
+
try:
|
|
800
|
+
_stats_publication_marker(db_path).unlink()
|
|
801
|
+
except FileNotFoundError:
|
|
802
|
+
pass
|
|
803
|
+
_cctally_journal._fsync_dir(pathlib.Path(db_path).parent)
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def _read_stats_publication_marker(db_path) -> "dict | None":
|
|
807
|
+
"""The marker's state as a MAPPING, or None when no marker exists.
|
|
808
|
+
|
|
809
|
+
A marker that is present but unreadable, or whose bytes are valid JSON that
|
|
810
|
+
is not an object (`null`, `[]`), reads as an empty mapping: it exists, and
|
|
811
|
+
it records nothing. `json.loads` returns whatever the bytes decode to, so
|
|
812
|
+
calling `.get(...)` on the raw result raises `AttributeError` on those
|
|
813
|
+
shapes — which, from `_raise_settled_publication_failure`, escapes the heal
|
|
814
|
+
hook's `except Exception` and surfaces as a raw traceback from `open_db`.
|
|
815
|
+
"""
|
|
816
|
+
try:
|
|
817
|
+
state = json.loads(_stats_publication_marker(db_path).read_text())
|
|
818
|
+
except FileNotFoundError:
|
|
819
|
+
return None
|
|
820
|
+
except (OSError, ValueError):
|
|
821
|
+
return {}
|
|
822
|
+
return state if isinstance(state, dict) else {}
|
|
823
|
+
|
|
824
|
+
|
|
825
|
+
def _discard_pending_stats_publication_marker(db_path) -> None:
|
|
826
|
+
"""Drop a PENDING marker whose own publication never replaced anything.
|
|
827
|
+
|
|
828
|
+
Such a marker's pinned high-water describes an index that was never
|
|
829
|
+
published, so validating the live destination against it would condemn a
|
|
830
|
+
healthy index (measured: destination cursor 303 against a pinned
|
|
831
|
+
high-water of 2852).
|
|
832
|
+
|
|
833
|
+
Callers must establish that fact first — see
|
|
834
|
+
`_pending_stats_publication_never_replaced`. **The maintenance lock does
|
|
835
|
+
NOT establish it.** A rebuild that dies releases its flock, so a scratch on
|
|
836
|
+
disk can belong to a strictly later run than the marker beside it, and the
|
|
837
|
+
two are then unrelated.
|
|
838
|
+
|
|
839
|
+
A `priorFailure` block is restored rather than dropped. It records a verdict
|
|
840
|
+
a PREVIOUS publication owed on bytes that are still live, carried forward by
|
|
841
|
+
the run whose marker this is; because that run never replaced the
|
|
842
|
+
destination, those bytes are exactly what a later opener would connect to.
|
|
843
|
+
|
|
844
|
+
A `failed` marker is a settled verdict about the CURRENT destination and is
|
|
845
|
+
never discarded here.
|
|
846
|
+
"""
|
|
847
|
+
state = _read_stats_publication_marker(db_path)
|
|
848
|
+
if state is None or str(state.get("status") or "") != "pending":
|
|
849
|
+
return
|
|
850
|
+
prior = state.get("priorFailure")
|
|
851
|
+
if isinstance(prior, dict) and prior:
|
|
852
|
+
import _cctally_journal
|
|
853
|
+
|
|
854
|
+
_cctally_db._atomic_write_private_json(
|
|
855
|
+
_stats_publication_marker(db_path), prior
|
|
856
|
+
)
|
|
857
|
+
_cctally_journal._fsync_dir(pathlib.Path(db_path).parent)
|
|
858
|
+
return
|
|
859
|
+
_remove_stats_publication_marker(db_path)
|
|
860
|
+
|
|
861
|
+
|
|
862
|
+
def _pending_stats_publication_never_replaced(db_path) -> bool:
|
|
863
|
+
"""Whether a PENDING marker's own publication never became live.
|
|
864
|
+
|
|
865
|
+
The marker STATES which protocol it belongs to, so the discriminator is
|
|
866
|
+
selected rather than inferred (#496 S3 §5), and the two never generalize
|
|
867
|
+
over each other.
|
|
868
|
+
|
|
869
|
+
**Physical replacement** answers with the scratch. `os.replace` is the only
|
|
870
|
+
thing that consumes ANOTHER run's scratch, so a marker still naming a live
|
|
871
|
+
scratch pathname proves that run never reached publication: the live
|
|
872
|
+
destination is the untouched predecessor and the marker is stale. When the
|
|
873
|
+
scratch is gone the replacement DID happen and the verdict on the published
|
|
874
|
+
bytes is still owed, so the marker must be resolved rather than discarded.
|
|
875
|
+
|
|
876
|
+
The stronger form of that claim — that `os.replace` is the only consumer of
|
|
877
|
+
any scratch — is false, and the difference is confined to the run's OWN
|
|
878
|
+
process. `_cctally_journal._cleanup_new_correction_scratches` removes the
|
|
879
|
+
scratch this run just created when `_recover_completed_correction`'s
|
|
880
|
+
rebuild raises, so a failing `os.replace` in that path leaves the marker
|
|
881
|
+
naming a scratch that its own process then deleted, and this predicate
|
|
882
|
+
reads that as "replaced". The proxy is used only across processes, where
|
|
883
|
+
that cleanup cannot reach, so the weaker property is the one it needs.
|
|
884
|
+
|
|
885
|
+
A marker carrying no `scratchPath` cannot prove it published, so it is
|
|
886
|
+
treated as never-replaced. No released binary has ever written one — the
|
|
887
|
+
marker and this field ship together — so the branch exists only to keep an
|
|
888
|
+
unreadable marker from wedging every open.
|
|
889
|
+
|
|
890
|
+
**In-place publication** answers with the publication's own stamp, because
|
|
891
|
+
it attaches the scratch read-only and the scratch survives commit and
|
|
892
|
+
rollback identically. Only `PROVEN_PREDECESSOR` discards; `INDETERMINATE`
|
|
893
|
+
fails closed and the marker is resolved instead.
|
|
894
|
+
|
|
895
|
+
Must be consulted BEFORE stale-artifact cleanup removes the scratch. It is
|
|
896
|
+
also what makes artifact-first recovery stamp-aware: a scratch surviving a
|
|
897
|
+
COMMITTED in-place publish is a spent artifact beside an owed verdict, not
|
|
898
|
+
an interrupted rebuild.
|
|
899
|
+
"""
|
|
900
|
+
state = _read_stats_publication_marker(db_path)
|
|
901
|
+
if not state:
|
|
902
|
+
return True
|
|
903
|
+
if str(state.get("status") or "") != "pending":
|
|
904
|
+
return False
|
|
905
|
+
if str(state.get("mechanism") or "replace") == "in_place":
|
|
906
|
+
import _cctally_journal
|
|
907
|
+
|
|
908
|
+
return _cctally_journal.in_place_publication_proven_predecessor(
|
|
909
|
+
db_path, state
|
|
910
|
+
)
|
|
911
|
+
scratch = state.get("scratchPath")
|
|
912
|
+
if not isinstance(scratch, str) or not scratch:
|
|
913
|
+
return True
|
|
914
|
+
return pathlib.Path(scratch).exists()
|
|
915
|
+
|
|
916
|
+
|
|
917
|
+
def _stats_publication_failed_error(
|
|
918
|
+
db_path, record_path, mechanism=None,
|
|
919
|
+
) -> _cctally_db.StatsPublicationFailedError:
|
|
920
|
+
"""The guided error for a settled publication failure.
|
|
921
|
+
|
|
922
|
+
The wording is selected by the mechanism the marker RECORDS, because the two
|
|
923
|
+
mechanisms leave different things on disk: physical replacement preserves
|
|
924
|
+
the damaged predecessor under `quarantine/`, and an in-place publication
|
|
925
|
+
preserves nothing at all. A marker written before the field existed reads as
|
|
926
|
+
`replace`, which is what those markers describe.
|
|
927
|
+
"""
|
|
928
|
+
template = (
|
|
929
|
+
_cctally_core.STATS_PUBLICATION_FAILED_IN_PLACE_MSG
|
|
930
|
+
if str(mechanism or "replace") == "in_place"
|
|
931
|
+
else _cctally_core.STATS_PUBLICATION_FAILED_MSG
|
|
932
|
+
)
|
|
933
|
+
return _cctally_db.StatsPublicationFailedError(
|
|
934
|
+
template.format(path=db_path, record=record_path or "<unrecorded>")
|
|
935
|
+
)
|
|
936
|
+
|
|
937
|
+
|
|
938
|
+
def _raise_settled_publication_failure(db_path) -> None:
|
|
939
|
+
"""Re-raise a `failed` publication verdict the caller's `except` swallowed.
|
|
940
|
+
|
|
941
|
+
Only a marker already written as `failed` reaches this; every other state
|
|
942
|
+
returns and leaves the caller's behaviour unchanged.
|
|
943
|
+
"""
|
|
944
|
+
state = _read_stats_publication_marker(db_path)
|
|
945
|
+
if not state:
|
|
946
|
+
return
|
|
947
|
+
if str(state.get("status") or "") == "failed":
|
|
948
|
+
raise _stats_publication_failed_error(
|
|
949
|
+
db_path, state.get("recordPath"), state.get("mechanism")
|
|
950
|
+
)
|
|
951
|
+
|
|
952
|
+
|
|
953
|
+
def _resolve_stats_publication_marker(db_path: pathlib.Path) -> None:
|
|
954
|
+
"""Honour a durable publication marker (#496 S1 F1).
|
|
955
|
+
|
|
956
|
+
Caller holds maintenance EXCLUSIVE, which is what makes the pending case
|
|
957
|
+
safe: a rebuild that is still in flight owns the lock, so reaching here
|
|
958
|
+
proves its outcome is settled.
|
|
959
|
+
|
|
960
|
+
Order of precedence is established by the caller: a `.rebuilding-*` scratch
|
|
961
|
+
artifact is classified FIRST, so the existing interrupted-rebuild recovery
|
|
962
|
+
keeps taking precedence and clears any stale marker when it republishes.
|
|
963
|
+
|
|
964
|
+
This function refuses and reports. It never decides to rebuild — choosing
|
|
965
|
+
when to rebuild is firing policy.
|
|
966
|
+
"""
|
|
967
|
+
import _cctally_journal
|
|
968
|
+
|
|
969
|
+
marker = _stats_publication_marker(db_path)
|
|
970
|
+
state = _read_stats_publication_marker(db_path)
|
|
971
|
+
if state is None:
|
|
972
|
+
return
|
|
973
|
+
status = str(state.get("status") or "")
|
|
974
|
+
record_path = state.get("recordPath")
|
|
975
|
+
mechanism = state.get("mechanism")
|
|
976
|
+
|
|
977
|
+
if status == "failed":
|
|
978
|
+
raise _stats_publication_failed_error(db_path, record_path, mechanism)
|
|
979
|
+
if status != "pending":
|
|
980
|
+
_remove_stats_publication_marker(db_path)
|
|
981
|
+
return
|
|
982
|
+
|
|
983
|
+
# The discriminator runs on EVERY path into this function, not only the
|
|
984
|
+
# one the opener reaches with a surviving `.rebuilding-*` family beside
|
|
985
|
+
# the marker. For a `replace` marker the two agree — a missing scratch
|
|
986
|
+
# proves `os.replace` ran, so this returns False and resolution proceeds
|
|
987
|
+
# exactly as before. For an `in_place` marker scratch absence proves
|
|
988
|
+
# NOTHING, and without this the record's pinned high-water would be
|
|
989
|
+
# validated against a generation that was never published: a publication
|
|
990
|
+
# the stamp shows never committed would condemn its own healthy
|
|
991
|
+
# predecessor and refuse every ordinary open. `INDETERMINATE` still fails
|
|
992
|
+
# closed, so an unreadable stamp resolves rather than discards.
|
|
993
|
+
if _pending_stats_publication_never_replaced(db_path):
|
|
994
|
+
_discard_pending_stats_publication_marker(db_path)
|
|
995
|
+
return
|
|
996
|
+
|
|
997
|
+
record = None
|
|
998
|
+
if isinstance(record_path, str):
|
|
999
|
+
try:
|
|
1000
|
+
record = json.loads(pathlib.Path(record_path).read_text())
|
|
1001
|
+
except (OSError, ValueError):
|
|
1002
|
+
record = None
|
|
1003
|
+
if not isinstance(record, dict) or "highWater" not in record:
|
|
1004
|
+
# Without its record the marker cannot be judged, and validating
|
|
1005
|
+
# against a guessed high-water would condemn a healthy index. This
|
|
1006
|
+
# marker is diagnostic scaffolding; it must not wedge every open.
|
|
1007
|
+
print(
|
|
1008
|
+
"[stats] discarding an unresolvable stats.db publication marker "
|
|
1009
|
+
f"(rebuild record: {record_path!r})",
|
|
1010
|
+
file=sys.stderr,
|
|
1011
|
+
)
|
|
1012
|
+
_remove_stats_publication_marker(db_path)
|
|
1013
|
+
return
|
|
1014
|
+
|
|
1015
|
+
raw = record.get("highWater")
|
|
1016
|
+
high_water = (
|
|
1017
|
+
(str(raw[0]), int(raw[1]))
|
|
1018
|
+
if isinstance(raw, (list, tuple)) and len(raw) == 2
|
|
1019
|
+
else None
|
|
1020
|
+
)
|
|
1021
|
+
error = _cctally_journal.validate_published_stats_index(db_path, high_water)
|
|
1022
|
+
if error is None:
|
|
1023
|
+
_remove_stats_publication_marker(db_path)
|
|
1024
|
+
return
|
|
1025
|
+
|
|
1026
|
+
state.update({"status": "failed", "error": error})
|
|
1027
|
+
try:
|
|
1028
|
+
_cctally_db._atomic_write_private_json(marker, state)
|
|
1029
|
+
record["status"] = "failed"
|
|
1030
|
+
record["postPublicationValidation"] = {"ok": False, "error": error}
|
|
1031
|
+
_cctally_db._atomic_write_private_json(
|
|
1032
|
+
pathlib.Path(record_path), record
|
|
1033
|
+
)
|
|
1034
|
+
except OSError:
|
|
1035
|
+
pass
|
|
1036
|
+
raise _stats_publication_failed_error(db_path, record_path, mechanism)
|
|
1037
|
+
|
|
1038
|
+
|
|
787
1039
|
def _resume_pending_quarantine(db_path: pathlib.Path) -> None:
|
|
788
1040
|
"""Finish a strict quarantine that a previous owner did not complete.
|
|
789
1041
|
|
|
@@ -993,12 +1245,27 @@ def _recover_or_reclaim_interrupted_stats_rebuild(
|
|
|
993
1245
|
# matching legacy prebuild-quarantine incident, exact scratch names are
|
|
994
1246
|
# unpublished Task A artifacts and are safe to reclaim under the
|
|
995
1247
|
# caller's maintenance EXCLUSIVE hold.
|
|
1248
|
+
#
|
|
1249
|
+
# The marker beside them is a separate question, decided BEFORE the
|
|
1250
|
+
# cleanup destroys the evidence: a scratch here need not belong to the
|
|
1251
|
+
# marker's run at all, because a crashed rebuild releases its flock and
|
|
1252
|
+
# a later run can leave its own scratch behind.
|
|
1253
|
+
stale_marker = _pending_stats_publication_never_replaced(db_path)
|
|
996
1254
|
_remove_stale_stats_rebuild_artifacts(artifacts)
|
|
1255
|
+
if stale_marker:
|
|
1256
|
+
_discard_pending_stats_publication_marker(db_path)
|
|
1257
|
+
else:
|
|
1258
|
+
_resolve_stats_publication_marker(db_path)
|
|
997
1259
|
return True
|
|
998
1260
|
if db_path.exists() and _cctally_journal.stats_index_matches_journal_prefix(
|
|
999
1261
|
db_path, high_water
|
|
1000
1262
|
):
|
|
1001
1263
|
_remove_stale_stats_rebuild_artifacts(artifacts)
|
|
1264
|
+
# This branch just PROVED the destination is a fully valid
|
|
1265
|
+
# materialization of the journal prefix, which is strictly stronger
|
|
1266
|
+
# than any publication marker's own check, so the proof supersedes
|
|
1267
|
+
# whatever the marker recorded.
|
|
1268
|
+
_remove_stats_publication_marker(db_path)
|
|
1002
1269
|
return True
|
|
1003
1270
|
if high_water is None or high_water[1] == 0:
|
|
1004
1271
|
return False
|
|
@@ -1010,8 +1277,14 @@ def _recover_or_reclaim_interrupted_stats_rebuild(
|
|
|
1010
1277
|
)
|
|
1011
1278
|
try:
|
|
1012
1279
|
with stats_write_scope("maintenance-interrupted-rebuild"):
|
|
1013
|
-
_cctally_journal.rebuild_stats_index(
|
|
1280
|
+
_cctally_journal.rebuild_stats_index(
|
|
1281
|
+
context=_cctally_journal.RebuildContext(
|
|
1282
|
+
trigger="interrupted-rebuild-recovery"
|
|
1283
|
+
),
|
|
1284
|
+
high_water=high_water,
|
|
1285
|
+
)
|
|
1014
1286
|
_remove_stale_stats_rebuild_artifacts(artifacts)
|
|
1287
|
+
_discard_pending_stats_publication_marker(db_path)
|
|
1015
1288
|
return True
|
|
1016
1289
|
finally:
|
|
1017
1290
|
_cctally_journal._release_ingest_lock(ingest_fd)
|
|
@@ -1102,6 +1375,13 @@ def stats_open_guarded(
|
|
|
1102
1375
|
raise _cctally_db.StatsDbMaintenanceError(
|
|
1103
1376
|
_STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
|
|
1104
1377
|
)
|
|
1378
|
+
# Recovery calls `rebuild_stats_index`, whose in-place publisher
|
|
1379
|
+
# reopens the live destination through `stats_open_guarded`.
|
|
1380
|
+
# `flock` conflicts are per open-file-DESCRIPTION and apply
|
|
1381
|
+
# WITHIN a process, so without this the nested SHARED request
|
|
1382
|
+
# would conflict with the EXCLUSIVE hold taken on the line
|
|
1383
|
+
# above and time out against the branch's own lock.
|
|
1384
|
+
_cctally_core.note_stats_maintenance_acquired()
|
|
1105
1385
|
recovered = False
|
|
1106
1386
|
try:
|
|
1107
1387
|
current_artifacts = _stats_rebuild_artifact_bases(db_path)
|
|
@@ -1114,6 +1394,11 @@ def stats_open_guarded(
|
|
|
1114
1394
|
except (
|
|
1115
1395
|
_cctally_db.ProdMigrationRefused,
|
|
1116
1396
|
_cctally_db.StatsDbMaintenanceError,
|
|
1397
|
+
# Recovery may resolve a publication marker whose run
|
|
1398
|
+
# DID replace the destination; that verdict carries its
|
|
1399
|
+
# own guided wording and must not be reworded into a
|
|
1400
|
+
# maintenance-in-progress error.
|
|
1401
|
+
_cctally_db.StatsPublicationFailedError,
|
|
1117
1402
|
):
|
|
1118
1403
|
raise
|
|
1119
1404
|
except Exception as exc:
|
|
@@ -1126,6 +1411,7 @@ def stats_open_guarded(
|
|
|
1126
1411
|
"`cctally db rebuild --db stats`."
|
|
1127
1412
|
) from exc
|
|
1128
1413
|
finally:
|
|
1414
|
+
_cctally_core.note_stats_maintenance_released()
|
|
1129
1415
|
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1130
1416
|
if recovered:
|
|
1131
1417
|
continue
|
|
@@ -1138,6 +1424,36 @@ def stats_open_guarded(
|
|
|
1138
1424
|
if marker.exists() or pending.exists():
|
|
1139
1425
|
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1140
1426
|
continue
|
|
1427
|
+
# #496 S1 F1: a durable publication marker, honoured AFTER the
|
|
1428
|
+
# scratch-artifact classification above so that path keeps
|
|
1429
|
+
# precedence. Steady state costs one stat() on a file that does not
|
|
1430
|
+
# exist. Suppressed exactly where interrupted recovery is, so
|
|
1431
|
+
# doctor's read-only gather stays read-only.
|
|
1432
|
+
if (
|
|
1433
|
+
recover_interruptions
|
|
1434
|
+
and _INTERRUPTED_RECOVERY_SUPPRESSED.get() == 0
|
|
1435
|
+
and _stats_publication_marker(db_path).exists()
|
|
1436
|
+
):
|
|
1437
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1438
|
+
if not _flock_bounded(
|
|
1439
|
+
lock_fh, fcntl.LOCK_EX, _STATS_OPEN_RESUME_WAIT_S
|
|
1440
|
+
):
|
|
1441
|
+
raise _cctally_db.StatsDbMaintenanceError(
|
|
1442
|
+
_STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
|
|
1443
|
+
)
|
|
1444
|
+
try:
|
|
1445
|
+
_resolve_stats_publication_marker(db_path)
|
|
1446
|
+
finally:
|
|
1447
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1448
|
+
if not _flock_bounded(
|
|
1449
|
+
lock_fh, fcntl.LOCK_SH, _STATS_OPEN_MAINTENANCE_WAIT_S
|
|
1450
|
+
):
|
|
1451
|
+
raise _cctally_db.StatsDbMaintenanceError(
|
|
1452
|
+
_STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
|
|
1453
|
+
)
|
|
1454
|
+
if marker.exists() or pending.exists():
|
|
1455
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1456
|
+
continue
|
|
1141
1457
|
try:
|
|
1142
1458
|
conn = _connect(db_path)
|
|
1143
1459
|
# Re-check inside the same shared hold: cheap, and it closes the
|
|
@@ -1179,9 +1495,11 @@ def _acquire_stats_maintenance_reentrant(path) -> "int | None":
|
|
|
1179
1495
|
and apply WITHIN a process: holding SHARED on one fd and then requesting
|
|
1180
1496
|
EXCLUSIVE on a second fd of the same file blocks the process against itself,
|
|
1181
1497
|
indefinitely. ``run_stats_ingest`` holds maintenance SHARED across its entire
|
|
1182
|
-
cycle, and
|
|
1183
|
-
|
|
1184
|
-
|
|
1498
|
+
cycle, and this helper's caller — the epoch resolver — is reachable from a
|
|
1499
|
+
nested ``open_db()`` inside that cycle. Without this check that nested open
|
|
1500
|
+
is an unconditional self-deadlock. The corruption heal applies the same
|
|
1501
|
+
ownership-first rule through ``_acquire_stats_maintenance_for_heal``, which
|
|
1502
|
+
additionally BOUNDS the acquire.
|
|
1185
1503
|
|
|
1186
1504
|
Proceeding on a shared hold is a deliberate, narrow weakening: the caller
|
|
1187
1505
|
still runs ``_stats_family_drained`` before any physical replacement, which
|
|
@@ -1231,6 +1549,54 @@ def _heal_release_maintenance_flock(fd: int) -> None:
|
|
|
1231
1549
|
_heal_release_flock(fd)
|
|
1232
1550
|
|
|
1233
1551
|
|
|
1552
|
+
_HEAL_MAINTENANCE_WAIT_S = 5.0
|
|
1553
|
+
|
|
1554
|
+
|
|
1555
|
+
def _acquire_stats_maintenance_for_heal(
|
|
1556
|
+
timeout_s: float = _HEAL_MAINTENANCE_WAIT_S,
|
|
1557
|
+
) -> "tuple[int | None, bool]":
|
|
1558
|
+
"""Ownership-first, mode-aware maintenance for the corruption heal (§6).
|
|
1559
|
+
|
|
1560
|
+
Returns ``(fd, True)`` when the heal may proceed — ``fd`` is ``None`` when
|
|
1561
|
+
an existing hold was REUSED and nothing was acquired — and ``(None,
|
|
1562
|
+
False)`` when the bounded acquire expired.
|
|
1563
|
+
|
|
1564
|
+
**Ownership-first, not mode-first.** ``flock`` conflicts are per open-file-
|
|
1565
|
+
description and apply WITHIN a process, so requesting the lock a second
|
|
1566
|
+
time on a second descriptor blocks this process against itself whenever the
|
|
1567
|
+
hold it already owns is EXCLUSIVE (`_cctally_core` documents that at the
|
|
1568
|
+
maintenance tracker, and ``run_stats_ingest`` can hold exclusive when it
|
|
1569
|
+
calls ``open_db()``). The tracker is a depth counter that records THAT a
|
|
1570
|
+
hold exists and never which mode, and it does not need to: the rule reuses
|
|
1571
|
+
any hold whatever its mode, so the two cases never have to be told apart.
|
|
1572
|
+
Upgrading a shared hold to exclusive is the one operation that would need
|
|
1573
|
+
the mode, and it is exactly the second acquire that deadlocks.
|
|
1574
|
+
|
|
1575
|
+
**Bounded, never blocking, when nothing is held.** The heal runs inside an
|
|
1576
|
+
ordinary open, and the detached worker owns maintenance EXCLUSIVE for the
|
|
1577
|
+
whole of its rebuild. An unbounded acquire here would make every statusline
|
|
1578
|
+
and dashboard open that meets corruption wait out that rebuild — the
|
|
1579
|
+
blocking this architecture exists to remove. A timeout means some OTHER
|
|
1580
|
+
holder owns it, and failing soft is correct: decline, and let a later open
|
|
1581
|
+
retry.
|
|
1582
|
+
"""
|
|
1583
|
+
if _cctally_core.holds_stats_maintenance():
|
|
1584
|
+
return (None, True)
|
|
1585
|
+
fd = _heal_flock_bounded(
|
|
1586
|
+
_cctally_core.STATS_LOCK_MAINTENANCE_PATH, timeout_s
|
|
1587
|
+
)
|
|
1588
|
+
if fd is None:
|
|
1589
|
+
return (None, False)
|
|
1590
|
+
_cctally_core.note_stats_maintenance_acquired()
|
|
1591
|
+
return (fd, True)
|
|
1592
|
+
|
|
1593
|
+
|
|
1594
|
+
def _release_stats_maintenance_for_heal(fd: "int | None") -> None:
|
|
1595
|
+
"""Release what ``_acquire_stats_maintenance_for_heal`` took, if anything."""
|
|
1596
|
+
if fd is not None:
|
|
1597
|
+
_heal_release_maintenance_flock(fd)
|
|
1598
|
+
|
|
1599
|
+
|
|
1234
1600
|
def _heal_flock_bounded(path, timeout_s: float) -> "int | None":
|
|
1235
1601
|
"""Bounded EX flock. Returns the HELD fd, or ``None`` on timeout.
|
|
1236
1602
|
|
|
@@ -1370,7 +1736,15 @@ def _stats_heal_hook(
|
|
|
1370
1736
|
the maintenance lock; False when it DECLINES — a non-corruption
|
|
1371
1737
|
``DatabaseError`` (BUSY / disk-full / permission), the dev-checkout-on-prod
|
|
1372
1738
|
guard, or re-entrancy. A False return leaves ``open_db`` to raise its guided
|
|
1373
|
-
``StatsDbCorruptError``.
|
|
1739
|
+
``StatsDbCorruptError``.
|
|
1740
|
+
|
|
1741
|
+
It can also RAISE ``_cctally_db.StatsPublicationFailedError`` (#496 S1 F1),
|
|
1742
|
+
and both callers depend on that: ``_cctally_tui._tui_heal_post_query_stats``
|
|
1743
|
+
catches it and degrades, while ``_cctally_core.open_db`` deliberately lets
|
|
1744
|
+
it propagate. The heal that replaced the index and then failed to validate
|
|
1745
|
+
it must report that itself, because ``open_db``'s decline branch would tell
|
|
1746
|
+
the user the database was "Not auto-recreated" — false once replacement has
|
|
1747
|
+
occurred."""
|
|
1374
1748
|
global _HEAL_ACTIVE
|
|
1375
1749
|
if store != "stats":
|
|
1376
1750
|
return False
|
|
@@ -1396,49 +1770,130 @@ def _stats_heal_hook(
|
|
|
1396
1770
|
return False
|
|
1397
1771
|
_HEAL_ACTIVE = True
|
|
1398
1772
|
try:
|
|
1399
|
-
|
|
1400
|
-
|
|
1773
|
+
# Ownership-first and BOUNDED (#496 S3 §6). A hold this context already
|
|
1774
|
+
# owns is reused whatever its mode; otherwise exclusive is acquired
|
|
1775
|
+
# within a bound, because the detached worker owns maintenance for the
|
|
1776
|
+
# whole of its rebuild and an ordinary open must never wait that out.
|
|
1777
|
+
maint_fd, acquired = _acquire_stats_maintenance_for_heal()
|
|
1778
|
+
if not acquired:
|
|
1779
|
+
probe = _probe_stats_integrity_ok if post_query else _probe_stats_ok
|
|
1780
|
+
if probe(path):
|
|
1781
|
+
# Some other maintenance owner already republished a readable
|
|
1782
|
+
# index while we waited — retry the open rather than decline.
|
|
1783
|
+
return True
|
|
1784
|
+
print(
|
|
1785
|
+
"[heal] stats.db auto-heal declined: another maintenance "
|
|
1786
|
+
"owner holds stats.db.maintenance.lock; a later open will "
|
|
1787
|
+
"retry.",
|
|
1788
|
+
file=sys.stderr,
|
|
1789
|
+
)
|
|
1790
|
+
return False
|
|
1401
1791
|
try:
|
|
1402
1792
|
probe = _probe_stats_integrity_ok if post_query else _probe_stats_ok
|
|
1403
1793
|
if probe(path):
|
|
1404
1794
|
return True # a sibling process already healed it — retry the open
|
|
1405
|
-
# Forensics FIRST — before anything disturbs the evidence.
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
|
|
1416
|
-
|
|
1417
|
-
|
|
1418
|
-
|
|
1419
|
-
|
|
1420
|
-
|
|
1421
|
-
|
|
1422
|
-
|
|
1423
|
-
|
|
1424
|
-
|
|
1425
|
-
|
|
1426
|
-
|
|
1427
|
-
|
|
1428
|
-
|
|
1429
|
-
|
|
1430
|
-
|
|
1795
|
+
# Forensics FIRST — before anything disturbs the evidence. The
|
|
1796
|
+
# trigger pair is what arms the #496 S1 forensics-time WAL capture
|
|
1797
|
+
# and what lets the quarantine incident name the bundle that
|
|
1798
|
+
# preceded it.
|
|
1799
|
+
#
|
|
1800
|
+
forensics = _cctally_db.write_corruption_forensics(
|
|
1801
|
+
path,
|
|
1802
|
+
db_label="stats",
|
|
1803
|
+
trigger_origin="corruption-heal",
|
|
1804
|
+
trigger_exception=exc,
|
|
1805
|
+
return_result=True,
|
|
1806
|
+
)
|
|
1807
|
+
request = _build_stats_heal_request(
|
|
1808
|
+
exc, forensics, post_query=post_query, high_water=hw,
|
|
1809
|
+
)
|
|
1810
|
+
# F4, first point (#496 S3 §7). Classifier gating stays a
|
|
1811
|
+
# PRECONDITION; this narrows within classified triggers exactly as
|
|
1812
|
+
# the cache path does at `_cctally_cache.py`. A disposition other
|
|
1813
|
+
# than CONFIRMED declines: no deferral, no worker, no replacement,
|
|
1814
|
+
# and a printed reason naming the bundle.
|
|
1815
|
+
confirmed = (
|
|
1816
|
+
forensics is not None
|
|
1817
|
+
and forensics.disposition
|
|
1818
|
+
is _cctally_db.CorruptionProbeDisposition.CONFIRMED
|
|
1819
|
+
and forensics.path is not None
|
|
1820
|
+
)
|
|
1821
|
+
if not confirmed:
|
|
1822
|
+
bundle = (
|
|
1823
|
+
str(forensics.path)
|
|
1824
|
+
if forensics is not None and forensics.path is not None
|
|
1825
|
+
else "unavailable"
|
|
1826
|
+
)
|
|
1827
|
+
reason = (
|
|
1828
|
+
forensics.reason if forensics is not None else "unavailable"
|
|
1829
|
+
)
|
|
1830
|
+
append_stats_heal_event({
|
|
1831
|
+
**build_stats_heal_event(request, "unconfirmed"),
|
|
1832
|
+
"outcome": "declined-unconfirmed",
|
|
1833
|
+
"declineReason": reason,
|
|
1834
|
+
})
|
|
1835
|
+
print(
|
|
1836
|
+
"[heal] stats.db auto-heal declined for classified "
|
|
1837
|
+
f"trigger: corruption was not confirmed ({reason}; "
|
|
1838
|
+
f"forensics: {bundle}); leaving the stats.db file family "
|
|
1839
|
+
"untouched.",
|
|
1840
|
+
file=sys.stderr,
|
|
1841
|
+
)
|
|
1842
|
+
return False
|
|
1843
|
+
append_stats_heal_event(build_stats_heal_event(request, "confirmed"))
|
|
1844
|
+
finally:
|
|
1845
|
+
# Released BEFORE deferring: the worker takes maintenance
|
|
1846
|
+
# EXCLUSIVE as a fresh process holding nothing, and a caller still
|
|
1847
|
+
# holding it here would make that acquire wait for a request it is
|
|
1848
|
+
# itself in the middle of filing.
|
|
1849
|
+
_release_stats_maintenance_for_heal(maint_fd)
|
|
1850
|
+
outcome = defer_stats_corruption_heal(request)
|
|
1851
|
+
# F15 (#496 S3 §7). Detachment supplies the timing for free: report at
|
|
1852
|
+
# DETECTION, naming the absolute forensics path and the heal id. It
|
|
1853
|
+
# cannot name an incident path, because the quarantine directory is
|
|
1854
|
+
# allocated only during preservation, after the worker has chosen
|
|
1855
|
+
# physical fallback and begun it; the worker adds that to the ring.
|
|
1856
|
+
bundle = request.get("forensicsPath") or "unavailable"
|
|
1857
|
+
print(
|
|
1858
|
+
f"[heal] stats.db is corrupt ({exc}); nothing was replaced by this "
|
|
1859
|
+
f"command. A rebuild from the journal was scheduled to run in the "
|
|
1860
|
+
f"background as heal {request['healId']}. Forensics: {bundle}.",
|
|
1861
|
+
file=sys.stderr,
|
|
1862
|
+
)
|
|
1863
|
+
# Escalation is REPORT-ONLY: no halt, and no throttle beyond the
|
|
1864
|
+
# admission marker's existing retry interval. Halting auto-heal after
|
|
1865
|
+
# N occurrences was considered and rejected (§3 Q4).
|
|
1866
|
+
recurrence = stats_heal_recurrence()
|
|
1867
|
+
if recurrence >= _STATS_HEAL_RECURRENCE_THRESHOLD:
|
|
1868
|
+
days = int(_STATS_HEAL_RECURRENCE_WINDOW_S // 86400)
|
|
1431
1869
|
print(
|
|
1432
|
-
f"[heal]
|
|
1433
|
-
"
|
|
1434
|
-
"
|
|
1870
|
+
f"[heal] this is a recurring stats.db corruption: "
|
|
1871
|
+
f"{recurrence} heals in the last {days} days. The heal still "
|
|
1872
|
+
f"runs; the bundles in {_cctally_core.LOG_DIR} and the events "
|
|
1873
|
+
f"in {_stats_heal_ring_path()} are the evidence to report.",
|
|
1435
1874
|
file=sys.stderr,
|
|
1436
1875
|
)
|
|
1437
|
-
|
|
1438
|
-
|
|
1439
|
-
|
|
1876
|
+
# The heal no longer runs on the caller's thread, so it no longer has
|
|
1877
|
+
# a boolean to return. The signal derives from `BaseException` for the
|
|
1878
|
+
# reason `StatsRebuildDeferred` records: a broad `except Exception`
|
|
1879
|
+
# fallback would turn "the index is being rebuilt" into a misleading
|
|
1880
|
+
# partial report.
|
|
1881
|
+
raise _cctally_db.StatsHealDeferred(
|
|
1882
|
+
outcome,
|
|
1883
|
+
heal_id=request["healId"],
|
|
1884
|
+
forensics_path=request.get("forensicsPath"),
|
|
1885
|
+
)
|
|
1440
1886
|
except Exception as heal_exc:
|
|
1441
1887
|
print(f"[heal] stats.db auto-heal failed: {heal_exc}", file=sys.stderr)
|
|
1888
|
+
# A post-publication validation failure has ALREADY replaced the index.
|
|
1889
|
+
# Declining here sends `open_db` to its pre-existing branch, which tells
|
|
1890
|
+
# the user the database was "Not auto-recreated" and to run
|
|
1891
|
+
# `db repair --db stats --yes` — both false once replacement occurred.
|
|
1892
|
+
# The durable marker makes the NEXT process say the right thing; the
|
|
1893
|
+
# process that caused the failure must say it too (#496 S1 F1).
|
|
1894
|
+
# Narrow by construction: it fires only on a `failed` marker, so every
|
|
1895
|
+
# other heal failure keeps its existing behaviour.
|
|
1896
|
+
_raise_settled_publication_failure(path)
|
|
1442
1897
|
return False
|
|
1443
1898
|
finally:
|
|
1444
1899
|
_HEAL_ACTIVE = False
|
|
@@ -1447,6 +1902,572 @@ def _stats_heal_hook(
|
|
|
1447
1902
|
HEAL_HOOK = _stats_heal_hook
|
|
1448
1903
|
|
|
1449
1904
|
|
|
1905
|
+
# --------------------------------------------------------------------------
|
|
1906
|
+
# #496 S3 §6 — the detached corruption heal
|
|
1907
|
+
# --------------------------------------------------------------------------
|
|
1908
|
+
#
|
|
1909
|
+
# The hook writes forensics and files a REQUEST; a detached worker does the
|
|
1910
|
+
# rebuild. Admission copies the three layers of `defer_stats_epoch_rebuild` — a
|
|
1911
|
+
# non-blocking admission flock whose loser returns immediately, a pending
|
|
1912
|
+
# marker with a retry window, and a worker-active probe that refreshes the
|
|
1913
|
+
# marker instead of spawning a duplicate — over its OWN files, so the two
|
|
1914
|
+
# deferrals can never suppress each other.
|
|
1915
|
+
#
|
|
1916
|
+
# The epoch path's marker is an empty touched file. This one is a durable JSON
|
|
1917
|
+
# document, because the worker runs later and in another process and needs
|
|
1918
|
+
# facts the hook established at detection: the heal id that correlates the
|
|
1919
|
+
# durable event record, the trigger evidence, the forensics bundle, the
|
|
1920
|
+
# journal information to revalidate, and — load-bearing — WHICH PROBE to run.
|
|
1921
|
+
# A `post_query` detection was established by a failed `quick_check` against a
|
|
1922
|
+
# file SQLite opens happily, so a worker that always used the cheap readability
|
|
1923
|
+
# probe would exit on exactly the readable-but-corrupt population this
|
|
1924
|
+
# architecture exists to serve.
|
|
1925
|
+
|
|
1926
|
+
STATS_CORRUPTION_HEAL_COMMAND = "_stats-corruption-heal"
|
|
1927
|
+
_STATS_HEAL_RETRY_SECONDS = 60.0
|
|
1928
|
+
_STATS_HEAL_WORKER_MAINTENANCE_WAIT_S = 120.0
|
|
1929
|
+
_STATS_HEAL_PROBE_INTEGRITY = "integrity"
|
|
1930
|
+
_STATS_HEAL_PROBE_READABILITY = "readability"
|
|
1931
|
+
|
|
1932
|
+
|
|
1933
|
+
def _stats_heal_path(name: str) -> pathlib.Path:
|
|
1934
|
+
return pathlib.Path(_cctally_core.APP_DIR) / name
|
|
1935
|
+
|
|
1936
|
+
|
|
1937
|
+
def _stats_heal_marker_path() -> pathlib.Path:
|
|
1938
|
+
return _stats_heal_path("stats-corruption-heal.pending")
|
|
1939
|
+
|
|
1940
|
+
|
|
1941
|
+
def _stats_heal_admission_path() -> pathlib.Path:
|
|
1942
|
+
return _stats_heal_path("stats-corruption-heal.admission.lock")
|
|
1943
|
+
|
|
1944
|
+
|
|
1945
|
+
def _stats_heal_worker_path() -> pathlib.Path:
|
|
1946
|
+
return _stats_heal_path("stats-corruption-heal.worker.lock")
|
|
1947
|
+
|
|
1948
|
+
|
|
1949
|
+
def _stats_heal_log_path() -> pathlib.Path:
|
|
1950
|
+
return pathlib.Path(_cctally_core.LOG_DIR) / "stats-corruption-heal.log"
|
|
1951
|
+
|
|
1952
|
+
|
|
1953
|
+
def _new_heal_id() -> str:
|
|
1954
|
+
"""A collision-free correlation id readable in a log line."""
|
|
1955
|
+
return (
|
|
1956
|
+
dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
|
|
1957
|
+
+ "-"
|
|
1958
|
+
+ os.urandom(4).hex()
|
|
1959
|
+
)
|
|
1960
|
+
|
|
1961
|
+
|
|
1962
|
+
def _build_stats_heal_request(
|
|
1963
|
+
exc: BaseException,
|
|
1964
|
+
forensics,
|
|
1965
|
+
*,
|
|
1966
|
+
post_query: bool,
|
|
1967
|
+
high_water,
|
|
1968
|
+
) -> dict:
|
|
1969
|
+
"""The durable request the worker acts on (#496 S3 §6)."""
|
|
1970
|
+
return {
|
|
1971
|
+
"schemaVersion": 1,
|
|
1972
|
+
"healId": _new_heal_id(),
|
|
1973
|
+
"detectedAtUtc": _cctally_core.now_utc_iso(),
|
|
1974
|
+
"postQuery": bool(post_query),
|
|
1975
|
+
"probeKind": (
|
|
1976
|
+
_STATS_HEAL_PROBE_INTEGRITY
|
|
1977
|
+
if post_query
|
|
1978
|
+
else _STATS_HEAL_PROBE_READABILITY
|
|
1979
|
+
),
|
|
1980
|
+
"triggerError": _cctally_db._bounded_forensics_text(
|
|
1981
|
+
exc, _cctally_db._FORENSICS_EXCEPTION_MESSAGE_MAX
|
|
1982
|
+
),
|
|
1983
|
+
"triggerType": type(exc).__name__,
|
|
1984
|
+
"forensicsPath": (
|
|
1985
|
+
str(forensics.path)
|
|
1986
|
+
if forensics is not None and forensics.path is not None
|
|
1987
|
+
else None
|
|
1988
|
+
),
|
|
1989
|
+
"forensicsDisposition": (
|
|
1990
|
+
forensics.disposition.value if forensics is not None else None
|
|
1991
|
+
),
|
|
1992
|
+
"journalHighWater": (
|
|
1993
|
+
[str(high_water[0]), int(high_water[1])]
|
|
1994
|
+
if high_water is not None
|
|
1995
|
+
else None
|
|
1996
|
+
),
|
|
1997
|
+
}
|
|
1998
|
+
|
|
1999
|
+
|
|
2000
|
+
def _read_stats_heal_request() -> "dict | None":
|
|
2001
|
+
try:
|
|
2002
|
+
payload = json.loads(_stats_heal_marker_path().read_text())
|
|
2003
|
+
except FileNotFoundError:
|
|
2004
|
+
return None
|
|
2005
|
+
except (OSError, ValueError):
|
|
2006
|
+
return {}
|
|
2007
|
+
return payload if isinstance(payload, dict) else {}
|
|
2008
|
+
|
|
2009
|
+
|
|
2010
|
+
def _unlink_stats_heal_marker() -> None:
|
|
2011
|
+
try:
|
|
2012
|
+
_stats_heal_marker_path().unlink()
|
|
2013
|
+
except FileNotFoundError:
|
|
2014
|
+
pass
|
|
2015
|
+
|
|
2016
|
+
|
|
2017
|
+
def _stats_heal_worker_active() -> bool:
|
|
2018
|
+
"""Probe the worker flock without waiting or disturbing its owner."""
|
|
2019
|
+
try:
|
|
2020
|
+
fd = os.open(
|
|
2021
|
+
_stats_heal_worker_path(), os.O_WRONLY | os.O_CREAT, 0o600
|
|
2022
|
+
)
|
|
2023
|
+
except OSError:
|
|
2024
|
+
return False
|
|
2025
|
+
try:
|
|
2026
|
+
try:
|
|
2027
|
+
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
2028
|
+
except BlockingIOError:
|
|
2029
|
+
return True
|
|
2030
|
+
except OSError:
|
|
2031
|
+
return False
|
|
2032
|
+
try:
|
|
2033
|
+
fcntl.flock(fd, fcntl.LOCK_UN)
|
|
2034
|
+
except OSError:
|
|
2035
|
+
pass
|
|
2036
|
+
return False
|
|
2037
|
+
finally:
|
|
2038
|
+
os.close(fd)
|
|
2039
|
+
|
|
2040
|
+
|
|
2041
|
+
def _log_stats_heal(
|
|
2042
|
+
outcome: str, *, heal_id: "str | None" = None,
|
|
2043
|
+
error: BaseException | None = None,
|
|
2044
|
+
) -> None:
|
|
2045
|
+
"""Append one path-safe worker result line.
|
|
2046
|
+
|
|
2047
|
+
Follows `stats-epoch-rebuild.log`'s restraint for exception text: the class
|
|
2048
|
+
plus a numeric SQLite/OS code, never free-form message text that may carry
|
|
2049
|
+
private paths. The heal id is our own generated token and carries nothing.
|
|
2050
|
+
"""
|
|
2051
|
+
try:
|
|
2052
|
+
log_path = _stats_heal_log_path()
|
|
2053
|
+
log_path.parent.mkdir(parents=True, exist_ok=True)
|
|
2054
|
+
detail = ""
|
|
2055
|
+
if heal_id:
|
|
2056
|
+
detail += f" heal={heal_id}"
|
|
2057
|
+
if error is not None:
|
|
2058
|
+
code = getattr(error, "sqlite_errorcode", None)
|
|
2059
|
+
if code is None:
|
|
2060
|
+
code = getattr(error, "errno", None)
|
|
2061
|
+
detail += f" error={type(error).__name__}"
|
|
2062
|
+
if code is not None:
|
|
2063
|
+
detail += f" code={int(code)}"
|
|
2064
|
+
line = (
|
|
2065
|
+
f"{_cctally_core.now_utc_iso()} worker=stats-corruption-heal "
|
|
2066
|
+
f"result={outcome}{detail}\n"
|
|
2067
|
+
).encode("utf-8")
|
|
2068
|
+
fd = os.open(log_path, os.O_WRONLY | os.O_APPEND | os.O_CREAT, 0o600)
|
|
2069
|
+
try:
|
|
2070
|
+
os.write(fd, line)
|
|
2071
|
+
finally:
|
|
2072
|
+
os.close(fd)
|
|
2073
|
+
except Exception:
|
|
2074
|
+
pass
|
|
2075
|
+
|
|
2076
|
+
|
|
2077
|
+
def defer_stats_corruption_heal(request: dict) -> str:
|
|
2078
|
+
"""Schedule one retryable detached corruption heal without blocking."""
|
|
2079
|
+
try:
|
|
2080
|
+
pathlib.Path(_cctally_core.APP_DIR).mkdir(parents=True, exist_ok=True)
|
|
2081
|
+
admission_fd = os.open(
|
|
2082
|
+
_stats_heal_admission_path(), os.O_WRONLY | os.O_CREAT, 0o600
|
|
2083
|
+
)
|
|
2084
|
+
except OSError:
|
|
2085
|
+
return "failed"
|
|
2086
|
+
try:
|
|
2087
|
+
try:
|
|
2088
|
+
fcntl.flock(admission_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
2089
|
+
except OSError:
|
|
2090
|
+
return "pending"
|
|
2091
|
+
marker = _stats_heal_marker_path()
|
|
2092
|
+
try:
|
|
2093
|
+
age = time.time() - marker.stat().st_mtime
|
|
2094
|
+
except FileNotFoundError:
|
|
2095
|
+
age = None
|
|
2096
|
+
except OSError:
|
|
2097
|
+
return "failed"
|
|
2098
|
+
if age is not None and age < _STATS_HEAL_RETRY_SECONDS:
|
|
2099
|
+
return "pending"
|
|
2100
|
+
if _stats_heal_worker_active():
|
|
2101
|
+
# A real rebuild outlives the marker retry interval. Refresh the
|
|
2102
|
+
# admission stamp instead of launching a process that can only lose
|
|
2103
|
+
# the worker flock and exit.
|
|
2104
|
+
try:
|
|
2105
|
+
os.utime(marker, None)
|
|
2106
|
+
except OSError:
|
|
2107
|
+
pass
|
|
2108
|
+
return "pending"
|
|
2109
|
+
try:
|
|
2110
|
+
_cctally_db._atomic_write_private_json(marker, request)
|
|
2111
|
+
except OSError:
|
|
2112
|
+
return "failed"
|
|
2113
|
+
from _cctally_update import _spawn_detached
|
|
2114
|
+
if _spawn_detached(STATS_CORRUPTION_HEAL_COMMAND):
|
|
2115
|
+
return "spawned"
|
|
2116
|
+
_unlink_stats_heal_marker()
|
|
2117
|
+
return "failed"
|
|
2118
|
+
finally:
|
|
2119
|
+
try:
|
|
2120
|
+
fcntl.flock(admission_fd, fcntl.LOCK_UN)
|
|
2121
|
+
except OSError:
|
|
2122
|
+
pass
|
|
2123
|
+
os.close(admission_fd)
|
|
2124
|
+
|
|
2125
|
+
|
|
2126
|
+
def _run_stats_corruption_heal(request: dict) -> str:
|
|
2127
|
+
"""The worker's body, under its own maintenance-EXCLUSIVE hold.
|
|
2128
|
+
|
|
2129
|
+
Three orderings here are load-bearing (#496 S3 §6):
|
|
2130
|
+
|
|
2131
|
+
* the authoritative probe runs UNDER exclusive, not before it, because
|
|
2132
|
+
epoch rebuilds, operator rebuilds and other maintenance classes hold
|
|
2133
|
+
distinct worker flocks and would otherwise race it;
|
|
2134
|
+
* the no-journal guard is re-checked under the lock, because the hook
|
|
2135
|
+
checked it before spawning and `rebuild_stats_index` accepts a `None`
|
|
2136
|
+
high-water and would build an EMPTY scratch — rebuilding a pre-cutover
|
|
2137
|
+
index to empty is exactly the silent data loss that guard exists to
|
|
2138
|
+
prevent;
|
|
2139
|
+
* the probe is the one the DETECTION established, carried in the request.
|
|
2140
|
+
"""
|
|
2141
|
+
import _cctally_journal
|
|
2142
|
+
|
|
2143
|
+
path = _cctally_core.DB_PATH
|
|
2144
|
+
heal_id = str(request.get("healId") or "")
|
|
2145
|
+
if _cctally_db._would_block_prod_stats(path):
|
|
2146
|
+
return "prod-refused"
|
|
2147
|
+
maint_fd = _heal_flock_bounded(
|
|
2148
|
+
_cctally_core.STATS_LOCK_MAINTENANCE_PATH,
|
|
2149
|
+
_STATS_HEAL_WORKER_MAINTENANCE_WAIT_S,
|
|
2150
|
+
)
|
|
2151
|
+
if maint_fd is None:
|
|
2152
|
+
return "maintenance-busy"
|
|
2153
|
+
_cctally_core.note_stats_maintenance_acquired()
|
|
2154
|
+
try:
|
|
2155
|
+
probe = (
|
|
2156
|
+
_probe_stats_integrity_ok
|
|
2157
|
+
if str(request.get("probeKind") or "")
|
|
2158
|
+
== _STATS_HEAL_PROBE_INTEGRITY
|
|
2159
|
+
else _probe_stats_ok
|
|
2160
|
+
)
|
|
2161
|
+
if probe(path):
|
|
2162
|
+
# F4's second point: a re-probe under a lock the hook never held
|
|
2163
|
+
# finds the index intact, so nothing is replaced.
|
|
2164
|
+
return "declined-readable"
|
|
2165
|
+
high_water = _cctally_journal.journal_high_water()
|
|
2166
|
+
if high_water is None or high_water[1] == 0:
|
|
2167
|
+
return "declined-no-journal"
|
|
2168
|
+
if holds_ingest_lock():
|
|
2169
|
+
ingest_fd = None
|
|
2170
|
+
else:
|
|
2171
|
+
ingest_fd = _heal_flock_bounded(
|
|
2172
|
+
_cctally_core.JOURNAL_INGEST_LOCK_PATH, 10.0
|
|
2173
|
+
)
|
|
2174
|
+
if ingest_fd is None:
|
|
2175
|
+
return "ingest-busy"
|
|
2176
|
+
try:
|
|
2177
|
+
with stats_write_scope("maintenance-heal"):
|
|
2178
|
+
result = _cctally_journal.rebuild_stats_index(
|
|
2179
|
+
context=_cctally_journal.RebuildContext(
|
|
2180
|
+
trigger="corruption-heal",
|
|
2181
|
+
trigger_error=str(request.get("triggerError") or ""),
|
|
2182
|
+
forensics_path=request.get("forensicsPath"),
|
|
2183
|
+
),
|
|
2184
|
+
high_water=high_water,
|
|
2185
|
+
)
|
|
2186
|
+
finally:
|
|
2187
|
+
if ingest_fd is not None:
|
|
2188
|
+
_heal_release_flock(ingest_fd)
|
|
2189
|
+
_record_stats_heal_outcome(heal_id, "rebuilt", result=result)
|
|
2190
|
+
return "success"
|
|
2191
|
+
finally:
|
|
2192
|
+
_heal_release_maintenance_flock(maint_fd)
|
|
2193
|
+
|
|
2194
|
+
|
|
2195
|
+
def cmd_stats_corruption_heal_internal(args) -> int:
|
|
2196
|
+
"""Hidden detached worker: heal one corrupt stats index exactly once."""
|
|
2197
|
+
del args
|
|
2198
|
+
try:
|
|
2199
|
+
pathlib.Path(_cctally_core.APP_DIR).mkdir(parents=True, exist_ok=True)
|
|
2200
|
+
worker_fd = os.open(
|
|
2201
|
+
_stats_heal_worker_path(), os.O_WRONLY | os.O_CREAT, 0o600
|
|
2202
|
+
)
|
|
2203
|
+
except OSError as exc:
|
|
2204
|
+
_log_stats_heal("error", error=exc)
|
|
2205
|
+
return 0
|
|
2206
|
+
try:
|
|
2207
|
+
try:
|
|
2208
|
+
fcntl.flock(worker_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
2209
|
+
except OSError:
|
|
2210
|
+
return 0
|
|
2211
|
+
request = _read_stats_heal_request()
|
|
2212
|
+
if not request:
|
|
2213
|
+
_unlink_stats_heal_marker()
|
|
2214
|
+
_log_stats_heal("no-request")
|
|
2215
|
+
return 0
|
|
2216
|
+
heal_id = str(request.get("healId") or "")
|
|
2217
|
+
try:
|
|
2218
|
+
outcome = _run_stats_corruption_heal(request)
|
|
2219
|
+
except Exception as exc:
|
|
2220
|
+
# Retryable: the marker stays so a later detection is admitted
|
|
2221
|
+
# once its retry window expires.
|
|
2222
|
+
_record_stats_heal_outcome(heal_id, "failed", error=exc)
|
|
2223
|
+
_log_stats_heal("error", heal_id=heal_id, error=exc)
|
|
2224
|
+
return 0
|
|
2225
|
+
if outcome in ("maintenance-busy", "ingest-busy"):
|
|
2226
|
+
_log_stats_heal(outcome, heal_id=heal_id)
|
|
2227
|
+
return 0
|
|
2228
|
+
if outcome != "success":
|
|
2229
|
+
_record_stats_heal_outcome(heal_id, outcome)
|
|
2230
|
+
_unlink_stats_heal_marker()
|
|
2231
|
+
_log_stats_heal(outcome, heal_id=heal_id)
|
|
2232
|
+
return 0
|
|
2233
|
+
finally:
|
|
2234
|
+
try:
|
|
2235
|
+
fcntl.flock(worker_fd, fcntl.LOCK_UN)
|
|
2236
|
+
except OSError:
|
|
2237
|
+
pass
|
|
2238
|
+
os.close(worker_fd)
|
|
2239
|
+
|
|
2240
|
+
|
|
2241
|
+
# --------------------------------------------------------------------------
|
|
2242
|
+
# F6 — the bounded durable heal ring (#496 S3 §7)
|
|
2243
|
+
# --------------------------------------------------------------------------
|
|
2244
|
+
#
|
|
2245
|
+
# Stderr alone does not work as the accountability channel: the statusline's
|
|
2246
|
+
# background writer forks with stderr at `/dev/null` and wraps its body in
|
|
2247
|
+
# `except BaseException: pass`, so a heal firing from there is invisible. The
|
|
2248
|
+
# ring is the durable channel; the stderr line remains for interactive callers.
|
|
2249
|
+
#
|
|
2250
|
+
# **A non-blocking flock is wrong here and would defeat the guarantee.** The
|
|
2251
|
+
# writer-guard log may drop a line under contention because it is advisory;
|
|
2252
|
+
# this ring is the only durable notification that a heal happened, so a loser
|
|
2253
|
+
# that silently discarded its event would make the accountability claim false.
|
|
2254
|
+
# The acquire is therefore a BOUNDED WAIT, and an expiry is reported rather
|
|
2255
|
+
# than swallowed.
|
|
2256
|
+
#
|
|
2257
|
+
# It holds absolute paths because F15 requires the user be told them, so it
|
|
2258
|
+
# stays a private `0600` file in the user's own data directory, like the
|
|
2259
|
+
# incident `manifest.json` beside it. Bounded by COUNT so it cannot grow, which
|
|
2260
|
+
# is also what makes it survive S6's retention by construction.
|
|
2261
|
+
|
|
2262
|
+
_STATS_HEAL_RING_CAPACITY = 50
|
|
2263
|
+
_STATS_HEAL_RING_WAIT_S = 10.0
|
|
2264
|
+
_STATS_HEAL_RECURRENCE_THRESHOLD = 3
|
|
2265
|
+
_STATS_HEAL_RECURRENCE_WINDOW_S = 7 * 86400.0
|
|
2266
|
+
|
|
2267
|
+
|
|
2268
|
+
def _stats_heal_ring_path() -> pathlib.Path:
|
|
2269
|
+
return pathlib.Path(_cctally_core.LOG_DIR) / "stats-heal-events.json"
|
|
2270
|
+
|
|
2271
|
+
|
|
2272
|
+
def _stats_heal_ring_lock_path() -> pathlib.Path:
|
|
2273
|
+
return pathlib.Path(_cctally_core.LOG_DIR) / "stats-heal-events.lock"
|
|
2274
|
+
|
|
2275
|
+
|
|
2276
|
+
def build_stats_heal_event(request: dict, disposition: str) -> dict:
|
|
2277
|
+
"""One ring entry, as the DETECTION knows it.
|
|
2278
|
+
|
|
2279
|
+
`changed` is `unknown` and stays that way. `RebuildResult` carries row
|
|
2280
|
+
counts and replay diagnostics but no comparison against the index it
|
|
2281
|
+
replaced, and `conflicts` / `protocol_violations` report replay ambiguity
|
|
2282
|
+
and omitted correction batches — which is not "the rebuilt index differs
|
|
2283
|
+
from the live one". Recording them is still required, because a user is
|
|
2284
|
+
entitled to know a rebuild reported conflicts.
|
|
2285
|
+
|
|
2286
|
+
`incidentPath` is `None` here and can only be `None` here: the quarantine
|
|
2287
|
+
directory is allocated during preservation, after the worker has chosen
|
|
2288
|
+
physical fallback and begun it (#496 S3 §7 F15).
|
|
2289
|
+
"""
|
|
2290
|
+
return {
|
|
2291
|
+
"schemaVersion": 1,
|
|
2292
|
+
"healId": str(request.get("healId") or ""),
|
|
2293
|
+
"detectedAtUtc": str(
|
|
2294
|
+
request.get("detectedAtUtc") or _cctally_core.now_utc_iso()
|
|
2295
|
+
),
|
|
2296
|
+
"updatedAtUtc": _cctally_core.now_utc_iso(),
|
|
2297
|
+
"trigger": {
|
|
2298
|
+
"origin": "corruption-heal",
|
|
2299
|
+
"type": request.get("triggerType"),
|
|
2300
|
+
"error": request.get("triggerError"),
|
|
2301
|
+
"postQuery": bool(request.get("postQuery")),
|
|
2302
|
+
},
|
|
2303
|
+
"disposition": disposition,
|
|
2304
|
+
"forensicsPath": request.get("forensicsPath"),
|
|
2305
|
+
"incidentPath": None,
|
|
2306
|
+
"publicationMechanism": None,
|
|
2307
|
+
"outcome": "detected",
|
|
2308
|
+
"changed": "unknown",
|
|
2309
|
+
}
|
|
2310
|
+
|
|
2311
|
+
|
|
2312
|
+
def _report_unreadable_stats_heal_ring(reason: str) -> None:
|
|
2313
|
+
"""Report a ring file that exists but cannot be read as a ring.
|
|
2314
|
+
|
|
2315
|
+
A ring that reads as empty is indistinguishable from a ring that never
|
|
2316
|
+
recorded anything, and the next writer overwrites it — so without this the
|
|
2317
|
+
accountability history would disappear with nothing said. Both channels are
|
|
2318
|
+
used because neither reaches every caller: the worker's streams are
|
|
2319
|
+
`/dev/null`, and an interactive caller does not read the heal log.
|
|
2320
|
+
"""
|
|
2321
|
+
_log_stats_heal(f"ring-unreadable-{reason}")
|
|
2322
|
+
print(
|
|
2323
|
+
f"[heal] the stats.db heal event log at {_stats_heal_ring_path()} "
|
|
2324
|
+
f"could not be read ({reason}) and reports no history; the next "
|
|
2325
|
+
"recorded heal replaces it.",
|
|
2326
|
+
file=sys.stderr,
|
|
2327
|
+
)
|
|
2328
|
+
|
|
2329
|
+
|
|
2330
|
+
def _read_stats_heal_ring() -> list:
|
|
2331
|
+
try:
|
|
2332
|
+
payload = json.loads(_stats_heal_ring_path().read_text())
|
|
2333
|
+
except FileNotFoundError:
|
|
2334
|
+
return []
|
|
2335
|
+
except (OSError, ValueError) as exc:
|
|
2336
|
+
_report_unreadable_stats_heal_ring(type(exc).__name__)
|
|
2337
|
+
return []
|
|
2338
|
+
events = payload.get("events") if isinstance(payload, dict) else None
|
|
2339
|
+
if not isinstance(events, list):
|
|
2340
|
+
_report_unreadable_stats_heal_ring("NoEventList")
|
|
2341
|
+
return []
|
|
2342
|
+
return [e for e in events if isinstance(e, dict)]
|
|
2343
|
+
|
|
2344
|
+
|
|
2345
|
+
def read_stats_heal_events() -> list:
|
|
2346
|
+
"""Every retained heal event, oldest first. Public: S6's F14 reads this."""
|
|
2347
|
+
return _read_stats_heal_ring()
|
|
2348
|
+
|
|
2349
|
+
|
|
2350
|
+
def _write_stats_heal_ring(events: list) -> None:
|
|
2351
|
+
_cctally_db._atomic_write_private_json(
|
|
2352
|
+
_stats_heal_ring_path(),
|
|
2353
|
+
{"schemaVersion": 1, "events": events[-_STATS_HEAL_RING_CAPACITY:]},
|
|
2354
|
+
)
|
|
2355
|
+
|
|
2356
|
+
|
|
2357
|
+
def _mutate_stats_heal_ring(mutate) -> bool:
|
|
2358
|
+
"""Read-modify-write the ring under a BOUNDED wait for its lock."""
|
|
2359
|
+
try:
|
|
2360
|
+
pathlib.Path(_cctally_core.LOG_DIR).mkdir(parents=True, exist_ok=True)
|
|
2361
|
+
except OSError:
|
|
2362
|
+
return False
|
|
2363
|
+
fd = _heal_flock_bounded(
|
|
2364
|
+
_stats_heal_ring_lock_path(), _STATS_HEAL_RING_WAIT_S
|
|
2365
|
+
)
|
|
2366
|
+
if fd is None:
|
|
2367
|
+
# Loud, never silent: the ring is the accountability guarantee, so a
|
|
2368
|
+
# writer that could not take the lock says so rather than discarding
|
|
2369
|
+
# its event.
|
|
2370
|
+
print(
|
|
2371
|
+
"[heal] could not record a stats.db heal event: the heal event "
|
|
2372
|
+
"log stayed locked; the heal itself is unaffected.",
|
|
2373
|
+
file=sys.stderr,
|
|
2374
|
+
)
|
|
2375
|
+
return False
|
|
2376
|
+
try:
|
|
2377
|
+
events = _read_stats_heal_ring()
|
|
2378
|
+
mutated = mutate(events)
|
|
2379
|
+
if mutated is None:
|
|
2380
|
+
return False
|
|
2381
|
+
_write_stats_heal_ring(mutated)
|
|
2382
|
+
return True
|
|
2383
|
+
except OSError:
|
|
2384
|
+
return False
|
|
2385
|
+
finally:
|
|
2386
|
+
_heal_release_flock(fd)
|
|
2387
|
+
|
|
2388
|
+
|
|
2389
|
+
def append_stats_heal_event(entry: dict) -> bool:
|
|
2390
|
+
"""Append one detection entry. Bounded by count, oldest dropped first."""
|
|
2391
|
+
def mutate(events):
|
|
2392
|
+
events.append(entry)
|
|
2393
|
+
return events
|
|
2394
|
+
|
|
2395
|
+
return _mutate_stats_heal_ring(mutate)
|
|
2396
|
+
|
|
2397
|
+
|
|
2398
|
+
def update_stats_heal_event(heal_id: str, **fields) -> bool:
|
|
2399
|
+
"""Update the entry MATCHING ``heal_id``, and no other.
|
|
2400
|
+
|
|
2401
|
+
Admission coalesces several detections into one run, so an update keyed by
|
|
2402
|
+
anything else (position, recency) would settle a heal whose worker never
|
|
2403
|
+
ran and hide the one that died.
|
|
2404
|
+
"""
|
|
2405
|
+
if not heal_id:
|
|
2406
|
+
return False
|
|
2407
|
+
|
|
2408
|
+
def mutate(events):
|
|
2409
|
+
for event in events:
|
|
2410
|
+
if event.get("healId") == heal_id:
|
|
2411
|
+
event.update(fields)
|
|
2412
|
+
event["updatedAtUtc"] = _cctally_core.now_utc_iso()
|
|
2413
|
+
return events
|
|
2414
|
+
return None
|
|
2415
|
+
|
|
2416
|
+
return _mutate_stats_heal_ring(mutate)
|
|
2417
|
+
|
|
2418
|
+
|
|
2419
|
+
def stats_heal_recurrence(
|
|
2420
|
+
window_s: float = _STATS_HEAL_RECURRENCE_WINDOW_S,
|
|
2421
|
+
) -> int:
|
|
2422
|
+
"""How many heals were detected inside the trailing window."""
|
|
2423
|
+
cutoff = dt.datetime.now(dt.timezone.utc) - dt.timedelta(seconds=window_s)
|
|
2424
|
+
count = 0
|
|
2425
|
+
for event in _read_stats_heal_ring():
|
|
2426
|
+
try:
|
|
2427
|
+
detected = dt.datetime.fromisoformat(
|
|
2428
|
+
str(event.get("detectedAtUtc") or "").replace("Z", "+00:00")
|
|
2429
|
+
)
|
|
2430
|
+
except ValueError:
|
|
2431
|
+
continue
|
|
2432
|
+
if detected.tzinfo is None:
|
|
2433
|
+
detected = detected.replace(tzinfo=dt.timezone.utc)
|
|
2434
|
+
if detected >= cutoff:
|
|
2435
|
+
count += 1
|
|
2436
|
+
return count
|
|
2437
|
+
|
|
2438
|
+
|
|
2439
|
+
def _record_stats_heal_outcome(
|
|
2440
|
+
heal_id: str, outcome: str, *, result=None, error: BaseException | None = None,
|
|
2441
|
+
) -> None:
|
|
2442
|
+
"""Settle the durable heal event this worker owns."""
|
|
2443
|
+
fields: dict = {"outcome": outcome}
|
|
2444
|
+
if result is not None:
|
|
2445
|
+
incident = getattr(result, "quarantine_dir", None)
|
|
2446
|
+
fields["incidentPath"] = str(incident) if incident is not None else None
|
|
2447
|
+
# An in-place publish never preserves, so `quarantine_dir is None` is
|
|
2448
|
+
# an exact discriminator for which mechanism published (#496 S3 §4.1).
|
|
2449
|
+
fields["publicationMechanism"] = (
|
|
2450
|
+
"replace" if incident is not None else "in_place"
|
|
2451
|
+
)
|
|
2452
|
+
fields["conflicts"] = len(getattr(result, "conflicts", ()) or ())
|
|
2453
|
+
fields["protocolViolations"] = len(
|
|
2454
|
+
getattr(result, "protocol_violations", ()) or ()
|
|
2455
|
+
)
|
|
2456
|
+
fields["rowsTotal"] = sum(
|
|
2457
|
+
int(v) for v in (getattr(result, "rows_by_table", {}) or {}).values()
|
|
2458
|
+
)
|
|
2459
|
+
if error is not None:
|
|
2460
|
+
# Structural only, like `stats-epoch-rebuild.log`: never free-form
|
|
2461
|
+
# exception text, which can carry private paths.
|
|
2462
|
+
fields["error"] = type(error).__name__
|
|
2463
|
+
if not update_stats_heal_event(heal_id, **fields):
|
|
2464
|
+
# The ring is the accountability record this session added, so a
|
|
2465
|
+
# verdict that never reached it must not vanish silently. The durable
|
|
2466
|
+
# log is the channel that works here: this runs only in the detached
|
|
2467
|
+
# worker, whose stdout and stderr are `/dev/null`.
|
|
2468
|
+
_log_stats_heal(f"ring-update-lost-{outcome}", heal_id=heal_id)
|
|
2469
|
+
|
|
2470
|
+
|
|
1450
2471
|
# --------------------------------------------------------------------------
|
|
1451
2472
|
# §7.1 stats.db epoch-mismatch resolution (Task 9)
|
|
1452
2473
|
# --------------------------------------------------------------------------
|