ossify 0.2.3__tar.gz → 0.2.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {ossify-0.2.3 → ossify-0.2.5}/PKG-INFO +1 -1
  2. {ossify-0.2.3 → ossify-0.2.5}/pyproject.toml +2 -2
  3. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/__init__.py +1 -1
  4. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartments.py +232 -10
  5. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/file_io.py +101 -12
  6. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/structured_prediction.py +47 -0
  7. {ossify-0.2.3 → ossify-0.2.5}/.github/workflows/mkdocs_publish.yml +0 -0
  8. {ossify-0.2.3 → ossify-0.2.5}/.github/workflows/python-package.yml +0 -0
  9. {ossify-0.2.3 → ossify-0.2.5}/.gitignore +0 -0
  10. {ossify-0.2.3 → ossify-0.2.5}/.pre-commit-config.yaml +0 -0
  11. {ossify-0.2.3 → ossify-0.2.5}/LICENSE +0 -0
  12. {ossify-0.2.3 → ossify-0.2.5}/README.md +0 -0
  13. {ossify-0.2.3 → ossify-0.2.5}/mkdocs.yml +0 -0
  14. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/__init__.py +0 -0
  15. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/base.py +0 -0
  16. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/graph.py +0 -0
  17. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/mapping.py +0 -0
  18. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/mesh.py +0 -0
  19. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/morph.py +0 -0
  20. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/points.py +0 -0
  21. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/table.py +0 -0
  22. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/algorithms.py +0 -0
  23. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/base.py +0 -0
  24. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartment_models/minnie65_ds15_us0_bd0.json +0 -0
  25. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartment_models/v1dd_ds15_us0_bd0.json +0 -0
  26. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartment_models/xgb/minnie65_ds15_us0_bd0.ubj +0 -0
  27. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartment_models/xgb/v1dd_ds15_us0_bd0.ubj +0 -0
  28. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/data_layers.py +0 -0
  29. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/graph_functions.py +0 -0
  30. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/plot.py +0 -0
  31. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/plot3d.py +0 -0
  32. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/plot_utils.py +0 -0
  33. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/sync_classes.py +0 -0
  34. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/translate.py +0 -0
  35. {ossify-0.2.3 → ossify-0.2.5}/src/ossify/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ossify
3
- Version: 0.2.3
3
+ Version: 0.2.5
4
4
  Summary: Mesh and skeleton analysis
5
5
  Author-email: Casey Schneider-Mizell <caseysm@gmail.com>
6
6
  License-File: LICENSE
@@ -6,7 +6,7 @@ build-backend = "hatchling.build"
6
6
  allow-direct-references = true
7
7
  [project]
8
8
  name = "ossify"
9
- version = "0.2.3"
9
+ version = "0.2.5"
10
10
  description = "Mesh and skeleton analysis"
11
11
  readme = "README.md"
12
12
  requires-python = ">=3.11"
@@ -95,7 +95,7 @@ default-groups = ["dev", "docs", "lint", "profile", "viz"]
95
95
 
96
96
 
97
97
  [tool.bumpversion]
98
- current_version = "0.2.3"
98
+ current_version = "0.2.5"
99
99
  parse = "(?P<major>\\d+)\\.(?P<minor>\\d+)\\.(?P<patch>\\d+)"
100
100
  serialize = ["{major}.{minor}.{patch}"]
101
101
  regex = false
@@ -5,7 +5,7 @@ from .base import *
5
5
  from .file_io import *
6
6
  from .translate import *
7
7
 
8
- __version__ = "0.2.3"
8
+ __version__ = "0.2.5"
9
9
 
10
10
 
11
11
  def __getattr__(name):
@@ -26,6 +26,7 @@ from dataclasses import dataclass, field
26
26
  from typing import Callable, List, Literal, Optional, Union
27
27
 
28
28
  import numpy as np
29
+ import orjson
29
30
  import pandas as pd
30
31
  from scipy import sparse
31
32
 
@@ -38,6 +39,7 @@ from .algorithms import (
38
39
  from .base import Cell
39
40
  from .structured_prediction import (
40
41
  TransitionSchema,
42
+ _ossify_version,
41
43
  absorb_small_compartments,
42
44
  decode_tree,
43
45
  )
@@ -954,6 +956,10 @@ class ProbaVertexModel(SkeletonVertexModel):
954
956
  See :func:`make_skel_prop_df`. Defaults to :data:`DEFAULT_FEATURE_SPEC`.
955
957
  """
956
958
 
959
+ #: Bumped when :meth:`to_dict`'s shape changes incompatibly. Independent of
960
+ #: the ossify package version.
961
+ _SCHEMA_VERSION = 1
962
+
957
963
  def __init__(
958
964
  self,
959
965
  estimator,
@@ -974,6 +980,11 @@ class ProbaVertexModel(SkeletonVertexModel):
974
980
  self._feature_columns = list(feature_columns)
975
981
  self._smooth_alpha = smooth_alpha
976
982
  self._unary_clip = float(unary_clip)
983
+ #: The ``config`` argument the instance was built from via
984
+ #: :meth:`from_config` (a bundled model name, a user config dict, or a
985
+ #: config file path); ``None`` when built directly from an estimator.
986
+ #: Lets :meth:`to_dict` round-trip without re-serializing the estimator.
987
+ self._config_ref: Optional[Union[dict, str]] = None
977
988
  super().__init__(
978
989
  downstream_hops=downstream_hops,
979
990
  upstream_hops=upstream_hops,
@@ -981,6 +992,12 @@ class ProbaVertexModel(SkeletonVertexModel):
981
992
  feature_spec=feature_spec,
982
993
  )
983
994
 
995
+ @staticmethod
996
+ def _resolve_config(config: Optional[Union[dict, str]]) -> dict:
997
+ if not isinstance(config, dict):
998
+ config = load_model_config(config, dir=None)
999
+ return config
1000
+
984
1001
  @classmethod
985
1002
  def from_config(
986
1003
  cls,
@@ -997,22 +1014,125 @@ class ProbaVertexModel(SkeletonVertexModel):
997
1014
  the config's ``spread_alpha``; ``unary_clip`` to the config's ``unary_clip``
998
1015
  key, falling back to :data:`DEFAULT_UNARY_CLIP`.
999
1016
  """
1000
- if not isinstance(config, dict):
1001
- config = load_model_config(config, dir=None)
1017
+ config_ref = config if config is not None else DEFAULT_MODEL
1018
+ resolved = cls._resolve_config(config)
1002
1019
  if smooth_alpha is None:
1003
- smooth_alpha = config.get("spread_alpha")
1020
+ smooth_alpha = resolved.get("spread_alpha")
1004
1021
  if unary_clip is None:
1005
- unary_clip = config.get("unary_clip", DEFAULT_UNARY_CLIP)
1006
- return cls(
1007
- _load_xgboost_classifier(config["model_file"]),
1008
- config["feature_columns"],
1009
- downstream_hops=config.get("downstream_hops", 0),
1010
- upstream_hops=config.get("upstream_hops", 0),
1011
- bidirectional_hops=config.get("bidirectional_hops", 0),
1022
+ unary_clip = resolved.get("unary_clip", DEFAULT_UNARY_CLIP)
1023
+ obj = cls(
1024
+ _load_xgboost_classifier(resolved["model_file"]),
1025
+ resolved["feature_columns"],
1026
+ downstream_hops=resolved.get("downstream_hops", 0),
1027
+ upstream_hops=resolved.get("upstream_hops", 0),
1028
+ bidirectional_hops=resolved.get("bidirectional_hops", 0),
1012
1029
  smooth_alpha=smooth_alpha,
1013
1030
  unary_clip=unary_clip,
1014
1031
  feature_spec=feature_spec,
1015
1032
  )
1033
+ obj._config_ref = config_ref
1034
+ return obj
1035
+
1036
+ def to_dict(self, *, model_file: Optional[Union[str, pathlib.Path]] = None) -> dict:
1037
+ """Plain-data representation, round-trippable via :meth:`from_dict`.
1038
+
1039
+ When built via :meth:`from_config`, the ``config`` it was given (a
1040
+ bundled model name, config file path, or dict) is stored directly, so
1041
+ reloading re-resolves the same estimator without touching it.
1042
+
1043
+ Otherwise (built directly from an ``estimator``), the estimator has no
1044
+ known source file: pass ``model_file`` to save it there (requires an
1045
+ estimator with a ``save_model(path)`` method, e.g. XGBoost) and a fresh
1046
+ config dict pointing at it is built automatically. Without
1047
+ ``model_file``, this raises -- there is nothing to persist besides the
1048
+ estimator itself, and this is deliberately not a pickle.
1049
+
1050
+ Raises
1051
+ ------
1052
+ ValueError
1053
+ If ``feature_spec`` is non-default (it may contain callables that
1054
+ cannot be serialized) or the estimator's source can't be
1055
+ determined.
1056
+ """
1057
+ if self._feature_spec is not None:
1058
+ raise ValueError(
1059
+ "This ProbaVertexModel has a custom feature_spec, which may "
1060
+ "contain callables and cannot be serialized. Reattach it "
1061
+ "explicitly when reloading: "
1062
+ "ProbaVertexModel.from_dict(d, feature_spec=...)."
1063
+ )
1064
+ config_ref = self._config_ref
1065
+ if config_ref is None:
1066
+ if model_file is None:
1067
+ raise ValueError(
1068
+ "This ProbaVertexModel was not built via from_config(...), "
1069
+ "so its estimator has no known source file. Pass "
1070
+ "model_file=<path> to save the fitted estimator there "
1071
+ "(requires an estimator with a save_model(path) method, "
1072
+ "e.g. XGBoost), or save/load the estimator yourself and "
1073
+ "build a config dict for from_dict()."
1074
+ )
1075
+ if not hasattr(self._estimator, "save_model"):
1076
+ raise TypeError(
1077
+ f"estimator of type {type(self._estimator).__name__} has no "
1078
+ "save_model(path) method; save it yourself and pass a "
1079
+ "config dict to from_dict() instead."
1080
+ )
1081
+ self._estimator.save_model(model_file)
1082
+ config_ref = {
1083
+ "model_file": str(model_file),
1084
+ "feature_columns": self._feature_columns,
1085
+ "downstream_hops": self._downstream_hops,
1086
+ "upstream_hops": self._upstream_hops,
1087
+ "bidirectional_hops": self._bidirectional_hops,
1088
+ }
1089
+ elif isinstance(config_ref, dict):
1090
+ config_ref = {
1091
+ k: (str(v) if isinstance(v, pathlib.Path) else v)
1092
+ for k, v in config_ref.items()
1093
+ }
1094
+ return {
1095
+ "type": "ProbaVertexModel",
1096
+ "version": self._SCHEMA_VERSION,
1097
+ "ossify_version": _ossify_version(),
1098
+ "config": config_ref,
1099
+ "smooth_alpha": self._smooth_alpha,
1100
+ "unary_clip": self._unary_clip,
1101
+ }
1102
+
1103
+ @classmethod
1104
+ def from_dict(
1105
+ cls,
1106
+ d: dict,
1107
+ *,
1108
+ feature_spec: Optional[List[FeatureDef]] = None,
1109
+ ) -> "ProbaVertexModel":
1110
+ """Reconstruct a :class:`ProbaVertexModel` from :meth:`to_dict` output.
1111
+
1112
+ Unlike :meth:`from_config`, ``smooth_alpha``/``unary_clip`` are taken
1113
+ as-is from ``d`` (not re-defaulted from the config file), so a
1114
+ deliberate ``smooth_alpha: null`` (no smoothing) round-trips correctly.
1115
+ """
1116
+ version = d.get("version", 1)
1117
+ if version > cls._SCHEMA_VERSION:
1118
+ raise ValueError(
1119
+ f"ProbaVertexModel dict has schema version {version}, newer "
1120
+ f"than the {cls._SCHEMA_VERSION} this ossify release supports; "
1121
+ "upgrade ossify to load it."
1122
+ )
1123
+ resolved = cls._resolve_config(d["config"])
1124
+ obj = cls(
1125
+ _load_xgboost_classifier(resolved["model_file"]),
1126
+ resolved["feature_columns"],
1127
+ downstream_hops=resolved.get("downstream_hops", 0),
1128
+ upstream_hops=resolved.get("upstream_hops", 0),
1129
+ bidirectional_hops=resolved.get("bidirectional_hops", 0),
1130
+ smooth_alpha=d.get("smooth_alpha"),
1131
+ unary_clip=d.get("unary_clip", DEFAULT_UNARY_CLIP),
1132
+ feature_spec=feature_spec,
1133
+ )
1134
+ obj._config_ref = d["config"]
1135
+ return obj
1016
1136
 
1017
1137
  @property
1018
1138
  def estimator(self):
@@ -1059,6 +1179,11 @@ class ProbaVertexModel(SkeletonVertexModel):
1059
1179
 
1060
1180
  # --- Structured compartment labeler -----------------------------------------
1061
1181
 
1182
+ #: Registry of encoder types :class:`StructuredLabeler` can (de)serialize.
1183
+ #: Extend when a new :class:`SkeletonVertexModel` subclass grows a
1184
+ #: to_dict/from_dict pair.
1185
+ _ENCODER_TYPES = {"ProbaVertexModel": ProbaVertexModel}
1186
+
1062
1187
 
1063
1188
  class StructuredLabeler:
1064
1189
  """Compose an encoder + transition priors into a tree-decoded labeler.
@@ -1095,6 +1220,10 @@ class StructuredLabeler:
1095
1220
  this or less.
1096
1221
  """
1097
1222
 
1223
+ #: Bumped when :meth:`to_dict`'s shape changes incompatibly. Independent of
1224
+ #: the ossify package version.
1225
+ _SCHEMA_VERSION = 1
1226
+
1098
1227
  def __init__(
1099
1228
  self,
1100
1229
  encoder: SkeletonVertexModel,
@@ -1198,3 +1327,96 @@ class StructuredLabeler:
1198
1327
  """
1199
1328
  labels, edge_cost = self._decode(cell)
1200
1329
  return self._as_labels(labels, return_labels_as), edge_cost
1330
+
1331
+ # --- Serialization -------------------------------------------------------
1332
+
1333
+ def to_dict(self, *, model_file: Optional[Union[str, pathlib.Path]] = None) -> dict:
1334
+ """Plain-data representation, round-trippable via :meth:`from_dict`.
1335
+
1336
+ Composes :meth:`TransitionSchema.to_dict` and the encoder's own
1337
+ ``to_dict`` -- so a labeler built from a bundled model config (the
1338
+ common case) serializes down to a small JSON-safe dict with no model
1339
+ weights inlined. ``model_file`` is forwarded to the encoder's
1340
+ ``to_dict`` for the case where its estimator has no known source file
1341
+ (see :meth:`ProbaVertexModel.to_dict`). Includes a ``version`` (this
1342
+ dict shape, bumped on breaking changes) and ``ossify_version`` (the
1343
+ release that wrote it, for provenance/debugging).
1344
+ """
1345
+ encoder_type = type(self._encoder).__name__
1346
+ if encoder_type not in _ENCODER_TYPES:
1347
+ raise TypeError(
1348
+ f"Serialization is not supported for encoder type "
1349
+ f"{encoder_type!r}; supported types: {sorted(_ENCODER_TYPES)}."
1350
+ )
1351
+ return {
1352
+ "type": "StructuredLabeler",
1353
+ "version": self._SCHEMA_VERSION,
1354
+ "ossify_version": _ossify_version(),
1355
+ "schema": self._schema.to_dict(),
1356
+ "encoder": self._encoder.to_dict(model_file=model_file),
1357
+ "absorb_min_size": self._absorb_min_size,
1358
+ "absorb_min_weight": self._absorb_min_weight,
1359
+ }
1360
+
1361
+ @classmethod
1362
+ def from_dict(
1363
+ cls,
1364
+ d: dict,
1365
+ *,
1366
+ feature_spec: Optional[List[FeatureDef]] = None,
1367
+ ) -> "StructuredLabeler":
1368
+ """Reconstruct a :class:`StructuredLabeler` from :meth:`to_dict` output.
1369
+
1370
+ ``feature_spec`` is forwarded to the encoder's ``from_dict`` and only
1371
+ needed when the original encoder was built with a non-default
1372
+ ``feature_spec`` (see :meth:`ProbaVertexModel.to_dict`).
1373
+ """
1374
+ version = d.get("version", 1)
1375
+ if version > cls._SCHEMA_VERSION:
1376
+ raise ValueError(
1377
+ f"StructuredLabeler dict has schema version {version}, newer "
1378
+ f"than the {cls._SCHEMA_VERSION} this ossify release supports; "
1379
+ "upgrade ossify to load it."
1380
+ )
1381
+ encoder_dict = d["encoder"]
1382
+ encoder_type = encoder_dict.get("type")
1383
+ encoder_cls = _ENCODER_TYPES.get(encoder_type)
1384
+ if encoder_cls is None:
1385
+ raise ValueError(
1386
+ f"Unknown or unsupported encoder type: {encoder_type!r}; "
1387
+ f"supported types: {sorted(_ENCODER_TYPES)}."
1388
+ )
1389
+ encoder = encoder_cls.from_dict(encoder_dict, feature_spec=feature_spec)
1390
+ schema = TransitionSchema.from_dict(d["schema"])
1391
+ return cls(
1392
+ encoder,
1393
+ schema,
1394
+ absorb_min_size=d.get("absorb_min_size"),
1395
+ absorb_min_weight=d.get("absorb_min_weight"),
1396
+ )
1397
+
1398
+ def save_config(
1399
+ self,
1400
+ filename: Union[str, pathlib.Path],
1401
+ *,
1402
+ model_file: Optional[Union[str, pathlib.Path]] = None,
1403
+ ) -> None:
1404
+ """Write :meth:`to_dict` to ``filename`` as JSON."""
1405
+ with open(filename, "wb") as f:
1406
+ f.write(
1407
+ orjson.dumps(
1408
+ self.to_dict(model_file=model_file), option=orjson.OPT_INDENT_2
1409
+ )
1410
+ )
1411
+
1412
+ @classmethod
1413
+ def load_config(
1414
+ cls,
1415
+ filename: Union[str, pathlib.Path],
1416
+ *,
1417
+ feature_spec: Optional[List[FeatureDef]] = None,
1418
+ ) -> "StructuredLabeler":
1419
+ """Load a :class:`StructuredLabeler` from a JSON file written by :meth:`save_config`."""
1420
+ with open(filename, "rb") as f:
1421
+ d = orjson.loads(f.read())
1422
+ return cls.from_dict(d, feature_spec=feature_spec)
@@ -913,6 +913,68 @@ def build_point_cloud(
913
913
  return cell
914
914
 
915
915
 
916
+ def _exact_integer_id_coverage(link_ids: pd.Index, source_ids: pd.Index) -> bool:
917
+ """Test whether two integer-ID domains are exactly the same set.
918
+
919
+ Both arguments must already be canonicalized ``int64`` pandas ``Index``
920
+ objects. The comparison is order-independent and stays entirely in the
921
+ integer domain -- it never coerces identifiers through ``float64`` -- so it
922
+ is safe for IDs above ``2**53``.
923
+
924
+ Parameters
925
+ ----------
926
+ link_ids :
927
+ The linkage column for a candidate source layer.
928
+ source_ids :
929
+ That layer's vertex index.
930
+
931
+ Returns
932
+ -------
933
+ :
934
+ ``True`` iff the two contain exactly the same set of IDs.
935
+ """
936
+ if len(link_ids) != len(source_ids):
937
+ return False
938
+ # ``symmetric_difference`` is empty iff neither side has an ID the other
939
+ # lacks. Using pandas ``Index`` set ops keeps the comparison in int64 and
940
+ # avoids materializing Python ``set`` objects over large ID arrays.
941
+ return len(link_ids.symmetric_difference(source_ids)) == 0
942
+
943
+
944
+ def _linkage_source_error(linkage_pair, link_df, cell, sample_size: int = 10) -> str:
945
+ """Build a concise diagnostic message for an unresolvable linkage source.
946
+
947
+ Reports, for each endpoint, the layer vertex count, the linkage row count,
948
+ the number of unique linkage IDs, and the missing/extra ID counts, plus a
949
+ small bounded sample of the offending IDs. This replaces the enormous
950
+ pandas ``KeyError`` that a bad ``.loc`` reindex would otherwise raise.
951
+ """
952
+ lines = [
953
+ f"Could not infer a valid source layer for linkage "
954
+ f"({linkage_pair[0]!r}, {linkage_pair[1]!r}) with {len(link_df)} rows.",
955
+ "A source endpoint must map exactly one linkage row to each of its "
956
+ "vertices (unique, complete coverage). Neither endpoint qualifies:",
957
+ ]
958
+ for source in linkage_pair:
959
+ source_ids = canonicalize_ids(
960
+ pd.Index(cell._all_objects[source].vertex_index), name=source
961
+ )
962
+ link_ids = canonicalize_ids(pd.Index(link_df[source]), name=source)
963
+ unique_link_ids = link_ids.unique()
964
+ missing = source_ids.difference(link_ids)
965
+ extra = link_ids.difference(source_ids)
966
+ lines.append(
967
+ f" - {source!r}: vertices={len(source_ids)}, "
968
+ f"link_rows={len(link_ids)}, unique_link_ids={len(unique_link_ids)}, "
969
+ f"missing_source_ids={len(missing)}, extra_link_ids={len(extra)}"
970
+ )
971
+ if len(missing):
972
+ lines.append(f" missing sample: {missing[:sample_size].tolist()}")
973
+ if len(extra):
974
+ lines.append(f" extra sample: {extra[:sample_size].tolist()}")
975
+ return "\n".join(lines)
976
+
977
+
916
978
  def build_linkage(
917
979
  linkage_pair,
918
980
  tf,
@@ -922,23 +984,50 @@ def build_linkage(
922
984
  link_df = load_dataframe(f"{prefix}/linkage.feather", tf)
923
985
  # Legacy .osy files may store link columns with mixed int64/uint64 dtypes
924
986
  # (dtype optimization only downcast signed ints, leaving uint64 untouched).
925
- # Canonicalize both columns to int64 before the label-based ``.loc`` reindex
926
- # below so that lookup cannot collide two IDs above 2**53 through a float
927
- # coercion of mismatched signed/unsigned keys.
987
+ # Canonicalize both columns to int64 before any label-based reindex below so
988
+ # that lookup cannot collide two IDs above 2**53 through a float coercion of
989
+ # mismatched signed/unsigned keys.
928
990
  for col in linkage_pair:
929
991
  if col in link_df.columns:
930
992
  link_df[col] = canonicalize_ids(link_df[col], name=col)
931
- # Determine source based on the length of the vertices in the mapping and in the skeleton layer
932
- if len(link_df) == len(cell._all_objects[linkage_pair[0]].nodes):
933
- source_layer = linkage_pair[0]
934
- target_layer = linkage_pair[1]
935
- elif len(link_df) == len(cell._all_objects[linkage_pair[1]].nodes):
936
- source_layer = linkage_pair[1]
937
- target_layer = linkage_pair[0]
938
- else:
939
- raise ValueError("Linkage DataFrame does not match any layer.")
993
+
994
+ # The archive sorts the pair names, so serialized column order does not
995
+ # preserve the original source direction. Infer the source by exact
996
+ # source-domain validation rather than row count alone: a layer qualifies
997
+ # as source only when its linkage column has no missing IDs, is unique, has
998
+ # one row per vertex, and covers exactly that layer's vertex index. Row
999
+ # count alone is ambiguous whenever the two layers have equal cardinality
1000
+ # (e.g. a many-to-one graph <-> annotation link where the counts coincide).
1001
+ candidates = []
1002
+ for source, target in (
1003
+ (linkage_pair[0], linkage_pair[1]),
1004
+ (linkage_pair[1], linkage_pair[0]),
1005
+ ):
1006
+ source_ids = canonicalize_ids(
1007
+ pd.Index(cell._all_objects[source].vertex_index), name=source
1008
+ )
1009
+ link_ids = canonicalize_ids(pd.Index(link_df[source]), name=source)
1010
+ if (
1011
+ len(link_ids) == len(source_ids)
1012
+ and link_ids.is_unique
1013
+ and _exact_integer_id_coverage(link_ids, source_ids)
1014
+ ):
1015
+ candidates.append((source, target))
1016
+
1017
+ if not candidates:
1018
+ raise ValueError(_linkage_source_error(linkage_pair, link_df, cell))
1019
+
1020
+ # Zero candidates raises above. One candidate is the unambiguous source.
1021
+ # Two candidates means a true bijection: both columns are unique and cover
1022
+ # their layers exactly, so the mapping is one-to-one. MorphSync stores every
1023
+ # link in both directions (see ``add_link``), so either choice yields the
1024
+ # same bidirectional link; pick the first deterministically.
1025
+ source_layer, target_layer = candidates[0]
940
1026
 
941
1027
  layer = cell._all_objects[source_layer]
1028
+ # Validation above guarantees the source column is unique and covers exactly
1029
+ # the source vertex index, so this reindex is a total, collision-free
1030
+ # reordering -- it can no longer raise a large KeyError.
942
1031
  cell._all_objects[source_layer]._process_linkage(
943
1032
  Link(
944
1033
  link_df.set_index(source_layer).loc[layer.vertex_index][target_layer],
@@ -26,6 +26,13 @@ __all__ = [
26
26
  ]
27
27
 
28
28
 
29
+ def _ossify_version() -> str:
30
+ """The installed ossify release, for provenance in serialized dicts."""
31
+ from . import __version__
32
+
33
+ return __version__
34
+
35
+
29
36
  class TransitionSchema:
30
37
  """Declarative label set with soft parent->child transition costs.
31
38
 
@@ -51,6 +58,10 @@ class TransitionSchema:
51
58
  label changes over the tree.
52
59
  """
53
60
 
61
+ #: Bumped when :meth:`to_dict`'s shape changes incompatibly. Independent of
62
+ #: the ossify package version.
63
+ _SCHEMA_VERSION = 1
64
+
54
65
  def __init__(
55
66
  self,
56
67
  classes: Sequence,
@@ -105,6 +116,42 @@ class TransitionSchema:
105
116
  mask[self._index[c]] = True
106
117
  return mask
107
118
 
119
+ def to_dict(self) -> dict:
120
+ """Plain-data representation, round-trippable via :meth:`from_dict`.
121
+
122
+ ``transitions`` is written as a list of ``[parent, child, cost]``
123
+ triples rather than a ``{(parent, child): cost}`` dict, since JSON/YAML
124
+ cannot use tuples as object keys. Includes a ``version`` (this dict
125
+ shape, bumped on breaking changes) and ``ossify_version`` (the release
126
+ that wrote it, for provenance/debugging) alongside the parameters.
127
+ """
128
+ return {
129
+ "type": "TransitionSchema",
130
+ "version": self._SCHEMA_VERSION,
131
+ "ossify_version": _ossify_version(),
132
+ "classes": list(self.classes),
133
+ "transitions": [[a, b, cost] for (a, b), cost in self.transitions.items()],
134
+ "root_classes": list(self.root_classes),
135
+ "default_cost": self.default_cost,
136
+ }
137
+
138
+ @classmethod
139
+ def from_dict(cls, d: dict) -> "TransitionSchema":
140
+ """Reconstruct a :class:`TransitionSchema` from :meth:`to_dict` output."""
141
+ version = d.get("version", 1)
142
+ if version > cls._SCHEMA_VERSION:
143
+ raise ValueError(
144
+ f"TransitionSchema dict has schema version {version}, newer than "
145
+ f"the {cls._SCHEMA_VERSION} this ossify release supports; "
146
+ "upgrade ossify to load it."
147
+ )
148
+ return cls(
149
+ classes=d["classes"],
150
+ transitions={(a, b): cost for a, b, cost in d.get("transitions", [])},
151
+ root_classes=d.get("root_classes"),
152
+ default_cost=d.get("default_cost", 0.0),
153
+ )
154
+
108
155
 
109
156
  def tree_map_decode(
110
157
  parent_array: np.ndarray,
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes