ossify 0.2.3__tar.gz → 0.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ossify-0.2.3 → ossify-0.2.5}/PKG-INFO +1 -1
- {ossify-0.2.3 → ossify-0.2.5}/pyproject.toml +2 -2
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/__init__.py +1 -1
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartments.py +232 -10
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/file_io.py +101 -12
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/structured_prediction.py +47 -0
- {ossify-0.2.3 → ossify-0.2.5}/.github/workflows/mkdocs_publish.yml +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/.github/workflows/python-package.yml +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/.gitignore +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/.pre-commit-config.yaml +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/LICENSE +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/README.md +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/mkdocs.yml +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/__init__.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/base.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/graph.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/mapping.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/mesh.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/morph.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/points.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/_sync/table.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/algorithms.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/base.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartment_models/minnie65_ds15_us0_bd0.json +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartment_models/v1dd_ds15_us0_bd0.json +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartment_models/xgb/minnie65_ds15_us0_bd0.ubj +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/compartment_models/xgb/v1dd_ds15_us0_bd0.ubj +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/data_layers.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/graph_functions.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/plot.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/plot3d.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/plot_utils.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/sync_classes.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/translate.py +0 -0
- {ossify-0.2.3 → ossify-0.2.5}/src/ossify/utils.py +0 -0
|
@@ -6,7 +6,7 @@ build-backend = "hatchling.build"
|
|
|
6
6
|
allow-direct-references = true
|
|
7
7
|
[project]
|
|
8
8
|
name = "ossify"
|
|
9
|
-
version = "0.2.
|
|
9
|
+
version = "0.2.5"
|
|
10
10
|
description = "Mesh and skeleton analysis"
|
|
11
11
|
readme = "README.md"
|
|
12
12
|
requires-python = ">=3.11"
|
|
@@ -95,7 +95,7 @@ default-groups = ["dev", "docs", "lint", "profile", "viz"]
|
|
|
95
95
|
|
|
96
96
|
|
|
97
97
|
[tool.bumpversion]
|
|
98
|
-
current_version = "0.2.
|
|
98
|
+
current_version = "0.2.5"
|
|
99
99
|
parse = "(?P<major>\\d+)\\.(?P<minor>\\d+)\\.(?P<patch>\\d+)"
|
|
100
100
|
serialize = ["{major}.{minor}.{patch}"]
|
|
101
101
|
regex = false
|
|
@@ -26,6 +26,7 @@ from dataclasses import dataclass, field
|
|
|
26
26
|
from typing import Callable, List, Literal, Optional, Union
|
|
27
27
|
|
|
28
28
|
import numpy as np
|
|
29
|
+
import orjson
|
|
29
30
|
import pandas as pd
|
|
30
31
|
from scipy import sparse
|
|
31
32
|
|
|
@@ -38,6 +39,7 @@ from .algorithms import (
|
|
|
38
39
|
from .base import Cell
|
|
39
40
|
from .structured_prediction import (
|
|
40
41
|
TransitionSchema,
|
|
42
|
+
_ossify_version,
|
|
41
43
|
absorb_small_compartments,
|
|
42
44
|
decode_tree,
|
|
43
45
|
)
|
|
@@ -954,6 +956,10 @@ class ProbaVertexModel(SkeletonVertexModel):
|
|
|
954
956
|
See :func:`make_skel_prop_df`. Defaults to :data:`DEFAULT_FEATURE_SPEC`.
|
|
955
957
|
"""
|
|
956
958
|
|
|
959
|
+
#: Bumped when :meth:`to_dict`'s shape changes incompatibly. Independent of
|
|
960
|
+
#: the ossify package version.
|
|
961
|
+
_SCHEMA_VERSION = 1
|
|
962
|
+
|
|
957
963
|
def __init__(
|
|
958
964
|
self,
|
|
959
965
|
estimator,
|
|
@@ -974,6 +980,11 @@ class ProbaVertexModel(SkeletonVertexModel):
|
|
|
974
980
|
self._feature_columns = list(feature_columns)
|
|
975
981
|
self._smooth_alpha = smooth_alpha
|
|
976
982
|
self._unary_clip = float(unary_clip)
|
|
983
|
+
#: The ``config`` argument the instance was built from via
|
|
984
|
+
#: :meth:`from_config` (a bundled model name, a user config dict, or a
|
|
985
|
+
#: config file path); ``None`` when built directly from an estimator.
|
|
986
|
+
#: Lets :meth:`to_dict` round-trip without re-serializing the estimator.
|
|
987
|
+
self._config_ref: Optional[Union[dict, str]] = None
|
|
977
988
|
super().__init__(
|
|
978
989
|
downstream_hops=downstream_hops,
|
|
979
990
|
upstream_hops=upstream_hops,
|
|
@@ -981,6 +992,12 @@ class ProbaVertexModel(SkeletonVertexModel):
|
|
|
981
992
|
feature_spec=feature_spec,
|
|
982
993
|
)
|
|
983
994
|
|
|
995
|
+
@staticmethod
|
|
996
|
+
def _resolve_config(config: Optional[Union[dict, str]]) -> dict:
|
|
997
|
+
if not isinstance(config, dict):
|
|
998
|
+
config = load_model_config(config, dir=None)
|
|
999
|
+
return config
|
|
1000
|
+
|
|
984
1001
|
@classmethod
|
|
985
1002
|
def from_config(
|
|
986
1003
|
cls,
|
|
@@ -997,22 +1014,125 @@ class ProbaVertexModel(SkeletonVertexModel):
|
|
|
997
1014
|
the config's ``spread_alpha``; ``unary_clip`` to the config's ``unary_clip``
|
|
998
1015
|
key, falling back to :data:`DEFAULT_UNARY_CLIP`.
|
|
999
1016
|
"""
|
|
1000
|
-
if not
|
|
1001
|
-
|
|
1017
|
+
config_ref = config if config is not None else DEFAULT_MODEL
|
|
1018
|
+
resolved = cls._resolve_config(config)
|
|
1002
1019
|
if smooth_alpha is None:
|
|
1003
|
-
smooth_alpha =
|
|
1020
|
+
smooth_alpha = resolved.get("spread_alpha")
|
|
1004
1021
|
if unary_clip is None:
|
|
1005
|
-
unary_clip =
|
|
1006
|
-
|
|
1007
|
-
_load_xgboost_classifier(
|
|
1008
|
-
|
|
1009
|
-
downstream_hops=
|
|
1010
|
-
upstream_hops=
|
|
1011
|
-
bidirectional_hops=
|
|
1022
|
+
unary_clip = resolved.get("unary_clip", DEFAULT_UNARY_CLIP)
|
|
1023
|
+
obj = cls(
|
|
1024
|
+
_load_xgboost_classifier(resolved["model_file"]),
|
|
1025
|
+
resolved["feature_columns"],
|
|
1026
|
+
downstream_hops=resolved.get("downstream_hops", 0),
|
|
1027
|
+
upstream_hops=resolved.get("upstream_hops", 0),
|
|
1028
|
+
bidirectional_hops=resolved.get("bidirectional_hops", 0),
|
|
1012
1029
|
smooth_alpha=smooth_alpha,
|
|
1013
1030
|
unary_clip=unary_clip,
|
|
1014
1031
|
feature_spec=feature_spec,
|
|
1015
1032
|
)
|
|
1033
|
+
obj._config_ref = config_ref
|
|
1034
|
+
return obj
|
|
1035
|
+
|
|
1036
|
+
def to_dict(self, *, model_file: Optional[Union[str, pathlib.Path]] = None) -> dict:
|
|
1037
|
+
"""Plain-data representation, round-trippable via :meth:`from_dict`.
|
|
1038
|
+
|
|
1039
|
+
When built via :meth:`from_config`, the ``config`` it was given (a
|
|
1040
|
+
bundled model name, config file path, or dict) is stored directly, so
|
|
1041
|
+
reloading re-resolves the same estimator without touching it.
|
|
1042
|
+
|
|
1043
|
+
Otherwise (built directly from an ``estimator``), the estimator has no
|
|
1044
|
+
known source file: pass ``model_file`` to save it there (requires an
|
|
1045
|
+
estimator with a ``save_model(path)`` method, e.g. XGBoost) and a fresh
|
|
1046
|
+
config dict pointing at it is built automatically. Without
|
|
1047
|
+
``model_file``, this raises -- there is nothing to persist besides the
|
|
1048
|
+
estimator itself, and this is deliberately not a pickle.
|
|
1049
|
+
|
|
1050
|
+
Raises
|
|
1051
|
+
------
|
|
1052
|
+
ValueError
|
|
1053
|
+
If ``feature_spec`` is non-default (it may contain callables that
|
|
1054
|
+
cannot be serialized) or the estimator's source can't be
|
|
1055
|
+
determined.
|
|
1056
|
+
"""
|
|
1057
|
+
if self._feature_spec is not None:
|
|
1058
|
+
raise ValueError(
|
|
1059
|
+
"This ProbaVertexModel has a custom feature_spec, which may "
|
|
1060
|
+
"contain callables and cannot be serialized. Reattach it "
|
|
1061
|
+
"explicitly when reloading: "
|
|
1062
|
+
"ProbaVertexModel.from_dict(d, feature_spec=...)."
|
|
1063
|
+
)
|
|
1064
|
+
config_ref = self._config_ref
|
|
1065
|
+
if config_ref is None:
|
|
1066
|
+
if model_file is None:
|
|
1067
|
+
raise ValueError(
|
|
1068
|
+
"This ProbaVertexModel was not built via from_config(...), "
|
|
1069
|
+
"so its estimator has no known source file. Pass "
|
|
1070
|
+
"model_file=<path> to save the fitted estimator there "
|
|
1071
|
+
"(requires an estimator with a save_model(path) method, "
|
|
1072
|
+
"e.g. XGBoost), or save/load the estimator yourself and "
|
|
1073
|
+
"build a config dict for from_dict()."
|
|
1074
|
+
)
|
|
1075
|
+
if not hasattr(self._estimator, "save_model"):
|
|
1076
|
+
raise TypeError(
|
|
1077
|
+
f"estimator of type {type(self._estimator).__name__} has no "
|
|
1078
|
+
"save_model(path) method; save it yourself and pass a "
|
|
1079
|
+
"config dict to from_dict() instead."
|
|
1080
|
+
)
|
|
1081
|
+
self._estimator.save_model(model_file)
|
|
1082
|
+
config_ref = {
|
|
1083
|
+
"model_file": str(model_file),
|
|
1084
|
+
"feature_columns": self._feature_columns,
|
|
1085
|
+
"downstream_hops": self._downstream_hops,
|
|
1086
|
+
"upstream_hops": self._upstream_hops,
|
|
1087
|
+
"bidirectional_hops": self._bidirectional_hops,
|
|
1088
|
+
}
|
|
1089
|
+
elif isinstance(config_ref, dict):
|
|
1090
|
+
config_ref = {
|
|
1091
|
+
k: (str(v) if isinstance(v, pathlib.Path) else v)
|
|
1092
|
+
for k, v in config_ref.items()
|
|
1093
|
+
}
|
|
1094
|
+
return {
|
|
1095
|
+
"type": "ProbaVertexModel",
|
|
1096
|
+
"version": self._SCHEMA_VERSION,
|
|
1097
|
+
"ossify_version": _ossify_version(),
|
|
1098
|
+
"config": config_ref,
|
|
1099
|
+
"smooth_alpha": self._smooth_alpha,
|
|
1100
|
+
"unary_clip": self._unary_clip,
|
|
1101
|
+
}
|
|
1102
|
+
|
|
1103
|
+
@classmethod
|
|
1104
|
+
def from_dict(
|
|
1105
|
+
cls,
|
|
1106
|
+
d: dict,
|
|
1107
|
+
*,
|
|
1108
|
+
feature_spec: Optional[List[FeatureDef]] = None,
|
|
1109
|
+
) -> "ProbaVertexModel":
|
|
1110
|
+
"""Reconstruct a :class:`ProbaVertexModel` from :meth:`to_dict` output.
|
|
1111
|
+
|
|
1112
|
+
Unlike :meth:`from_config`, ``smooth_alpha``/``unary_clip`` are taken
|
|
1113
|
+
as-is from ``d`` (not re-defaulted from the config file), so a
|
|
1114
|
+
deliberate ``smooth_alpha: null`` (no smoothing) round-trips correctly.
|
|
1115
|
+
"""
|
|
1116
|
+
version = d.get("version", 1)
|
|
1117
|
+
if version > cls._SCHEMA_VERSION:
|
|
1118
|
+
raise ValueError(
|
|
1119
|
+
f"ProbaVertexModel dict has schema version {version}, newer "
|
|
1120
|
+
f"than the {cls._SCHEMA_VERSION} this ossify release supports; "
|
|
1121
|
+
"upgrade ossify to load it."
|
|
1122
|
+
)
|
|
1123
|
+
resolved = cls._resolve_config(d["config"])
|
|
1124
|
+
obj = cls(
|
|
1125
|
+
_load_xgboost_classifier(resolved["model_file"]),
|
|
1126
|
+
resolved["feature_columns"],
|
|
1127
|
+
downstream_hops=resolved.get("downstream_hops", 0),
|
|
1128
|
+
upstream_hops=resolved.get("upstream_hops", 0),
|
|
1129
|
+
bidirectional_hops=resolved.get("bidirectional_hops", 0),
|
|
1130
|
+
smooth_alpha=d.get("smooth_alpha"),
|
|
1131
|
+
unary_clip=d.get("unary_clip", DEFAULT_UNARY_CLIP),
|
|
1132
|
+
feature_spec=feature_spec,
|
|
1133
|
+
)
|
|
1134
|
+
obj._config_ref = d["config"]
|
|
1135
|
+
return obj
|
|
1016
1136
|
|
|
1017
1137
|
@property
|
|
1018
1138
|
def estimator(self):
|
|
@@ -1059,6 +1179,11 @@ class ProbaVertexModel(SkeletonVertexModel):
|
|
|
1059
1179
|
|
|
1060
1180
|
# --- Structured compartment labeler -----------------------------------------
|
|
1061
1181
|
|
|
1182
|
+
#: Registry of encoder types :class:`StructuredLabeler` can (de)serialize.
|
|
1183
|
+
#: Extend when a new :class:`SkeletonVertexModel` subclass grows a
|
|
1184
|
+
#: to_dict/from_dict pair.
|
|
1185
|
+
_ENCODER_TYPES = {"ProbaVertexModel": ProbaVertexModel}
|
|
1186
|
+
|
|
1062
1187
|
|
|
1063
1188
|
class StructuredLabeler:
|
|
1064
1189
|
"""Compose an encoder + transition priors into a tree-decoded labeler.
|
|
@@ -1095,6 +1220,10 @@ class StructuredLabeler:
|
|
|
1095
1220
|
this or less.
|
|
1096
1221
|
"""
|
|
1097
1222
|
|
|
1223
|
+
#: Bumped when :meth:`to_dict`'s shape changes incompatibly. Independent of
|
|
1224
|
+
#: the ossify package version.
|
|
1225
|
+
_SCHEMA_VERSION = 1
|
|
1226
|
+
|
|
1098
1227
|
def __init__(
|
|
1099
1228
|
self,
|
|
1100
1229
|
encoder: SkeletonVertexModel,
|
|
@@ -1198,3 +1327,96 @@ class StructuredLabeler:
|
|
|
1198
1327
|
"""
|
|
1199
1328
|
labels, edge_cost = self._decode(cell)
|
|
1200
1329
|
return self._as_labels(labels, return_labels_as), edge_cost
|
|
1330
|
+
|
|
1331
|
+
# --- Serialization -------------------------------------------------------
|
|
1332
|
+
|
|
1333
|
+
def to_dict(self, *, model_file: Optional[Union[str, pathlib.Path]] = None) -> dict:
|
|
1334
|
+
"""Plain-data representation, round-trippable via :meth:`from_dict`.
|
|
1335
|
+
|
|
1336
|
+
Composes :meth:`TransitionSchema.to_dict` and the encoder's own
|
|
1337
|
+
``to_dict`` -- so a labeler built from a bundled model config (the
|
|
1338
|
+
common case) serializes down to a small JSON-safe dict with no model
|
|
1339
|
+
weights inlined. ``model_file`` is forwarded to the encoder's
|
|
1340
|
+
``to_dict`` for the case where its estimator has no known source file
|
|
1341
|
+
(see :meth:`ProbaVertexModel.to_dict`). Includes a ``version`` (this
|
|
1342
|
+
dict shape, bumped on breaking changes) and ``ossify_version`` (the
|
|
1343
|
+
release that wrote it, for provenance/debugging).
|
|
1344
|
+
"""
|
|
1345
|
+
encoder_type = type(self._encoder).__name__
|
|
1346
|
+
if encoder_type not in _ENCODER_TYPES:
|
|
1347
|
+
raise TypeError(
|
|
1348
|
+
f"Serialization is not supported for encoder type "
|
|
1349
|
+
f"{encoder_type!r}; supported types: {sorted(_ENCODER_TYPES)}."
|
|
1350
|
+
)
|
|
1351
|
+
return {
|
|
1352
|
+
"type": "StructuredLabeler",
|
|
1353
|
+
"version": self._SCHEMA_VERSION,
|
|
1354
|
+
"ossify_version": _ossify_version(),
|
|
1355
|
+
"schema": self._schema.to_dict(),
|
|
1356
|
+
"encoder": self._encoder.to_dict(model_file=model_file),
|
|
1357
|
+
"absorb_min_size": self._absorb_min_size,
|
|
1358
|
+
"absorb_min_weight": self._absorb_min_weight,
|
|
1359
|
+
}
|
|
1360
|
+
|
|
1361
|
+
@classmethod
|
|
1362
|
+
def from_dict(
|
|
1363
|
+
cls,
|
|
1364
|
+
d: dict,
|
|
1365
|
+
*,
|
|
1366
|
+
feature_spec: Optional[List[FeatureDef]] = None,
|
|
1367
|
+
) -> "StructuredLabeler":
|
|
1368
|
+
"""Reconstruct a :class:`StructuredLabeler` from :meth:`to_dict` output.
|
|
1369
|
+
|
|
1370
|
+
``feature_spec`` is forwarded to the encoder's ``from_dict`` and only
|
|
1371
|
+
needed when the original encoder was built with a non-default
|
|
1372
|
+
``feature_spec`` (see :meth:`ProbaVertexModel.to_dict`).
|
|
1373
|
+
"""
|
|
1374
|
+
version = d.get("version", 1)
|
|
1375
|
+
if version > cls._SCHEMA_VERSION:
|
|
1376
|
+
raise ValueError(
|
|
1377
|
+
f"StructuredLabeler dict has schema version {version}, newer "
|
|
1378
|
+
f"than the {cls._SCHEMA_VERSION} this ossify release supports; "
|
|
1379
|
+
"upgrade ossify to load it."
|
|
1380
|
+
)
|
|
1381
|
+
encoder_dict = d["encoder"]
|
|
1382
|
+
encoder_type = encoder_dict.get("type")
|
|
1383
|
+
encoder_cls = _ENCODER_TYPES.get(encoder_type)
|
|
1384
|
+
if encoder_cls is None:
|
|
1385
|
+
raise ValueError(
|
|
1386
|
+
f"Unknown or unsupported encoder type: {encoder_type!r}; "
|
|
1387
|
+
f"supported types: {sorted(_ENCODER_TYPES)}."
|
|
1388
|
+
)
|
|
1389
|
+
encoder = encoder_cls.from_dict(encoder_dict, feature_spec=feature_spec)
|
|
1390
|
+
schema = TransitionSchema.from_dict(d["schema"])
|
|
1391
|
+
return cls(
|
|
1392
|
+
encoder,
|
|
1393
|
+
schema,
|
|
1394
|
+
absorb_min_size=d.get("absorb_min_size"),
|
|
1395
|
+
absorb_min_weight=d.get("absorb_min_weight"),
|
|
1396
|
+
)
|
|
1397
|
+
|
|
1398
|
+
def save_config(
|
|
1399
|
+
self,
|
|
1400
|
+
filename: Union[str, pathlib.Path],
|
|
1401
|
+
*,
|
|
1402
|
+
model_file: Optional[Union[str, pathlib.Path]] = None,
|
|
1403
|
+
) -> None:
|
|
1404
|
+
"""Write :meth:`to_dict` to ``filename`` as JSON."""
|
|
1405
|
+
with open(filename, "wb") as f:
|
|
1406
|
+
f.write(
|
|
1407
|
+
orjson.dumps(
|
|
1408
|
+
self.to_dict(model_file=model_file), option=orjson.OPT_INDENT_2
|
|
1409
|
+
)
|
|
1410
|
+
)
|
|
1411
|
+
|
|
1412
|
+
@classmethod
|
|
1413
|
+
def load_config(
|
|
1414
|
+
cls,
|
|
1415
|
+
filename: Union[str, pathlib.Path],
|
|
1416
|
+
*,
|
|
1417
|
+
feature_spec: Optional[List[FeatureDef]] = None,
|
|
1418
|
+
) -> "StructuredLabeler":
|
|
1419
|
+
"""Load a :class:`StructuredLabeler` from a JSON file written by :meth:`save_config`."""
|
|
1420
|
+
with open(filename, "rb") as f:
|
|
1421
|
+
d = orjson.loads(f.read())
|
|
1422
|
+
return cls.from_dict(d, feature_spec=feature_spec)
|
|
@@ -913,6 +913,68 @@ def build_point_cloud(
|
|
|
913
913
|
return cell
|
|
914
914
|
|
|
915
915
|
|
|
916
|
+
def _exact_integer_id_coverage(link_ids: pd.Index, source_ids: pd.Index) -> bool:
|
|
917
|
+
"""Test whether two integer-ID domains are exactly the same set.
|
|
918
|
+
|
|
919
|
+
Both arguments must already be canonicalized ``int64`` pandas ``Index``
|
|
920
|
+
objects. The comparison is order-independent and stays entirely in the
|
|
921
|
+
integer domain -- it never coerces identifiers through ``float64`` -- so it
|
|
922
|
+
is safe for IDs above ``2**53``.
|
|
923
|
+
|
|
924
|
+
Parameters
|
|
925
|
+
----------
|
|
926
|
+
link_ids :
|
|
927
|
+
The linkage column for a candidate source layer.
|
|
928
|
+
source_ids :
|
|
929
|
+
That layer's vertex index.
|
|
930
|
+
|
|
931
|
+
Returns
|
|
932
|
+
-------
|
|
933
|
+
:
|
|
934
|
+
``True`` iff the two contain exactly the same set of IDs.
|
|
935
|
+
"""
|
|
936
|
+
if len(link_ids) != len(source_ids):
|
|
937
|
+
return False
|
|
938
|
+
# ``symmetric_difference`` is empty iff neither side has an ID the other
|
|
939
|
+
# lacks. Using pandas ``Index`` set ops keeps the comparison in int64 and
|
|
940
|
+
# avoids materializing Python ``set`` objects over large ID arrays.
|
|
941
|
+
return len(link_ids.symmetric_difference(source_ids)) == 0
|
|
942
|
+
|
|
943
|
+
|
|
944
|
+
def _linkage_source_error(linkage_pair, link_df, cell, sample_size: int = 10) -> str:
|
|
945
|
+
"""Build a concise diagnostic message for an unresolvable linkage source.
|
|
946
|
+
|
|
947
|
+
Reports, for each endpoint, the layer vertex count, the linkage row count,
|
|
948
|
+
the number of unique linkage IDs, and the missing/extra ID counts, plus a
|
|
949
|
+
small bounded sample of the offending IDs. This replaces the enormous
|
|
950
|
+
pandas ``KeyError`` that a bad ``.loc`` reindex would otherwise raise.
|
|
951
|
+
"""
|
|
952
|
+
lines = [
|
|
953
|
+
f"Could not infer a valid source layer for linkage "
|
|
954
|
+
f"({linkage_pair[0]!r}, {linkage_pair[1]!r}) with {len(link_df)} rows.",
|
|
955
|
+
"A source endpoint must map exactly one linkage row to each of its "
|
|
956
|
+
"vertices (unique, complete coverage). Neither endpoint qualifies:",
|
|
957
|
+
]
|
|
958
|
+
for source in linkage_pair:
|
|
959
|
+
source_ids = canonicalize_ids(
|
|
960
|
+
pd.Index(cell._all_objects[source].vertex_index), name=source
|
|
961
|
+
)
|
|
962
|
+
link_ids = canonicalize_ids(pd.Index(link_df[source]), name=source)
|
|
963
|
+
unique_link_ids = link_ids.unique()
|
|
964
|
+
missing = source_ids.difference(link_ids)
|
|
965
|
+
extra = link_ids.difference(source_ids)
|
|
966
|
+
lines.append(
|
|
967
|
+
f" - {source!r}: vertices={len(source_ids)}, "
|
|
968
|
+
f"link_rows={len(link_ids)}, unique_link_ids={len(unique_link_ids)}, "
|
|
969
|
+
f"missing_source_ids={len(missing)}, extra_link_ids={len(extra)}"
|
|
970
|
+
)
|
|
971
|
+
if len(missing):
|
|
972
|
+
lines.append(f" missing sample: {missing[:sample_size].tolist()}")
|
|
973
|
+
if len(extra):
|
|
974
|
+
lines.append(f" extra sample: {extra[:sample_size].tolist()}")
|
|
975
|
+
return "\n".join(lines)
|
|
976
|
+
|
|
977
|
+
|
|
916
978
|
def build_linkage(
|
|
917
979
|
linkage_pair,
|
|
918
980
|
tf,
|
|
@@ -922,23 +984,50 @@ def build_linkage(
|
|
|
922
984
|
link_df = load_dataframe(f"{prefix}/linkage.feather", tf)
|
|
923
985
|
# Legacy .osy files may store link columns with mixed int64/uint64 dtypes
|
|
924
986
|
# (dtype optimization only downcast signed ints, leaving uint64 untouched).
|
|
925
|
-
# Canonicalize both columns to int64 before
|
|
926
|
-
#
|
|
927
|
-
#
|
|
987
|
+
# Canonicalize both columns to int64 before any label-based reindex below so
|
|
988
|
+
# that lookup cannot collide two IDs above 2**53 through a float coercion of
|
|
989
|
+
# mismatched signed/unsigned keys.
|
|
928
990
|
for col in linkage_pair:
|
|
929
991
|
if col in link_df.columns:
|
|
930
992
|
link_df[col] = canonicalize_ids(link_df[col], name=col)
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
993
|
+
|
|
994
|
+
# The archive sorts the pair names, so serialized column order does not
|
|
995
|
+
# preserve the original source direction. Infer the source by exact
|
|
996
|
+
# source-domain validation rather than row count alone: a layer qualifies
|
|
997
|
+
# as source only when its linkage column has no missing IDs, is unique, has
|
|
998
|
+
# one row per vertex, and covers exactly that layer's vertex index. Row
|
|
999
|
+
# count alone is ambiguous whenever the two layers have equal cardinality
|
|
1000
|
+
# (e.g. a many-to-one graph <-> annotation link where the counts coincide).
|
|
1001
|
+
candidates = []
|
|
1002
|
+
for source, target in (
|
|
1003
|
+
(linkage_pair[0], linkage_pair[1]),
|
|
1004
|
+
(linkage_pair[1], linkage_pair[0]),
|
|
1005
|
+
):
|
|
1006
|
+
source_ids = canonicalize_ids(
|
|
1007
|
+
pd.Index(cell._all_objects[source].vertex_index), name=source
|
|
1008
|
+
)
|
|
1009
|
+
link_ids = canonicalize_ids(pd.Index(link_df[source]), name=source)
|
|
1010
|
+
if (
|
|
1011
|
+
len(link_ids) == len(source_ids)
|
|
1012
|
+
and link_ids.is_unique
|
|
1013
|
+
and _exact_integer_id_coverage(link_ids, source_ids)
|
|
1014
|
+
):
|
|
1015
|
+
candidates.append((source, target))
|
|
1016
|
+
|
|
1017
|
+
if not candidates:
|
|
1018
|
+
raise ValueError(_linkage_source_error(linkage_pair, link_df, cell))
|
|
1019
|
+
|
|
1020
|
+
# Zero candidates raises above. One candidate is the unambiguous source.
|
|
1021
|
+
# Two candidates means a true bijection: both columns are unique and cover
|
|
1022
|
+
# their layers exactly, so the mapping is one-to-one. MorphSync stores every
|
|
1023
|
+
# link in both directions (see ``add_link``), so either choice yields the
|
|
1024
|
+
# same bidirectional link; pick the first deterministically.
|
|
1025
|
+
source_layer, target_layer = candidates[0]
|
|
940
1026
|
|
|
941
1027
|
layer = cell._all_objects[source_layer]
|
|
1028
|
+
# Validation above guarantees the source column is unique and covers exactly
|
|
1029
|
+
# the source vertex index, so this reindex is a total, collision-free
|
|
1030
|
+
# reordering -- it can no longer raise a large KeyError.
|
|
942
1031
|
cell._all_objects[source_layer]._process_linkage(
|
|
943
1032
|
Link(
|
|
944
1033
|
link_df.set_index(source_layer).loc[layer.vertex_index][target_layer],
|
|
@@ -26,6 +26,13 @@ __all__ = [
|
|
|
26
26
|
]
|
|
27
27
|
|
|
28
28
|
|
|
29
|
+
def _ossify_version() -> str:
|
|
30
|
+
"""The installed ossify release, for provenance in serialized dicts."""
|
|
31
|
+
from . import __version__
|
|
32
|
+
|
|
33
|
+
return __version__
|
|
34
|
+
|
|
35
|
+
|
|
29
36
|
class TransitionSchema:
|
|
30
37
|
"""Declarative label set with soft parent->child transition costs.
|
|
31
38
|
|
|
@@ -51,6 +58,10 @@ class TransitionSchema:
|
|
|
51
58
|
label changes over the tree.
|
|
52
59
|
"""
|
|
53
60
|
|
|
61
|
+
#: Bumped when :meth:`to_dict`'s shape changes incompatibly. Independent of
|
|
62
|
+
#: the ossify package version.
|
|
63
|
+
_SCHEMA_VERSION = 1
|
|
64
|
+
|
|
54
65
|
def __init__(
|
|
55
66
|
self,
|
|
56
67
|
classes: Sequence,
|
|
@@ -105,6 +116,42 @@ class TransitionSchema:
|
|
|
105
116
|
mask[self._index[c]] = True
|
|
106
117
|
return mask
|
|
107
118
|
|
|
119
|
+
def to_dict(self) -> dict:
|
|
120
|
+
"""Plain-data representation, round-trippable via :meth:`from_dict`.
|
|
121
|
+
|
|
122
|
+
``transitions`` is written as a list of ``[parent, child, cost]``
|
|
123
|
+
triples rather than a ``{(parent, child): cost}`` dict, since JSON/YAML
|
|
124
|
+
cannot use tuples as object keys. Includes a ``version`` (this dict
|
|
125
|
+
shape, bumped on breaking changes) and ``ossify_version`` (the release
|
|
126
|
+
that wrote it, for provenance/debugging) alongside the parameters.
|
|
127
|
+
"""
|
|
128
|
+
return {
|
|
129
|
+
"type": "TransitionSchema",
|
|
130
|
+
"version": self._SCHEMA_VERSION,
|
|
131
|
+
"ossify_version": _ossify_version(),
|
|
132
|
+
"classes": list(self.classes),
|
|
133
|
+
"transitions": [[a, b, cost] for (a, b), cost in self.transitions.items()],
|
|
134
|
+
"root_classes": list(self.root_classes),
|
|
135
|
+
"default_cost": self.default_cost,
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
@classmethod
|
|
139
|
+
def from_dict(cls, d: dict) -> "TransitionSchema":
|
|
140
|
+
"""Reconstruct a :class:`TransitionSchema` from :meth:`to_dict` output."""
|
|
141
|
+
version = d.get("version", 1)
|
|
142
|
+
if version > cls._SCHEMA_VERSION:
|
|
143
|
+
raise ValueError(
|
|
144
|
+
f"TransitionSchema dict has schema version {version}, newer than "
|
|
145
|
+
f"the {cls._SCHEMA_VERSION} this ossify release supports; "
|
|
146
|
+
"upgrade ossify to load it."
|
|
147
|
+
)
|
|
148
|
+
return cls(
|
|
149
|
+
classes=d["classes"],
|
|
150
|
+
transitions={(a, b): cost for a, b, cost in d.get("transitions", [])},
|
|
151
|
+
root_classes=d.get("root_classes"),
|
|
152
|
+
default_cost=d.get("default_cost", 0.0),
|
|
153
|
+
)
|
|
154
|
+
|
|
108
155
|
|
|
109
156
|
def tree_map_decode(
|
|
110
157
|
parent_array: np.ndarray,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|