batchlens-bio 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
batchlens/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Metadata-only experimental-design audits. No expression correction."""
2
+
3
+ __version__ = "0.1.0"
batchlens/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from batchlens.cli import main
2
+
3
+ raise SystemExit(main())
batchlens/audit.py ADDED
@@ -0,0 +1,232 @@
1
+ """Evidence-backed findings over experimental units and declared contrasts."""
2
+
3
+ from typing import Any
4
+
5
+ from batchlens import __version__
6
+ from batchlens.config import StudySpec
7
+ from batchlens.design import analyse_design
8
+ from batchlens.metadata import Inputs, deidentify
9
+
10
+ LIMITATION = (
11
+ "Metadata and the declared additive model only. This audit does not establish power, "
12
+ "independence, randomization, absence of unmeasured confounding, or causality."
13
+ )
14
+
15
+
16
+ def audit(inputs: Inputs, spec: StudySpec) -> dict[str, Any]:
17
+ inputs = deidentify(inputs, spec)
18
+ samples, observations, assays = inputs.samples, inputs.observations, inputs.assays
19
+ findings: list[dict[str, Any]] = []
20
+
21
+ def finding(rule: str, severity: str, message: str, refs: list[str], next_step: str) -> None:
22
+ findings.append(
23
+ {
24
+ "rule_id": rule,
25
+ "rule_version": "1.0",
26
+ "severity": severity,
27
+ "scope": "declared study",
28
+ "message": message,
29
+ "evidence_refs": refs,
30
+ "limitations": LIMITATION,
31
+ "suggested_next_step": next_step,
32
+ }
33
+ )
34
+
35
+ units: list[dict[str, Any]] = [
36
+ {
37
+ "unit_alias": unit,
38
+ "samples": len(group),
39
+ "sample_aliases": sorted(group[spec.sample_id]),
40
+ "target_levels": sorted(set(group[spec.target])),
41
+ }
42
+ for unit, group in samples.groupby(spec.unit_id, sort=True)
43
+ ]
44
+ unsupported: list[str] = []
45
+ if any(unit["samples"] > 1 for unit in units):
46
+ finding(
47
+ "BL-UNIT-001",
48
+ "info",
49
+ "Repeated samples from an experimental unit are present.",
50
+ ["/units"],
51
+ "Verify the declared experimental unit and repeated-sampling structure.",
52
+ )
53
+ if spec.design_mode == "independent" and any(unit["samples"] != 1 for unit in units):
54
+ unsupported.append("independent mode requires exactly one sample per experimental unit")
55
+ if spec.design_mode == "paired":
56
+ expected = set(spec.variables[spec.target].levels or [])
57
+ for _, group in samples.groupby(spec.unit_id):
58
+ if len(group) != 2 or set(group[spec.target]) != expected:
59
+ unsupported.append(
60
+ "paired mode needs one sample at each target level for every unit"
61
+ )
62
+ break
63
+
64
+ target_counts = []
65
+ for level in spec.variables[spec.target].levels or []:
66
+ subset = samples[samples[spec.target] == level]
67
+ count = int(subset[spec.unit_id].nunique())
68
+ target_counts.append({"target_level": level, "samples": len(subset), "units": count})
69
+ if count < 2:
70
+ finding(
71
+ "BL-UNIT-002",
72
+ "warning",
73
+ f"Target level {level!r} has fewer than two units.",
74
+ ["/target_counts"],
75
+ "Review biological replication; this is not a power analysis.",
76
+ )
77
+ coverage = []
78
+ for name, var in spec.variables.items():
79
+ if var.role != "batch":
80
+ continue
81
+ for target in spec.variables[spec.target].levels or []:
82
+ for batch in var.levels or []:
83
+ subset = samples[(samples[spec.target] == target) & (samples[name] == batch)]
84
+ coverage.append(
85
+ {
86
+ "batch_variable": name,
87
+ "batch_level": batch,
88
+ "target_level": target,
89
+ "samples": len(subset),
90
+ "units": int(subset[spec.unit_id].nunique()),
91
+ }
92
+ )
93
+ if any(row["samples"] == 0 for row in coverage):
94
+ finding(
95
+ "BL-COVER-001",
96
+ "warning",
97
+ "Some target-by-batch combinations have no samples.",
98
+ ["/coverage"],
99
+ "Review coverage. An empty combination alone is not non-estimability.",
100
+ )
101
+
102
+ observation_coverage = []
103
+ if observations is not None:
104
+ lookup = samples.set_index(spec.sample_id)
105
+ observed = set(observations[spec.sample_id])
106
+ absent = sorted(set(samples[spec.sample_id]) - observed)
107
+ if absent:
108
+ finding(
109
+ "BL-COVER-002",
110
+ "warning",
111
+ f"{len(absent)} samples have no linked observations.",
112
+ ["/counts"],
113
+ "Confirm whether the observation export is intentionally partial.",
114
+ )
115
+ for column in ["cell_type", "region"]:
116
+ if column not in observations:
117
+ continue
118
+ for label, group in observations.groupby(column, sort=True):
119
+ for target in spec.variables[spec.target].levels or []:
120
+ selected = group[group[spec.sample_id].map(lookup[spec.target]) == target]
121
+ observation_coverage.append(
122
+ {
123
+ "annotation": column,
124
+ "label": label,
125
+ "target_level": target,
126
+ "observations": len(selected),
127
+ "samples": int(selected[spec.sample_id].nunique()),
128
+ "units": int(
129
+ selected[spec.sample_id].map(lookup[spec.unit_id]).nunique()
130
+ ),
131
+ }
132
+ )
133
+
134
+ assay_summary: dict[str, Any] = {"links": 0, "assays": 0, "multi_batch_samples": []}
135
+ if assays is not None:
136
+ assay_summary.update(links=len(assays), assays=int(assays.assay_id.nunique()))
137
+ for name, var in spec.variables.items():
138
+ if var.role != "batch" or name not in assays:
139
+ continue
140
+ for sample, group in assays.groupby(spec.sample_id, sort=True):
141
+ if group[name].nunique() > 1:
142
+ assay_summary["multi_batch_samples"].append(
143
+ {
144
+ "sample_alias": sample,
145
+ "batch_variable": name,
146
+ "batch_levels": sorted(set(group[name])),
147
+ }
148
+ )
149
+ for column in ["slide_id", "section_id"]:
150
+ if column in assays:
151
+ assay_summary[column + "_count"] = int(assays[column].nunique())
152
+ if assay_summary["multi_batch_samples"]:
153
+ unsupported.append(
154
+ "some samples span multiple batches; sample-level batch is not unique"
155
+ )
156
+
157
+ matrix = None
158
+ if unsupported:
159
+ comparisons = [{**c.model_dump(), "status": "NOT_ASSESSED"} for c in spec.contrasts]
160
+ finding(
161
+ "BL-SUPPORT-001",
162
+ "warning",
163
+ "; ".join(unsupported),
164
+ ["/not_assessed"],
165
+ "Review the analysis unit/model explicitly; no samples are automatically dropped.",
166
+ )
167
+ else:
168
+ matrix, comparisons = analyse_design(samples, spec)
169
+ if matrix["rank"] < matrix["n_columns"]:
170
+ finding(
171
+ "BL-DESIGN-001",
172
+ "warning",
173
+ "The declared design matrix is rank deficient.",
174
+ ["/matrix_diagnostics/dependencies"],
175
+ "Inspect the dependencies and each contrast separately; do not drop batch blindly.",
176
+ )
177
+ if matrix["numerically_sensitive"]:
178
+ finding(
179
+ "BL-NUMERIC-001",
180
+ "warning",
181
+ "The design is nearly singular and sensitive to tolerance.",
182
+ ["/matrix_diagnostics"],
183
+ "Review covariate redundancy and the numeric diagnostics.",
184
+ )
185
+ for index, comparison in enumerate(comparisons):
186
+ if comparison["status"] == "NON_ESTIMABLE":
187
+ finding(
188
+ "BL-CONTRAST-001",
189
+ "critical",
190
+ f"{comparison['id']}: the requested contrast is not estimable in this model.",
191
+ [f"/contrasts/{index}", "/matrix_diagnostics/dependencies"],
192
+ "Discuss crossed sampling or a revised scientific question; correction cannot "
193
+ "supply missing design information.",
194
+ )
195
+
196
+ findings.sort(
197
+ key=lambda f: (
198
+ {"critical": 0, "warning": 1, "info": 2}[f["severity"]],
199
+ f["rule_id"],
200
+ f["message"],
201
+ )
202
+ )
203
+ return {
204
+ "schema_version": "1.0",
205
+ "tool_version": __version__,
206
+ "ruleset_version": "1.0",
207
+ "run_status": "completed",
208
+ "declared_design": spec.model_dump(),
209
+ "counts": {
210
+ "experimental_units": len(units),
211
+ "samples": len(samples),
212
+ "observations": 0 if observations is None else len(observations),
213
+ },
214
+ "units": units,
215
+ "target_counts": target_counts,
216
+ "coverage": coverage,
217
+ "observation_coverage": observation_coverage,
218
+ "assay_summary": assay_summary,
219
+ "matrix_diagnostics": matrix,
220
+ "contrasts": comparisons,
221
+ "findings": findings,
222
+ "not_assessed": unsupported,
223
+ "limitations": LIMITATION,
224
+ "sharing_notice": "IDs use aliases. Labels and design values may still identify people; "
225
+ "review every output before sharing. No observation-level data are exported.",
226
+ }
227
+
228
+
229
+ def policy_exit(result: dict[str, Any], fail_on: str) -> int:
230
+ levels = {"none": 99, "critical": 2, "warning": 1}
231
+ severity = {"info": 0, "warning": 1, "critical": 2}
232
+ return 3 if any(severity[f["severity"]] >= levels[fail_on] for f in result["findings"]) else 0
batchlens/cli.py ADDED
@@ -0,0 +1,104 @@
1
+ """Command-line interface. Exit 3 means findings, not execution failure."""
2
+
3
+ import argparse
4
+ import hashlib
5
+ import json
6
+ import sys
7
+ from importlib import resources
8
+ from pathlib import Path
9
+
10
+ from batchlens import __version__
11
+ from batchlens.audit import audit, policy_exit
12
+ from batchlens.config import InputError, read_spec
13
+ from batchlens.metadata import load_inputs
14
+ from batchlens.reporting import write_bundle
15
+
16
+ CASES = [
17
+ "balanced",
18
+ "confounded-time",
19
+ "partial-overlap",
20
+ "redundant-nuisance",
21
+ "paired",
22
+ "spatial-replicates",
23
+ "mixed-assays",
24
+ ]
25
+
26
+
27
+ def parser() -> argparse.ArgumentParser:
28
+ result = argparse.ArgumentParser(
29
+ prog="batchlens",
30
+ description="Audit experimental units and declared batch-confounded contrasts.",
31
+ epilog="Exit codes: 0 completed; 1 internal failure; 2 input/output error; "
32
+ "3 findings threshold. "
33
+ "No expression correction. See docs/input-schema.md and docs/interpretation.md.",
34
+ )
35
+ result.add_argument("--version", action="version", version=f"batchlens {__version__}")
36
+ sub = result.add_subparsers(dest="command", required=True)
37
+ for name in ["validate", "audit", "demo"]:
38
+ command = sub.add_parser(name)
39
+ if name == "demo":
40
+ command.add_argument("--case", choices=CASES, default="confounded-time")
41
+ else:
42
+ command.add_argument(
43
+ "--samples", required=True, type=Path, help="CSV/TSV; unique sample ID"
44
+ )
45
+ command.add_argument(
46
+ "--design", required=True, type=Path, help="Strict YAML model contract"
47
+ )
48
+ command.add_argument("--observations", type=Path, help="Optional cells/spots metadata")
49
+ command.add_argument("--assays", type=Path, help="Optional sample-to-assay links")
50
+ if name != "validate":
51
+ command.add_argument(
52
+ "--out", required=True, type=Path, help="New directory; parent must exist"
53
+ )
54
+ command.add_argument(
55
+ "--fail-on", choices=["critical", "warning", "none"], default="critical"
56
+ )
57
+ return result
58
+
59
+
60
+ def main(argv: list[str] | None = None) -> int:
61
+ args = parser().parse_args(argv)
62
+ try:
63
+ dataset = None
64
+ if args.command == "demo":
65
+ folder = Path(str(resources.files("batchlens").joinpath("resources/demo", args.case)))
66
+ args.samples, args.design = folder / "samples.tsv", folder / "design.yaml"
67
+ args.observations = (
68
+ folder / "observations.tsv" if (folder / "observations.tsv").is_file() else None
69
+ )
70
+ args.assays = folder / "assays.tsv" if (folder / "assays.tsv").is_file() else None
71
+ dataset = {"kind": "SYNTHETIC", "title": args.case}
72
+ spec = read_spec(args.design)
73
+ inputs = load_inputs(args.samples, spec, args.observations, args.assays)
74
+ if args.command == "validate":
75
+ print("Metadata valid. Model support and contrast estimability have not been assessed.")
76
+ return 0
77
+ result = audit(inputs, spec)
78
+ inputs.provenance["design"] = {
79
+ "sha256": hashlib.sha256(args.design.read_bytes()).hexdigest(),
80
+ "canonical_sha256": hashlib.sha256(
81
+ json.dumps(spec.model_dump(), sort_keys=True, ensure_ascii=False).encode()
82
+ ).hexdigest(),
83
+ }
84
+ write_bundle(result, inputs.provenance, args.out, dataset)
85
+ print(
86
+ f"BatchLens audit completed: {result['counts']['experimental_units']} "
87
+ "experimental units; "
88
+ f"{result['counts']['samples']} samples"
89
+ )
90
+ for comparison in result["contrasts"]:
91
+ print(f"{comparison['id']}: {comparison['status']}")
92
+ print(f"Report: {args.out / 'report.html'}")
93
+ print(f"JSON: {args.out / 'result.json'}")
94
+ return policy_exit(result, args.fail_on)
95
+ except (InputError, OSError) as exc:
96
+ print(f"Input/output error: {exc}", file=sys.stderr)
97
+ return 2
98
+ except Exception as exc:
99
+ print(
100
+ f"Internal error ({type(exc).__name__}). No successful audit is claimed. "
101
+ "Please report a minimal, de-identified reproducer.",
102
+ file=sys.stderr,
103
+ )
104
+ return 1
batchlens/config.py ADDED
@@ -0,0 +1,132 @@
1
+ """A bounded, declarative model contract; no formula evaluation."""
2
+
3
+ from pathlib import Path
4
+ from typing import Any, Literal, Self
5
+
6
+ import yaml
7
+ from pydantic import BaseModel, ConfigDict, Field, ValidationError, model_validator
8
+
9
+
10
+ class InputError(ValueError):
11
+ """An actionable input problem, distinct from a scientific finding."""
12
+
13
+
14
+ class StrictModel(BaseModel):
15
+ model_config = ConfigDict(extra="forbid", strict=True)
16
+
17
+
18
+ class Variable(StrictModel):
19
+ role: Literal["target", "batch", "covariate"]
20
+ type: Literal["categorical", "numeric"]
21
+ levels: list[str] | None = None
22
+ reference: str | None = None
23
+
24
+ @model_validator(mode="after")
25
+ def validate_type(self) -> Self:
26
+ if self.type == "categorical":
27
+ if not self.levels or len(set(self.levels)) != len(self.levels):
28
+ raise ValueError("Categorical levels must be a nonempty unique list of strings")
29
+ if any(not x.strip() or x != x.strip() for x in self.levels):
30
+ raise ValueError("Levels must be nonempty and have no surrounding whitespace")
31
+ if self.reference not in self.levels:
32
+ raise ValueError("Categorical reference must belong to levels")
33
+ elif self.levels is not None or self.reference is not None:
34
+ raise ValueError("Numeric variables cannot declare levels or reference")
35
+ return self
36
+
37
+
38
+ class Contrast(StrictModel):
39
+ id: str = Field(min_length=1, max_length=200)
40
+ numerator: str
41
+ denominator: str
42
+
43
+
44
+ class StudySpec(StrictModel):
45
+ schema_version: Literal["1.0"]
46
+ design_mode: Literal["independent", "paired"]
47
+ sample_id: str
48
+ unit_id: str
49
+ variables: dict[str, Variable]
50
+ target: str
51
+ adjust_for: list[str]
52
+ contrasts: list[Contrast] = Field(min_length=1, max_length=100)
53
+
54
+ @model_validator(mode="after")
55
+ def validate_model(self) -> Self:
56
+ names = [self.sample_id, self.unit_id, *self.variables]
57
+ if any(not n.strip() or n != n.strip() for n in names):
58
+ raise ValueError("Column names must be nonempty with no surrounding whitespace")
59
+ if self.sample_id == self.unit_id or {self.sample_id, self.unit_id} & self.variables.keys():
60
+ raise ValueError("ID columns and model variables must have distinct names")
61
+ reserved = {"observation_id", "assay_id", "slide_id", "section_id", "cell_type", "region"}
62
+ if {self.sample_id, self.unit_id} & reserved:
63
+ raise ValueError(
64
+ "Sample/unit ID column names cannot reuse optional-table interface names"
65
+ )
66
+ if {"observation_id", "assay_id"} & self.variables.keys():
67
+ raise ValueError("Observation and assay identity columns cannot be model variables")
68
+ target = self.variables.get(self.target)
69
+ if target is None or target.role != "target" or target.type != "categorical":
70
+ raise ValueError("target must name a categorical variable with role target")
71
+ if sum(v.role == "target" for v in self.variables.values()) != 1:
72
+ raise ValueError("Exactly one target variable is supported")
73
+ if len(target.levels or []) < 2:
74
+ raise ValueError("target needs at least two levels")
75
+ if len(set(self.adjust_for)) != len(self.adjust_for):
76
+ raise ValueError("adjust_for must not contain duplicates")
77
+ if set(self.adjust_for) != set(self.variables) - {self.target}:
78
+ raise ValueError(
79
+ "adjust_for must contain every declared non-target variable exactly once"
80
+ )
81
+ if not any(v.role == "batch" for v in self.variables.values()):
82
+ raise ValueError("Declare at least one batch variable")
83
+ if any(v.role == "batch" and v.type != "categorical" for v in self.variables.values()):
84
+ raise ValueError("Batch variables must be categorical in v0.1")
85
+ if self.design_mode == "paired" and len(target.levels or []) != 2:
86
+ raise ValueError("paired mode requires exactly two target levels")
87
+ if len({c.id for c in self.contrasts}) != len(self.contrasts):
88
+ raise ValueError("Contrast IDs must be unique")
89
+ for contrast in self.contrasts:
90
+ if contrast.numerator == contrast.denominator:
91
+ raise ValueError("Contrast numerator and denominator must differ")
92
+ if not {contrast.numerator, contrast.denominator} <= set(target.levels or []):
93
+ raise ValueError("Contrast levels must be declared target levels")
94
+ return self
95
+
96
+
97
+ class UniqueSafeLoader(yaml.SafeLoader):
98
+ """Reject duplicate keys instead of silently changing the declared design."""
99
+
100
+
101
+ def _mapping(loader: UniqueSafeLoader, node: yaml.MappingNode) -> dict[Any, Any]:
102
+ result: dict[Any, Any] = {}
103
+ for key_node, value_node in node.value:
104
+ key = loader.construct_object(key_node, deep=True)
105
+ if not isinstance(key, str):
106
+ raise InputError("YAML mapping keys must be strings")
107
+ if key in result:
108
+ raise InputError(f"Duplicate YAML key at line {key_node.start_mark.line + 1}")
109
+ result[key] = loader.construct_object(value_node, deep=True)
110
+ return result
111
+
112
+
113
+ UniqueSafeLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, _mapping)
114
+
115
+
116
+ def read_spec(path: Path) -> StudySpec:
117
+ try:
118
+ raw = path.read_text(encoding="utf-8-sig")
119
+ if len(raw) > 1_000_000:
120
+ raise InputError("Design configuration exceeds 1 MB")
121
+ # Aliases are unnecessary here and can create recursive/expanding documents.
122
+ if any(isinstance(t, (yaml.AliasToken, yaml.AnchorToken)) for t in yaml.scan(raw)):
123
+ raise InputError("YAML anchors and aliases are unsupported; use explicit values")
124
+ return StudySpec.model_validate(yaml.load(raw, Loader=UniqueSafeLoader))
125
+ except ValidationError as exc:
126
+ details = "; ".join(
127
+ f"{'.'.join(map(str, e['loc'])) or 'design'}: {e['msg']}"
128
+ for e in exc.errors(include_input=False, include_url=False)
129
+ )
130
+ raise InputError(details) from exc
131
+ except (OSError, UnicodeError, yaml.YAMLError) as exc:
132
+ raise InputError(f"Cannot read design ({type(exc).__name__}); check path and YAML") from exc
batchlens/design.py ADDED
@@ -0,0 +1,105 @@
1
+ """Linear-model estimability, without fitting outcomes or interpreting causality."""
2
+
3
+ from typing import Any
4
+
5
+ import numpy as np
6
+ import pandas as pd
7
+
8
+ from batchlens.config import InputError, StudySpec
9
+
10
+
11
+ def analyse_design(samples: pd.DataFrame, spec: StudySpec) -> tuple[dict[str, Any], list[dict]]:
12
+ n_columns = 1 + sum(
13
+ len(v.levels or []) - 1 if v.type == "categorical" else 1 for v in spec.variables.values()
14
+ )
15
+ if spec.design_mode == "paired":
16
+ n_columns += int(samples[spec.unit_id].nunique()) - 1
17
+ if n_columns > 256 or len(samples) > 100_000:
18
+ raise InputError("Design exceeds v0.1 limit: 100,000 samples or 256 encoded columns")
19
+ vectors = [np.ones(len(samples), dtype=float)]
20
+ columns: list[dict[str, Any]] = [{"name": "Intercept", "kind": "intercept"}]
21
+ target_indices: dict[str, int] = {}
22
+ terms = [spec.target, *spec.adjust_for]
23
+ if spec.design_mode == "paired":
24
+ terms.append(spec.unit_id)
25
+ for term in terms:
26
+ var = spec.variables.get(term)
27
+ if var is None or var.type == "categorical":
28
+ levels = sorted(set(samples[term])) if var is None else (var.levels or [])
29
+ reference = levels[0] if var is None else var.reference
30
+ for level in levels:
31
+ if level == reference:
32
+ continue
33
+ if term == spec.target:
34
+ target_indices[level] = len(vectors)
35
+ vectors.append((samples[term] == level).to_numpy(dtype=float))
36
+ columns.append(
37
+ {"name": term, "kind": "categorical", "level": level, "reference": reference}
38
+ )
39
+ else:
40
+ values = samples[term].to_numpy(dtype=float)
41
+ # Rescale before centering to avoid overflow for large finite input values.
42
+ magnitude = float(np.max(np.abs(values))) or 1.0
43
+ scaled = values / magnitude
44
+ center = float(scaled.mean())
45
+ scale = float(np.std(scaled)) or 1.0
46
+ vectors.append((scaled - center) / scale)
47
+ columns.append(
48
+ {
49
+ "name": term,
50
+ "kind": "numeric",
51
+ "magnitude": magnitude,
52
+ "scaled_center": center,
53
+ "scaled_std": scale,
54
+ }
55
+ )
56
+ if len(vectors) > 256 or len(samples) > 100_000:
57
+ raise InputError("Design exceeds v0.1 limit: 100,000 samples or 256 encoded columns")
58
+ matrix = np.column_stack(vectors)
59
+ n, p = matrix.shape
60
+ _, singular, vt = np.linalg.svd(matrix, full_matrices=n < p)
61
+ tolerance = float(np.finfo(float).eps * max(n, p) * singular[0])
62
+ rank = int(np.count_nonzero(singular > tolerance))
63
+ basis = vt[:rank]
64
+ dependencies = []
65
+ for vector in vt[rank:]:
66
+ vector = vector / np.max(np.abs(vector))
67
+ first = next((value for value in vector if abs(value) > 1e-10), 1)
68
+ if first < 0:
69
+ vector = -vector
70
+ dependencies.append([round(float(v), 12) for v in vector])
71
+ condition = float(singular[0] / singular[rank - 1]) if rank else None
72
+ sensitive = condition is not None and condition > 1e8
73
+ comparisons = []
74
+ for contrast in spec.contrasts:
75
+ vector = np.zeros(p)
76
+ for level, sign in [(contrast.numerator, 1), (contrast.denominator, -1)]:
77
+ if level in target_indices:
78
+ vector[target_indices[level]] += sign
79
+ residual = float(
80
+ np.linalg.norm(vector - basis.T @ (basis @ vector)) / max(1, np.linalg.norm(vector))
81
+ )
82
+ comparisons.append(
83
+ {
84
+ **contrast.model_dump(),
85
+ "status": "ESTIMABLE" if residual <= 1e-10 else "NON_ESTIMABLE",
86
+ "vector": vector.tolist(),
87
+ "row_space_residual": residual,
88
+ "tolerance": 1e-10,
89
+ }
90
+ )
91
+ return {
92
+ "n_rows": n,
93
+ "n_columns": p,
94
+ "rank": rank,
95
+ "row_residual_df": n - rank,
96
+ "columns": columns,
97
+ "encoded_rows": matrix.tolist(),
98
+ "sample_aliases": samples[spec.sample_id].tolist(),
99
+ "singular_values": singular.tolist(),
100
+ "rank_tolerance": tolerance,
101
+ "condition_on_nonzero_subspace": condition,
102
+ "numerically_sensitive": sensitive,
103
+ "dependencies": dependencies,
104
+ "assumptions": "Additive fixed-effects design; estimability is not power or causality.",
105
+ }, comparisons