protcross 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. {protcross-0.2.2 → protcross-0.2.3}/PKG-INFO +247 -184
  2. {protcross-0.2.2 → protcross-0.2.3}/README.md +245 -182
  3. {protcross-0.2.2 → protcross-0.2.3}/constraints/py310-ci.txt +2 -1
  4. {protcross-0.2.2 → protcross-0.2.3}/pyproject.toml +2 -2
  5. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/__init__.py +1 -1
  6. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/assets.py +89 -12
  7. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/cli/inspect.py +13 -1
  8. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/cli/map_labels.py +1 -1
  9. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/cli/predict.py +0 -20
  10. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/af2.py +0 -3
  11. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/dataset.py +1 -11
  12. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/inspection.py +1 -1
  13. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/label_mapping.py +22 -13
  14. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/preprocess.py +14 -5
  15. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/structure.py +10 -10
  16. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/experiments/multiseed_benchmark.py +5 -5
  17. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/experiments/strategy_search.py +15 -4
  18. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/inference/predictor.py +42 -196
  19. {protcross-0.2.2 → protcross-0.2.3}/tests/test_assets.py +62 -6
  20. {protcross-0.2.2 → protcross-0.2.3}/tests/test_cli.py +26 -35
  21. {protcross-0.2.2 → protcross-0.2.3}/tests/test_docs_drift.py +34 -1
  22. {protcross-0.2.2 → protcross-0.2.3}/tests/test_inference_result.py +77 -157
  23. {protcross-0.2.2 → protcross-0.2.3}/tests/test_label_mapping.py +19 -0
  24. {protcross-0.2.2 → protcross-0.2.3}/tests/test_pca_and_dataset.py +14 -3
  25. {protcross-0.2.2 → protcross-0.2.3}/tests/test_strategy_search.py +37 -0
  26. {protcross-0.2.2 → protcross-0.2.3}/tests/test_structure_and_pdb.py +28 -0
  27. {protcross-0.2.2 → protcross-0.2.3}/tests/test_structure_inspection.py +2 -1
  28. {protcross-0.2.2 → protcross-0.2.3}/LICENSE +0 -0
  29. {protcross-0.2.2 → protcross-0.2.3}/MANIFEST.in +0 -0
  30. {protcross-0.2.2 → protcross-0.2.3}/configs/data/protein_seg.yaml +0 -0
  31. {protcross-0.2.2 → protcross-0.2.3}/configs/model/da_module.yaml +0 -0
  32. {protcross-0.2.2 → protcross-0.2.3}/configs/train.yaml +0 -0
  33. {protcross-0.2.2 → protcross-0.2.3}/configs/trainer/default.yaml +0 -0
  34. {protcross-0.2.2 → protcross-0.2.3}/environment.yml +0 -0
  35. {protcross-0.2.2 → protcross-0.2.3}/examples/6fhu.pdb +0 -0
  36. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/analyze_geometric.py +0 -0
  37. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/eval_dataset.py +0 -0
  38. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/eval_run.py +0 -0
  39. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/eval_utils.py +0 -0
  40. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/get_af2.py +0 -0
  41. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/map_labels-o.py +0 -0
  42. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/map_labels.py +0 -0
  43. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/pdb_uniprot_mapping.json +0 -0
  44. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/preprocess_esm.py +0 -0
  45. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/run_Predict_ProtCross.py +0 -0
  46. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/run_Strategy.py +0 -0
  47. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/run_multiseed_benchmark.py +0 -0
  48. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/sensitivity-cutoff.py +0 -0
  49. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/setup_assets.py +0 -0
  50. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/test_adaptive.py +0 -0
  51. {protcross-0.2.2 → protcross-0.2.3}/reproduction/legacy/train.py +0 -0
  52. {protcross-0.2.2 → protcross-0.2.3}/setup.cfg +0 -0
  53. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/__init__.py +0 -0
  54. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/_compat.py +0 -0
  55. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/assets.py +0 -0
  56. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/cli/__init__.py +0 -0
  57. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/data/__init__.py +0 -0
  58. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/evaluation/__init__.py +0 -0
  59. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/experiments/__init__.py +0 -0
  60. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/inference/__init__.py +0 -0
  61. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/models/__init__.py +0 -0
  62. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/models/backbones/__init__.py +0 -0
  63. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/models/heads/__init__.py +0 -0
  64. {protcross-0.2.2 → protcross-0.2.3}/src/evopoint_da/training/__init__.py +0 -0
  65. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/__main__.py +0 -0
  66. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/cli/__init__.py +0 -0
  67. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/cli/download_af2.py +0 -0
  68. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/cli/main.py +0 -0
  69. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/cli/preprocess.py +0 -0
  70. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/cli/setup_assets.py +0 -0
  71. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/cli/train.py +0 -0
  72. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/configs/__init__.py +0 -0
  73. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/configs/data/protein_seg.yaml +0 -0
  74. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/configs/model/da_module.yaml +0 -0
  75. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/configs/train.yaml +0 -0
  76. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/configs/trainer/default.yaml +0 -0
  77. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/__init__.py +0 -0
  78. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/components.py +0 -0
  79. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/datamodule.py +0 -0
  80. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/esm.py +0 -0
  81. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/data/pca.py +0 -0
  82. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/evaluation/__init__.py +0 -0
  83. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/evaluation/adaptive.py +0 -0
  84. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/evaluation/metrics.py +0 -0
  85. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/experiments/__init__.py +0 -0
  86. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/inference/__init__.py +0 -0
  87. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/inference/pdb.py +0 -0
  88. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/models/__init__.py +0 -0
  89. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/models/backbones/__init__.py +0 -0
  90. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/models/backbones/pointnet2.py +0 -0
  91. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/models/domain_weights.py +0 -0
  92. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/models/heads/__init__.py +0 -0
  93. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/models/heads/classifier.py +0 -0
  94. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/models/module.py +0 -0
  95. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/training/__init__.py +0 -0
  96. {protcross-0.2.2 → protcross-0.2.3}/src/protcross/training/run.py +0 -0
  97. {protcross-0.2.2 → protcross-0.2.3}/src/protcross.egg-info/SOURCES.txt +0 -0
  98. {protcross-0.2.2 → protcross-0.2.3}/tests/conftest.py +0 -0
  99. {protcross-0.2.2 → protcross-0.2.3}/tests/test_checkpoint_smoke.py +0 -0
  100. {protcross-0.2.2 → protcross-0.2.3}/tests/test_legacy_archive.py +0 -0
  101. {protcross-0.2.2 → protcross-0.2.3}/tests/test_metrics.py +0 -0
  102. {protcross-0.2.2 → protcross-0.2.3}/tests/test_packaging_and_compat.py +0 -0
  103. {protcross-0.2.2 → protcross-0.2.3}/tests/test_pdb_output_preservation.py +0 -0
  104. {protcross-0.2.2 → protcross-0.2.3}/tests/test_pointnet2_fallback.py +0 -0
@@ -1,11 +1,11 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: protcross
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Domain-adaptive protein point-cloud binding-site prediction.
5
5
  Author: Shuyu Zhong, Yuying Jiang
6
6
  License-Expression: MIT
7
7
  Project-URL: Publication, https://doi.org/10.1021/acs.jcim.5c03224
8
- Project-URL: Documentation, https://github.com/GeraltZeroZhong/ProtCross/blob/v0.2.2/README.md
8
+ Project-URL: Documentation, https://github.com/GeraltZeroZhong/ProtCross/blob/v0.2.3/README.md
9
9
  Project-URL: Repository, https://github.com/GeraltZeroZhong/ProtCross
10
10
  Project-URL: Issues, https://github.com/GeraltZeroZhong/ProtCross/issues
11
11
  Classifier: Development Status :: 3 - Alpha
@@ -52,7 +52,7 @@ Dynamic: license-file
52
52
  [![PyPI](https://img.shields.io/pypi/v/protcross?label=PyPI&color=0f766e)](https://pypi.org/project/protcross/)
53
53
  [![Windows Desktop](https://img.shields.io/badge/Windows-10%2F11%20x64-0078d4?logo=windows11&logoColor=white)](https://github.com/GeraltZeroZhong/ProtCross/releases)
54
54
  [![macOS Desktop](https://img.shields.io/badge/macOS-12%2B%20Apple%20Silicon-111827?logo=apple&logoColor=white)](https://github.com/GeraltZeroZhong/ProtCross/releases)
55
- [![Version](https://img.shields.io/badge/version-0.2.2-2563eb)](#version-history)
55
+ [![Version](https://img.shields.io/badge/version-0.2.3-2563eb)](#version-history)
56
56
  [![Python](https://img.shields.io/badge/python-3.10-3776ab)](https://www.python.org/)
57
57
  [![License](https://img.shields.io/badge/license-MIT-16a34a)](LICENSE)
58
58
  [![Paper](https://img.shields.io/badge/DOI-10.1021%2Facs.jcim.5c03224-ca8a04)](https://doi.org/10.1021/acs.jcim.5c03224)
@@ -82,11 +82,17 @@ poses and virtual-screening enrichment
82
82
 
83
83
  ## Quick start
84
84
 
85
- ProtCross 0.2.2 uses Python 3.10:
85
+ Choose the interface that matches your task:
86
86
 
87
- Review the [ESM-C model terms](https://www.evolutionaryscale.ai/policies/cambrian-non-commercial-license-agreement)
88
- before asset setup. A fresh setup downloads approximately 2.14 GiB of model
89
- weights; subsequent predictions reuse the local asset cache.
87
+ | Goal | Start here |
88
+ | --- | --- |
89
+ | Run a local prediction from a terminal | Install the CLI below |
90
+ | Call ProtCross from a Python workflow | Install the CLI, then open [Python API](#python-api) |
91
+ | Use a guided interface and 3D viewer | Download [ProtCross Desktop](#desktop-application) |
92
+
93
+ ProtCross 0.2.3 requires Python 3.10. This first run installs the prediction
94
+ dependencies, prepares the managed model assets, checks a structure, and writes
95
+ one result package:
90
96
 
91
97
  ```bash
92
98
  python3.10 -m venv .venv
@@ -99,6 +105,10 @@ protcross inspect input.pdb
99
105
  protcross predict input.pdb --out-dir protcross-results
100
106
  ```
101
107
 
108
+ Review the [ESM-C model terms](https://www.evolutionaryscale.ai/policies/cambrian-non-commercial-license-agreement)
109
+ before recording acceptance. Initial asset setup downloads approximately
110
+ 2.14 GiB; later predictions reuse the verified local cache.
111
+
102
112
  The prediction command creates:
103
113
 
104
114
  ```text
@@ -109,10 +119,10 @@ protcross-results/
109
119
  └── input.protcross.summary.json
110
120
  ```
111
121
 
112
- Windows x64 and macOS Apple Silicon Desktop builds are available through
113
- [GitHub Releases](https://github.com/GeraltZeroZhong/ProtCross/releases). The
114
- Desktop workflow installs its local runtime, manages assets, inspects inputs,
115
- runs single or batch predictions, and displays results with Mol*.
122
+ For a graphical workflow, install the Windows x64 or macOS Apple Silicon build
123
+ from [GitHub Releases](https://github.com/GeraltZeroZhong/ProtCross/releases),
124
+ then follow the three readiness steps in **Setup**. Desktop manages its own
125
+ runtime and assets and opens completed predictions in Mol*.
116
126
 
117
127
  ## Contents
118
128
 
@@ -121,11 +131,11 @@ runs single or batch predictions, and displays results with Mol*.
121
131
  - [Installation](#installation)
122
132
  - [Run predictions](#run-predictions)
123
133
  - [Output package](#output-package)
124
- - [Model and inference pipeline](#model-and-inference-pipeline)
125
134
  - [Batch inference](#batch-inference)
126
135
  - [Python API](#python-api)
127
136
  - [Assets](#assets)
128
137
  - [Desktop application](#desktop-application)
138
+ - [Model and inference pipeline](#model-and-inference-pipeline)
129
139
  - [Training and development](#training-and-development)
130
140
  - [Troubleshooting](#troubleshooting)
131
141
  - [Version history](#version-history)
@@ -134,14 +144,13 @@ runs single or batch predictions, and displays results with Mol*.
134
144
 
135
145
  ## Highlights
136
146
 
137
- - PDB, mmCIF, and AlphaFold coordinate input
138
- - Per-chain ESM-C 600M embeddings with paired 128-dimensional PCA features
139
- - PointNet++ residue segmentation over centered Cα point clouds
140
- - PDB/mmCIF annotation, extended TSV, cluster JSON, and provenance JSON
141
- - CPU, CUDA, and Apple MPS device selection
142
- - Bounded ESM-C and PointNet++ microbatching for high-throughput inference
143
- - Deterministic pure-PyTorch FPS, radius, and KNN geometry operators
144
- - Local Desktop, unified CLI, and reusable Python API
147
+ - Inspect PDB, mmCIF, and AlphaFold coordinate files before model loading
148
+ - Score residues and rank spatial binding-site clusters with centroids
149
+ - Export annotated coordinates, a full residue table, cluster JSON, and run provenance
150
+ - Run on CPU, CUDA, or Apple MPS through the CLI, Python API, or local Desktop
151
+ - Reuse verified assets and reduced ESM/PCA features across repeated work
152
+ - Process structure collections with bounded ESM-C and PointNet++ microbatches
153
+ - Review persistent Desktop batches and regroup completed results interactively
145
154
 
146
155
  ## Installation
147
156
 
@@ -155,8 +164,8 @@ runs single or batch predictions, and displays results with Mol*.
155
164
  | Desktop development | Node.js 20, Rust 1.88, Tauri 2 system packages |
156
165
 
157
166
  ProtCross declares `python >=3.10,<3.11`. Create a dedicated Python 3.10
158
- environment for installation. The platform commands below install the PyPI
159
- distribution.
167
+ environment, choose the platform command below, and confirm the installed
168
+ version before preparing assets.
160
169
 
161
170
  ### Linux CPU
162
171
 
@@ -216,13 +225,9 @@ py -3.10 -m venv .venv
216
225
  Activate the environment to use the shorter commands shown throughout this
217
226
  README.
218
227
 
219
- ### Development environment
220
-
221
- ```bash
222
- conda env create -f environment.yml
223
- conda activate protcross
224
- python -m pip install -e ".[dev,esm]"
225
- ```
228
+ After installation, run `protcross setup-assets --accept-esm-license` once, or
229
+ open [Assets](#assets) to select another cache location or an existing ESM-C
230
+ file.
226
231
 
227
232
  ## Run predictions
228
233
 
@@ -230,7 +235,7 @@ python -m pip install -e ".[dev,esm]"
230
235
 
231
236
  `protcross inspect` parses coordinate metadata without loading model assets.
232
237
  The repository includes [`examples/6fhu.pdb`](examples/6fhu.pdb) for a first
233
- run.
238
+ check, so input preparation can be tested before the 2.14 GiB ESM-C download.
234
239
 
235
240
  ```bash
236
241
  protcross inspect input.cif
@@ -238,9 +243,10 @@ protcross inspect input.cif --chain A
238
243
  protcross inspect input.cif --json
239
244
  ```
240
245
 
241
- The report includes coordinate models, chains, scorable residues, missing Cα
242
- atoms, modified residues, alternate conformers, coordinate breaks, numbering
243
- gaps, and ESM-C context length.
246
+ Use the report to choose a chain, identify missing Cα atoms or modified
247
+ residues, and check whether a chain exceeds the 1,022-residue ESM-C context.
248
+ `--json` sends successful and failed inspections to stdout as machine-readable
249
+ JSON and uses the process exit code to signal success or failure.
244
250
 
245
251
  ```text
246
252
  Input: examples/6fhu.pdb
@@ -255,11 +261,14 @@ Ready for prediction.
255
261
  ### Predict one structure
256
262
 
257
263
  ```bash
258
- protcross predict input.pdb \
259
- --out-dir results \
260
- --device cpu \
261
- --threshold 0.5 \
262
- --pocket-cluster-cutoff 8.0
264
+ protcross predict input.pdb --out-dir results
265
+ ```
266
+
267
+ Select one chain or an available accelerator when the task requires it:
268
+
269
+ ```bash
270
+ protcross predict input.cif --chain A --out-dir results
271
+ protcross predict input.pdb --device auto --out-dir results
263
272
  ```
264
273
 
265
274
  Common options:
@@ -268,15 +277,19 @@ Common options:
268
277
  | --- | ---: | --- |
269
278
  | `--chain ID` | all chains | Select one author chain ID |
270
279
  | `--device` | `cpu` | Select `cpu`, `cuda`, `cuda:N`, `mps`, or `auto` |
271
- | `--threshold` | `0.5` | Set the strict residue selection threshold |
280
+ | `--threshold` | `0.5` | Select residues with `score > threshold` |
272
281
  | `--pocket-cluster-cutoff` | `8.0` | Set the Cα graph cutoff in Å |
273
282
  | `--max-len` | `1022` | Set the per-chain ESM-C residue limit |
274
- | `--allow-truncation` | disabled | Keep the leading `max_len` residues of each long chain |
283
+ | `--allow-truncation` | disabled | Score the leading `max_len` residues of each long chain |
275
284
  | `--embedding-cache-dir` | unset | Cache reduced ESM/PCA residue features |
276
285
  | `--overwrite` | disabled | Replace an existing result package |
277
286
  | `--offline` | disabled | Restrict asset resolution to local files |
278
287
 
279
- Use explicit paths when integrating ProtCross into a workflow:
288
+ The CLI writes four result files unless `--summary-only` is selected. Progress
289
+ messages go to stderr; the terminal summary goes to stdout. Use `--quiet` when
290
+ another process only needs the files and exit code.
291
+
292
+ Set explicit output paths when a workflow owns the file layout:
280
293
 
281
294
  ```bash
282
295
  protcross predict input.cif \
@@ -328,12 +341,12 @@ protcross COMMAND --help
328
341
 
329
342
  ### Files
330
343
 
331
- | File | Contents |
332
- | --- | --- |
333
- | `input.protcross.pdb` or `.cif` | Input structure with residue scores in B-factor fields |
334
- | `input.protcross.scores.tsv` | Residue identifiers, scores, calls, coordinates, cluster IDs, and ranks |
335
- | `input.protcross.pockets.json` | Thresholded residue clusters and spatial statistics |
336
- | `input.protcross.summary.json` | Parameters, assets, runtime, hashes, warnings, and top-ranked results |
344
+ | Task | File to use | Contents |
345
+ | --- | --- | --- |
346
+ | Color or share the scored structure | `input.protcross.pdb` or `.cif` | Input coordinates with residue scores in B-factor fields |
347
+ | Rank and filter every scored residue | `input.protcross.scores.tsv` | Identifiers, scores, calls, coordinates, cluster IDs, and ranks |
348
+ | Use predicted sites in a script | `input.protcross.pockets.json` | Thresholded residue clusters, members, centroids, and spatial statistics |
349
+ | Audit or reproduce a run | `input.protcross.summary.json` | Parameters, asset and input hashes, runtime, warnings, and top results |
337
350
 
338
351
  PDB annotation preserves record order and updates B-factor columns on
339
352
  `ATOM`/`HETATM` records. mmCIF annotation retains coordinate categories and
@@ -354,13 +367,15 @@ coordinate models retain their original values.
354
367
  | Cluster order | Descending count/mean/maximum, then ascending canonical index |
355
368
  | Cluster center | Score-weighted Cα centroid |
356
369
 
357
- `model_score` is the canonical TSV field. `probability` remains available as a
358
- schema compatibility alias. Higher values indicate stronger support from the
359
- model's binding-site class; residue ranks preserve the continuous ordering.
360
- The threshold creates binary calls and cluster membership. Empty selections
361
- produce zero clusters and null aggregate/top-cluster entries.
362
- Selected chains share one geometry graph, so connected components can span a
363
- chain interface.
370
+ Start analysis with `model_score` and `rank` in the TSV. Higher scores indicate
371
+ stronger model support for the binding-site class. Scores are continuous model
372
+ outputs and are not independently calibrated probabilities. `probability` is a
373
+ schema compatibility alias for `model_score`.
374
+
375
+ The threshold controls binary calls and cluster membership; it does not change
376
+ the underlying scores. Empty selections produce zero clusters and null
377
+ aggregate/top-cluster entries. Selected chains share one geometry graph, so a
378
+ cluster can span a chain interface.
364
379
 
365
380
  ### Schemas
366
381
 
@@ -372,89 +387,48 @@ The JSON package records the application version, scoring procedure, selected
372
387
  asset bundle, asset hashes, input SHA256, threshold, clustering parameters,
373
388
  device, precision, and effective microbatch size.
374
389
 
375
- ## Model and inference pipeline
376
-
377
- ```mermaid
378
- flowchart LR
379
- accTitle: ProtCross inference pipeline
380
- accDescr: Coordinate files are parsed into per-chain sequences and a shared C-alpha graph, embedded with ESM-C and PCA, scored by PointNet++, and serialized as annotated coordinates, scores TSV, pockets JSON, and summary JSON.
381
-
382
- coordinates["PDB or mmCIF"] --> parser["Structure parser"]
383
- parser --> sequence["Per-chain sequence"]
384
- parser --> geometry["Centered Cα graph"]
385
- sequence --> esmc["ESM-C 600M"]
386
- esmc --> pca["PCA 128"]
387
- pca --> pointnet["PointNet++"]
388
- geometry --> pointnet
389
- pointnet --> scores["Residue scores"]
390
- scores --> clusters["Threshold and cluster"]
391
- clusters --> outputs["Four-file result package"]
392
- ```
393
-
394
- ### Components
395
-
396
- | Component | Configuration |
397
- | --- | --- |
398
- | ESM-C | 600M, hidden size 1,152, 36 layers, 18 attention heads |
399
- | PCA | Paired reducer, 128 output dimensions |
400
- | Set abstraction 1 | Sampling ratio `0.5`, radius `10 Å`, 64 neighbors |
401
- | Set abstraction 2 | Sampling ratio `0.25`, radius `20 Å`, 64 neighbors |
402
- | Set abstraction 3 | Sampling ratio `0.1`, radius `40 Å`, 64 neighbors |
403
- | Feature propagation | Three `k=3` interpolation stages |
404
- | Segmentation head | `128 -> 64 -> 32 -> 2`, dropout `0.5` |
405
-
406
- The inference parser creates centered Cα geometry and per-chain sequence
407
- chunks. ESM-C embeddings are reduced with the PCA asset paired to the selected
408
- checkpoint. PointNet++ processes every input structure as an independent graph
409
- and returns one two-class logit vector per residue.
410
-
411
- The geometry backend uses pure-PyTorch farthest-point sampling, radius search,
412
- and stable KNN interpolation. Radius neighborhoods retain the first 64 source
413
- neighbors in canonical input order. Inference runs in FP32 and records the
414
- execution mode in `summary.json`. Canonical ordering and neighbor selection are
415
- deterministic; floating-point reductions remain device- and kernel-dependent.
416
-
417
- ### Training architecture
418
-
419
- ProtCross uses a source-domain residue segmentation objective and adversarial
420
- domain adaptation between PDB and matched AF2 structures. The target-domain
421
- adversarial term supports pLDDT weighting. The maintained model configuration
422
- uses `feature_dim=128`, `use_esm=true`, `use_da=true`, and `da_weight=0.2`.
423
-
424
- Training labels are generated from standard-residue Cα atoms within `6 Å` of
425
- eligible hetero-residue atoms. The parser applies a versioned residue-name
426
- filter for waters, common crystallization additives, salts, ions, and terminal
427
- caps.
428
-
429
390
  ## Batch inference
430
391
 
431
- ProtCross 0.2.2 batches ESM-C feature extraction and PointNet++ graph inference
432
- and preserves input order and per-structure outputs.
392
+ Use one `ProtCrossPredictor` to process a directory of structures. ProtCross
393
+ 0.2.3 preserves input order, keeps each structure in its own result directory,
394
+ and bounds ESM-C and PointNet++ microbatches by count and residue cost.
433
395
 
434
396
  ```python
435
397
  from pathlib import Path
436
398
 
437
399
  from protcross.inference import ProtCrossPredictor
438
400
 
439
- inputs = sorted(Path("structures").glob("*.pdb"))
401
+ structure_dir = Path("structures")
402
+ inputs = sorted(
403
+ path
404
+ for path in structure_dir.iterdir()
405
+ if path.is_file() and path.suffix.lower() in {".pdb", ".cif", ".mmcif"}
406
+ )
407
+ if not inputs:
408
+ raise FileNotFoundError(f"No PDB/mmCIF structures found in {structure_dir}")
409
+
440
410
  output_dir = Path("batch-results")
441
411
  output_dir.mkdir(parents=True, exist_ok=True)
442
412
 
443
413
  predictor = ProtCrossPredictor.from_default_assets(
444
- device="cuda",
414
+ device="auto",
445
415
  embedding_cache_dir=".protcross-feature-cache",
446
416
  accept_esm_license=True,
447
417
  )
448
418
 
449
- output_paths = [
450
- {
451
- "output_pdb": output_dir / f"{path.stem}.protcross{path.suffix}",
452
- "scores_tsv": output_dir / f"{path.stem}.protcross.scores.tsv",
453
- "pocket_json": output_dir / f"{path.stem}.protcross.pockets.json",
454
- "summary_json": output_dir / f"{path.stem}.protcross.summary.json",
455
- }
456
- for path in inputs
457
- ]
419
+ output_paths = []
420
+ for index, path in enumerate(inputs, start=1):
421
+ result_dir = output_dir / f"{index:04d}-{path.stem}"
422
+ result_dir.mkdir(parents=True, exist_ok=True)
423
+ structure_suffix = ".cif" if path.suffix.lower() in {".cif", ".mmcif"} else ".pdb"
424
+ output_paths.append(
425
+ {
426
+ "output_pdb": result_dir / f"{path.stem}.protcross{structure_suffix}",
427
+ "scores_tsv": result_dir / f"{path.stem}.protcross.scores.tsv",
428
+ "pocket_json": result_dir / f"{path.stem}.protcross.pockets.json",
429
+ "summary_json": result_dir / f"{path.stem}.protcross.summary.json",
430
+ }
431
+ )
458
432
 
459
433
  results = predictor.predict_many(
460
434
  inputs,
@@ -462,9 +436,22 @@ results = predictor.predict_many(
462
436
  batch_size=4,
463
437
  max_batch_residues=4096,
464
438
  max_batch_quadratic_cost=4 * 1022**2,
439
+ return_exceptions=True,
465
440
  )
441
+
442
+ for path, result in zip(inputs, results):
443
+ if isinstance(result, Exception):
444
+ print(f"FAILED {path}: {result}")
445
+ else:
446
+ print(f"DONE {path}: {result.output_files['summary_json']}")
466
447
  ```
467
448
 
449
+ The per-input result directories keep files distinct when structures share a
450
+ stem or use different coordinate formats. `return_exceptions=True` lets the
451
+ remaining inputs finish and keeps each exception in its original list position.
452
+ Pass `chain_ids=[None, "A", ...]` to choose a chain independently for each
453
+ input; `None` selects all scorable chains. The list must follow `inputs` order.
454
+
468
455
  ### Scheduler controls
469
456
 
470
457
  | Parameter | Default | Budget |
@@ -476,40 +463,29 @@ results = predictor.predict_many(
476
463
  | `max_feature_padded_tokens` | `2048` | Maximum padded ESM-C token matrix |
477
464
  | `return_exceptions` | `False` | Per-item exception collection |
478
465
 
479
- The scheduler keeps each structure in one PointNet++ graph and each chain in one
480
- ESM-C context. Identical chain sequences share one feature extraction
481
- within a microbatch. The feature cache validates tensor shape, PCA dimension,
482
- finite values, cache schema, and ESM/PCA asset identity.
483
-
484
- The residue and quadratic-cost limits are hard bounds for both a microbatch and
485
- each individual structure. An input graph that exceeds either limit fails before
486
- feature extraction; select a chain or raise the corresponding explicit limit.
487
-
488
- Accelerator memory errors trigger recursive microbatch splitting. With
489
- `return_exceptions=True`, failed items occupy their original list positions as
490
- exception objects. Completed items retain `PredictionResult` values.
491
- Desktop batch jobs use the same predictor API with a microbatch size of four.
492
- When a single-structure microbatch exhausts device memory, the item returns or
493
- raises that exception according to `return_exceptions`; chain selection and a
494
- smaller structure scope reduce its graph size.
495
-
496
- Cache keys include the chain sequence, cache schema, PCA dimension, maximum
497
- context, asset version, and ESM/PCA asset identity. Cache writes use atomic
498
- temporary-file replacement. Remove the cache directory to reclaim space or
499
- force feature regeneration. Predictors constructed with an injected ESM
500
- extractor or PCA reducer require a non-empty `feature_pipeline_fingerprint`
501
- before persistent feature caching can be enabled.
502
-
503
- The batch return type is `list[PredictionResult]` with the default exception
504
- mode and `list[PredictionResult | Exception]` with `return_exceptions=True`.
505
- Each `output_paths` mapping accepts `output_pdb`, `scores_tsv`, `pocket_json`,
506
- and `summary_json`; the result-schema aliases `structure` and `pockets_json`
507
- are accepted as well.
466
+ Each structure remains one PointNet++ graph and each chain remains one ESM-C
467
+ context. Identical chain sequences share feature extraction within a
468
+ microbatch. Persistent cache entries include the sequence, PCA dimension,
469
+ context limit, cache schema, and ESM/PCA asset identity.
470
+
471
+ The residue and quadratic-cost settings bound both a microbatch and each
472
+ individual structure. Select one chain or raise an explicit limit when a graph
473
+ exceeds that budget. Accelerator memory errors trigger recursive microbatch
474
+ splitting; a single item that still exhausts memory is reported through the
475
+ selected exception mode. Desktop batch jobs use this API with groups of four.
476
+
477
+ With the default exception mode, the return type is
478
+ `list[PredictionResult]`. With `return_exceptions=True`, it is
479
+ `list[PredictionResult | Exception]`. Each `output_paths` entry accepts the four
480
+ writer names shown in the example.
508
481
 
509
482
  ## Python API
510
483
 
511
484
  ### Single-structure helper
512
485
 
486
+ Use `predict_pdb` for a script that scores one structure and writes a complete
487
+ result package:
488
+
513
489
  ```python
514
490
  from pathlib import Path
515
491
 
@@ -531,9 +507,12 @@ result = predict_pdb(
531
507
  print(result.format_summary())
532
508
  ```
533
509
 
510
+ `predict_pdb` resolves and downloads missing managed assets by default. Set
511
+ `offline=True` for a local-cache-only run.
512
+
534
513
  ### Reusable predictor
535
514
 
536
- Load `ProtCrossPredictor` once for repeated calls:
515
+ Load `ProtCrossPredictor` once when a process will score several structures:
537
516
 
538
517
  ```python
539
518
  from protcross.inference import ProtCrossPredictor
@@ -550,9 +529,9 @@ summary = result.to_summary_dict()
550
529
  ```
551
530
 
552
531
  Use one predictor per device worker and serialize calls that share an instance.
553
- Independent processes load independent model instances. CLI output protection
554
- uses `--overwrite`; Python writers publish each file through atomic replacement
555
- and require the annotated structure extension to match the input format.
532
+ Independent processes load independent model instances. Python writers publish
533
+ each output file through atomic replacement and keep the annotated structure in
534
+ the input coordinate format.
556
535
 
557
536
  Structure inspection is also available from Python:
558
537
 
@@ -581,14 +560,14 @@ inspection = inspect_structure("examples/6fhu.pdb")
581
560
 
582
561
  ### Managed assets
583
562
 
584
- Prediction requires a checkpoint, its paired PCA reducer, and ESM-C 600M
585
- weights. Install the default bundle with:
563
+ Prediction uses three matched assets: a ProtCross checkpoint, its PCA reducer,
564
+ and ESM-C 600M weights. For most users, install the managed bundle once:
586
565
 
587
566
  ```bash
588
567
  protcross setup-assets --accept-esm-license
589
568
  ```
590
569
 
591
- The command installs these files under
570
+ The command verifies and installs these files under
592
571
  `~/.cache/protcross/assets/v0.1.2`:
593
572
 
594
573
  ```text
@@ -598,9 +577,10 @@ esmc_600m_2024_12_v0.pth
598
577
  protcross-assets.json
599
578
  ```
600
579
 
601
- The ESM-C download is approximately 2.14 GiB. Asset setup supports partial
602
- download resumption, file locking, SHA256 verification, and atomic publication.
603
- Review the ESM-C model terms before recording acceptance.[^1]
580
+ The ESM-C download is approximately 2.14 GiB. Interrupted transfers resume from
581
+ retained partial data. Setup verifies SHA256 hashes and publishes completed
582
+ files atomically. Later predictions reuse the manifest verification while file
583
+ size and modification time remain unchanged.
604
584
 
605
585
  | Bundle | Checkpoint and PCA |
606
586
  | --- | --- |
@@ -611,13 +591,13 @@ Release compatibility:
611
591
 
612
592
  | Interface | Version |
613
593
  | --- | --- |
614
- | Application and Desktop | `0.2.2` |
594
+ | Application and Desktop | `0.2.3` |
615
595
  | Default checkpoint/PCA bundle | `0.1.2` |
616
596
  | Paper reproduction bundle | `0.1.1-paper` |
617
597
  | Pocket and summary schemas | `protcross-pocket-v2`, `protcross-summary-v2` |
618
598
 
619
599
  `default` and `latest` resolve to the bundle pinned by the installed package.
620
- The checkpoint and PCA reducer come from the same bundle.
600
+ Keep the checkpoint and PCA reducer from the same bundle.
621
601
 
622
602
  Configure another managed directory with either interface:
623
603
 
@@ -630,8 +610,8 @@ protcross setup-assets \
630
610
  --accept-esm-license
631
611
  ```
632
612
 
633
- Use `--refresh-assets` to rebuild and verify the managed cache. Use `--offline`
634
- or `--no-auto-assets` for local-only asset resolution.
613
+ Use `--refresh-assets` for a fresh download and verification. Use `--offline`
614
+ or `--no-auto-assets` to limit prediction to local files.
635
615
 
636
616
  ### Existing or custom assets
637
617
 
@@ -659,7 +639,7 @@ protcross predict input.pdb \
659
639
  ```
660
640
 
661
641
  Checkpoint, PCA, and PyTorch weight files can contain executable serialized
662
- objects. Use assets from controlled storage. ESM-C weights are distributed
642
+ objects. Load them from controlled storage. ESM-C weights are distributed
663
643
  through the EvolutionaryScale model repository.[^2]
664
644
 
665
645
  CLI and Desktop assets use separate storage roots. Desktop records its selected
@@ -674,11 +654,11 @@ uses a per-session token for local API requests.
674
654
  ### Install
675
655
 
676
656
  Download the matching release artifact and `SHA256SUMS.txt` from
677
- [the v0.2.2 release](https://github.com/GeraltZeroZhong/ProtCross/releases/tag/v0.2.2):
657
+ [the v0.2.3 release](https://github.com/GeraltZeroZhong/ProtCross/releases/tag/v0.2.3):
678
658
 
679
659
  ```text
680
- ProtCross_Desktop_0.2.2_x64-setup.exe
681
- ProtCross_Desktop_0.2.2_macos-aarch64.dmg
660
+ ProtCross_Desktop_0.2.3_x64-setup.exe
661
+ ProtCross_Desktop_0.2.3_macos-aarch64.dmg
682
662
  ```
683
663
 
684
664
  The guided first-launch workflow installs a CPU runtime, records ESM-C term
@@ -710,27 +690,88 @@ Tauri 2 shell
710
690
  -> ProtCross predictor and batch scheduler
711
691
  ```
712
692
 
713
- Desktop batch jobs reuse one predictor, reuse input inspection reports, expose
714
- per-item status, and support cancellation between microbatches.
693
+ Desktop batch jobs reuse one predictor and the input inspection reports. Each
694
+ staged file has its own chain selector, including all scorable chains and blank
695
+ chain IDs. The monitor exposes per-item status and full errors, supports
696
+ cancellation between microbatches, and can start a new queue containing only
697
+ failed or interrupted items.
715
698
 
716
- For a batch run, open **Batch**, add and review the deduplicated structure list,
717
- select one output root, and start the queue. The staging list supports repeated
718
- file selection, per-file removal, and clearing. Runtime progress, completed,
719
- failed, and remaining counts stay visible while the queue runs. Each input
720
- receives a unique subdirectory containing the four-file output package. Select a
721
- completed row to open it in **Results**; completed items remain available when
722
- the queue is cancelled.
699
+ For a batch run, open **Batch**, add the structures, review the deduplicated
700
+ list, choose one output root, and start the queue. Progress and per-item status
701
+ remain visible while the queue runs. Each input receives a unique subdirectory
702
+ with the four-file result package. Select any completed row to inspect it in
703
+ **Results**. Recent batch history is stored in the Desktop application-data
704
+ directory and restored on the next launch. Work that was active during a
705
+ restart appears as interrupted and can be retried; completed items and their
706
+ files remain available.
723
707
 
724
708
  When the output field is empty, Desktop writes single predictions under its
725
709
  application-data `outputs/<structure>/` directory and batch predictions under
726
710
  `outputs/batch/<job-id>/`. The active platform path is displayed below the
727
711
  output field.
728
712
 
729
- The Results workspace maps annotated B-factor values to a continuous ProtCross
730
- model-score color theme and overlays the selected cluster in ball-and-stick
731
- representation. A previous result can be reopened by selecting its
732
- `*.protcross.summary.json` file. Diagnostics presents backend and asset health
733
- before the expandable technical report.
713
+ The **Results** workspace colors scored residues by model score and gives
714
+ unscored residues a neutral gray color, including residues outside a selected
715
+ chain or a truncated sequence context. Adjust the displayed score cutoff and
716
+ Cα clustering distance to regroup the complete residue table immediately;
717
+ this updates the viewer and cluster inspector without running the model or
718
+ changing output files. Reopen a previous package by selecting its
719
+ `*.protcross.summary.json` file. Use **Diagnostics** to test the runtime, review
720
+ asset health, and export a sanitized support ZIP with bounded log excerpts.
721
+
722
+ ## Model and inference pipeline
723
+
724
+ ```mermaid
725
+ flowchart LR
726
+ accTitle: ProtCross inference pipeline
727
+ accDescr: Coordinate files are parsed into per-chain sequences and a shared C-alpha graph, embedded with ESM-C and PCA, scored by PointNet++, and serialized as annotated coordinates, scores TSV, pockets JSON, and summary JSON.
728
+
729
+ coordinates["PDB or mmCIF"] --> parser["Structure parser"]
730
+ parser --> sequence["Per-chain sequence"]
731
+ parser --> geometry["Centered Cα graph"]
732
+ sequence --> esmc["ESM-C 600M"]
733
+ esmc --> pca["PCA 128"]
734
+ pca --> pointnet["PointNet++"]
735
+ geometry --> pointnet
736
+ pointnet --> scores["Residue scores"]
737
+ scores --> clusters["Threshold and cluster"]
738
+ clusters --> outputs["Four-file result package"]
739
+ ```
740
+
741
+ ### Components
742
+
743
+ | Component | Configuration |
744
+ | --- | --- |
745
+ | ESM-C | 600M, hidden size 1,152, 36 layers, 18 attention heads |
746
+ | PCA | Paired reducer, 128 output dimensions |
747
+ | Set abstraction 1 | Sampling ratio `0.5`, radius `10 Å`, 64 neighbors |
748
+ | Set abstraction 2 | Sampling ratio `0.25`, radius `20 Å`, 64 neighbors |
749
+ | Set abstraction 3 | Sampling ratio `0.1`, radius `40 Å`, 64 neighbors |
750
+ | Feature propagation | Three `k=3` interpolation stages |
751
+ | Segmentation head | `128 -> 64 -> 32 -> 2`, dropout `0.5` |
752
+
753
+ The inference parser creates centered Cα geometry and per-chain sequence
754
+ chunks. ESM-C embeddings are reduced with the PCA asset paired to the selected
755
+ checkpoint. PointNet++ processes every input structure as an independent graph
756
+ and returns one two-class logit vector per residue.
757
+
758
+ The geometry backend uses pure-PyTorch farthest-point sampling, radius search,
759
+ and stable KNN interpolation. Radius neighborhoods retain the first 64 source
760
+ neighbors in canonical input order. Inference runs in FP32 and records the
761
+ execution mode in `summary.json`. Canonical ordering and neighbor selection are
762
+ deterministic; floating-point reductions remain device- and kernel-dependent.
763
+
764
+ ### Training architecture
765
+
766
+ ProtCross uses a source-domain residue segmentation objective and adversarial
767
+ domain adaptation between PDB and matched AF2 structures. The target-domain
768
+ adversarial term supports pLDDT weighting. The maintained model configuration
769
+ uses `feature_dim=128`, `use_esm=true`, `use_da=true`, and `da_weight=0.2`.
770
+
771
+ Training labels are generated from standard-residue Cα atoms within `6 Å` of
772
+ eligible hetero-residue atoms. The parser applies a versioned residue-name
773
+ filter for waters, common crystallization additives, salts, ions, and terminal
774
+ caps.
734
775
 
735
776
  ## Training and development
736
777
 
@@ -752,6 +793,14 @@ reproduction/ archived paper-era workflows
752
793
  examples/ example coordinate files
753
794
  ```
754
795
 
796
+ ### Development environment
797
+
798
+ ```bash
799
+ conda env create -f environment.yml
800
+ conda activate protcross
801
+ python -m pip install -e ".[dev,esm]"
802
+ ```
803
+
755
804
  ### Maintained training workflow
756
805
 
757
806
  Place source coordinate files in `data/raw_pdb`. `protcross download-af2`
@@ -794,7 +843,9 @@ protcross train
794
843
  Preprocessing writes one `.pt` tensor package per structure and an atomic
795
844
  `protcross-preprocess-manifest.json`. The manifest records completion state,
796
845
  input hashes, generated outputs, failures, and skipped files. PCA fitting uses
797
- the configured preprocessing seed.
846
+ the configured preprocessing seed. Training cache freshness uses `.pt` file
847
+ names, sizes, and modification times, so dataset startup does not reread every
848
+ tensor package solely to hash it.
798
849
 
799
850
  Hydra configuration entry points:
800
851
 
@@ -883,6 +934,18 @@ diagnostics.
883
934
 
884
935
  ## Version history
885
936
 
937
+ ### 0.2.3
938
+
939
+ - Reused verified asset manifests and metadata-based dataset signatures to cut
940
+ repeated hashing and startup work; accelerated AF2 indexing, preprocessing,
941
+ strategy search, and long-log assembly.
942
+ - Removed redundant prediction, CLI, data-loading, and output-rollback layers;
943
+ retained bounded scheduling, input contracts, and atomic per-file outputs.
944
+ - Improved machine-readable inspection errors, chain guidance, Desktop error
945
+ display and diagnostic exports, per-file batch chain selection, failed-item
946
+ retry, restart-safe batch history, interactive result regrouping, and neutral
947
+ rendering for unscored residues.
948
+
886
949
  ### 0.2.2
887
950
 
888
951
  - Added bounded ESM-C and PointNet++ microbatching with graph, residue,
@@ -892,7 +955,7 @@ diagnostics.
892
955
  - Accelerated deterministic geometry and inference parsing; corrected small-set
893
956
  split leakage, label-alignment statistics, AF2 mapping, and mmCIF residue
894
957
  identity; added finite-value gates, isolated feature-cache namespaces, and
895
- transactional dataset and result publication.
958
+ transactional dataset publication and atomic per-file result writes.
896
959
  - Rebuilt ProtCross Desktop around responsive task workspaces, semantic OKLCH
897
960
  themes, accessible interaction states, score-aware Mol* rendering, structured
898
961
  diagnostics, persistent batch feedback, and result-package reopening.