countmut 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. {countmut-0.2.0/countmut.egg-info → countmut-0.2.2}/PKG-INFO +24 -7
  2. {countmut-0.2.0 → countmut-0.2.2}/README.md +23 -6
  3. {countmut-0.2.0 → countmut-0.2.2}/backend/countmut_core.c +126 -67
  4. {countmut-0.2.0 → countmut-0.2.2}/backend/countmut_core.h +18 -1
  5. {countmut-0.2.0 → countmut-0.2.2}/backend/countmut_core_main.c +4 -0
  6. {countmut-0.2.0 → countmut-0.2.2}/backend/countmut_expr.c +126 -35
  7. {countmut-0.2.0 → countmut-0.2.2}/backend/countmut_expr.h +26 -11
  8. countmut-0.2.2/countmut/_core/countmut_core +0 -0
  9. {countmut-0.2.0 → countmut-0.2.2}/countmut/backend.py +2 -0
  10. {countmut-0.2.0 → countmut-0.2.2}/countmut/cli.py +16 -1
  11. {countmut-0.2.0 → countmut-0.2.2}/countmut/model.py +1 -0
  12. {countmut-0.2.0 → countmut-0.2.2/countmut.egg-info}/PKG-INFO +24 -7
  13. {countmut-0.2.0 → countmut-0.2.2}/countmut.egg-info/SOURCES.txt +1 -0
  14. {countmut-0.2.0 → countmut-0.2.2}/pyproject.toml +1 -1
  15. countmut-0.2.2/tests/test_router.py +229 -0
  16. countmut-0.2.0/countmut/_core/countmut_core +0 -0
  17. {countmut-0.2.0 → countmut-0.2.2}/LICENSE +0 -0
  18. {countmut-0.2.0 → countmut-0.2.2}/MANIFEST.in +0 -0
  19. {countmut-0.2.0 → countmut-0.2.2}/backend/Makefile +0 -0
  20. {countmut-0.2.0 → countmut-0.2.2}/backend/bedidx.c +0 -0
  21. {countmut-0.2.0 → countmut-0.2.2}/backend/bgzf.c +0 -0
  22. {countmut-0.2.0 → countmut-0.2.2}/backend/bgzf.h +0 -0
  23. {countmut-0.2.0 → countmut-0.2.2}/backend/faidx.c +0 -0
  24. {countmut-0.2.0 → countmut-0.2.2}/backend/faidx.h +0 -0
  25. {countmut-0.2.0 → countmut-0.2.2}/backend/hts.c +0 -0
  26. {countmut-0.2.0 → countmut-0.2.2}/backend/hts.h +0 -0
  27. {countmut-0.2.0 → countmut-0.2.2}/backend/ketopt.h +0 -0
  28. {countmut-0.2.0 → countmut-0.2.2}/backend/khash.h +0 -0
  29. {countmut-0.2.0 → countmut-0.2.2}/backend/kseq.h +0 -0
  30. {countmut-0.2.0 → countmut-0.2.2}/backend/ksort.h +0 -0
  31. {countmut-0.2.0 → countmut-0.2.2}/backend/kstdint.h +0 -0
  32. {countmut-0.2.0 → countmut-0.2.2}/backend/kstring.h +0 -0
  33. {countmut-0.2.0 → countmut-0.2.2}/backend/razf.c +0 -0
  34. {countmut-0.2.0 → countmut-0.2.2}/backend/razf.h +0 -0
  35. {countmut-0.2.0 → countmut-0.2.2}/backend/sam.c +0 -0
  36. {countmut-0.2.0 → countmut-0.2.2}/backend/sam.h +0 -0
  37. {countmut-0.2.0 → countmut-0.2.2}/countmut/__init__.py +0 -0
  38. {countmut-0.2.0 → countmut-0.2.2}/countmut/bam_tags.py +0 -0
  39. {countmut-0.2.0 → countmut-0.2.2}/countmut/core.py +0 -0
  40. {countmut-0.2.0 → countmut-0.2.2}/countmut/utils.py +0 -0
  41. {countmut-0.2.0 → countmut-0.2.2}/countmut.egg-info/dependency_links.txt +0 -0
  42. {countmut-0.2.0 → countmut-0.2.2}/countmut.egg-info/entry_points.txt +0 -0
  43. {countmut-0.2.0 → countmut-0.2.2}/countmut.egg-info/requires.txt +0 -0
  44. {countmut-0.2.0 → countmut-0.2.2}/countmut.egg-info/top_level.txt +0 -0
  45. {countmut-0.2.0 → countmut-0.2.2}/setup.cfg +0 -0
  46. {countmut-0.2.0 → countmut-0.2.2}/tests/test_cli.py +0 -0
  47. {countmut-0.2.0 → countmut-0.2.2}/tests/test_core.py +0 -0
  48. {countmut-0.2.0 → countmut-0.2.2}/tests/test_correctness.py +0 -0
  49. {countmut-0.2.0 → countmut-0.2.2}/tests/test_unified.py +0 -0
  50. {countmut-0.2.0 → countmut-0.2.2}/tests/test_utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: countmut
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: Unified ultra-fast strand-aware mutation counter (C backend + samtools-style -e/-p filters)
5
5
  Author-email: Ye Chang <yech1990@gmail.com>
6
6
  License-Expression: MIT
@@ -99,6 +99,22 @@ countmut -i x -r ref -o out -e "[NM] <= 3 and not (flag.dup ~= 0) and flag.read1
99
99
  countmut -i x -r ref -o out -p "depth >= 5 and g >= 2"
100
100
  ```
101
101
 
102
+ **`-e` is also a group router.** A bare boolean expression is a filter
103
+ (`true` → count, `nil`/`false` → drop), but an expression that returns an
104
+ integer `0..3` routes each kept base into that **group**; `true` routes to
105
+ group 0. Anything else drops the base (with a stderr warning). The split
106
+ shows up in `--output-format` templates as per-group cells `{a.0}` … `{n.3}`
107
+ (plain `{a}` stays the total over all groups):
108
+
109
+ ```bash
110
+ # bisulfite A->G, 2-group router: group 1 = high-conversion bases, group 0 =
111
+ # everything else that passes the hard NS gate (low quality / read-end trim)
112
+ countmut -i x -r ref -o out \
113
+ -e "([NS] <= 1) and (([Yf] >= 1 and [Zf] <= 3 and bq >= 20 and qpos >= 2 and qlen - qpos > 2) and 1 or 0)" \
114
+ --output-format "{chrom}\t{pos+1}\t{strand}\t{motif}\t{a.0}\t{a.1}\t{g.0}\t{g.1}" \
115
+ --motif-pad 15 --fmt-header "chrom\tpos\tstrand\tmotif\tu0\tu1\tm0\tm1"
116
+ ```
117
+
102
118
  Most filters use roughly ten variables — `mapq`, `bq` (base quality), `flags`,
103
119
  `qpos` (position in the read), `dist5`/`dist3` (distance to the read ends),
104
120
  `base`/`ref`, `tag('XX')`, and `rname`. A couple of things are worth knowing.
@@ -116,10 +132,11 @@ reference in [`docs/expression_reference.md`](docs/expression_reference.md).
116
132
  ## Output format
117
133
 
118
134
  `--output-format` takes a **row template**: literal text plus `{expr}`
119
- placeholders evaluated per site over the site values (`pos`, `ref`, `depth`,
120
- `a c g t n`, `ins del ref_skip fail`). Placeholders run real Lua, so you can
121
- compute cells — a conversion ratio is just `{t}/({c}+{t})` — and `round(x, n)`
122
- and `int(x)` are helpers for formatting:
135
+ placeholders evaluated per site over the site values (`chrom`, `pos`,
136
+ `strand`, `motif`, `ref`, `depth`, `a c g t n`, `ins del ref_skip fail`, and
137
+ per-group counts `a.0` … `n.3` whenever `-e` routes into groups). Placeholders
138
+ run real Lua, so you can compute cells — a conversion ratio is just
139
+ `{t}/({c}+{t})` — and `round(x, n)` and `int(x)` are helpers for formatting:
123
140
 
124
141
  ```bash
125
142
  countmut -i x -r ref -o out \
@@ -138,8 +155,8 @@ Two BAM-walking strategies live in the C core and emit identical output, so
138
155
  the engine choice only affects speed (`--engine auto` uses the pileup walk for
139
156
  the per-position counting). The options are few: input/reference/output,
140
157
  `--region`, `--threads/-t`, `--engine`, `--strandless`, `--count-indels`,
141
- `--vcf` (+ `--min-depth`/`--min-allele-support`), `-e`/`-p`, and
142
- `--output-format`/`--fmt-header`.
158
+ `--vcf` (+ `--min-depth`/`--min-allele-support`), `-e`/`-p`, `--motif-pad`,
159
+ and `--output-format`/`--fmt-header`.
143
160
 
144
161
  ## Input formats
145
162
 
@@ -62,6 +62,22 @@ countmut -i x -r ref -o out -e "[NM] <= 3 and not (flag.dup ~= 0) and flag.read1
62
62
  countmut -i x -r ref -o out -p "depth >= 5 and g >= 2"
63
63
  ```
64
64
 
65
+ **`-e` is also a group router.** A bare boolean expression is a filter
66
+ (`true` → count, `nil`/`false` → drop), but an expression that returns an
67
+ integer `0..3` routes each kept base into that **group**; `true` routes to
68
+ group 0. Anything else drops the base (with a stderr warning). The split
69
+ shows up in `--output-format` templates as per-group cells `{a.0}` … `{n.3}`
70
+ (plain `{a}` stays the total over all groups):
71
+
72
+ ```bash
73
+ # bisulfite A->G, 2-group router: group 1 = high-conversion bases, group 0 =
74
+ # everything else that passes the hard NS gate (low quality / read-end trim)
75
+ countmut -i x -r ref -o out \
76
+ -e "([NS] <= 1) and (([Yf] >= 1 and [Zf] <= 3 and bq >= 20 and qpos >= 2 and qlen - qpos > 2) and 1 or 0)" \
77
+ --output-format "{chrom}\t{pos+1}\t{strand}\t{motif}\t{a.0}\t{a.1}\t{g.0}\t{g.1}" \
78
+ --motif-pad 15 --fmt-header "chrom\tpos\tstrand\tmotif\tu0\tu1\tm0\tm1"
79
+ ```
80
+
65
81
  Most filters use roughly ten variables — `mapq`, `bq` (base quality), `flags`,
66
82
  `qpos` (position in the read), `dist5`/`dist3` (distance to the read ends),
67
83
  `base`/`ref`, `tag('XX')`, and `rname`. A couple of things are worth knowing.
@@ -79,10 +95,11 @@ reference in [`docs/expression_reference.md`](docs/expression_reference.md).
79
95
  ## Output format
80
96
 
81
97
  `--output-format` takes a **row template**: literal text plus `{expr}`
82
- placeholders evaluated per site over the site values (`pos`, `ref`, `depth`,
83
- `a c g t n`, `ins del ref_skip fail`). Placeholders run real Lua, so you can
84
- compute cells — a conversion ratio is just `{t}/({c}+{t})` — and `round(x, n)`
85
- and `int(x)` are helpers for formatting:
98
+ placeholders evaluated per site over the site values (`chrom`, `pos`,
99
+ `strand`, `motif`, `ref`, `depth`, `a c g t n`, `ins del ref_skip fail`, and
100
+ per-group counts `a.0` … `n.3` whenever `-e` routes into groups). Placeholders
101
+ run real Lua, so you can compute cells — a conversion ratio is just
102
+ `{t}/({c}+{t})` — and `round(x, n)` and `int(x)` are helpers for formatting:
86
103
 
87
104
  ```bash
88
105
  countmut -i x -r ref -o out \
@@ -101,8 +118,8 @@ Two BAM-walking strategies live in the C core and emit identical output, so
101
118
  the engine choice only affects speed (`--engine auto` uses the pileup walk for
102
119
  the per-position counting). The options are few: input/reference/output,
103
120
  `--region`, `--threads/-t`, `--engine`, `--strandless`, `--count-indels`,
104
- `--vcf` (+ `--min-depth`/`--min-allele-support`), `-e`/`-p`, and
105
- `--output-format`/`--fmt-header`.
121
+ `--vcf` (+ `--min-depth`/`--min-allele-support`), `-e`/`-p`, `--motif-pad`,
122
+ and `--output-format`/`--fmt-header`.
106
123
 
107
124
  ## Input formats
108
125
 
@@ -75,12 +75,8 @@ static int read_trim_skip(const bam1_t *b, int qpos, int r1_end, int r2_start) {
75
75
  return 0;
76
76
  }
77
77
 
78
- /* per-site accumulator */
79
- typedef struct {
80
- int cnt[2][3][5]; /* [strand][category][base] */
81
- int ins[2], del[2], refskip[2], fail[2];
82
- } site_t;
83
-
78
+ /* per-site accumulator: site_t lives in countmut_core.h (cnt is
79
+ * [strand][CM_CAT_MAX][base]) so the expression layer can index it too. */
84
80
  static void site_zero(site_t *s) { memset(s, 0, sizeof(*s)); }
85
81
 
86
82
  static int better(int mapq, int r1, int q, int omapq, int or1, int oq) {
@@ -99,6 +95,17 @@ static int base_to_index(char c) {
99
95
  }
100
96
  }
101
97
 
98
+ /* Reverse-complement a single uppercase motif base (A<->T, C<->G, else N). */
99
+ static char rc_nt(char c) {
100
+ switch (c) {
101
+ case 'A': return 'T';
102
+ case 'T': return 'A';
103
+ case 'C': return 'G';
104
+ case 'G': return 'C';
105
+ default: return 'N';
106
+ }
107
+ }
108
+
102
109
  /* read-level filters (samtools reqflags/exclflags + mapq + NS + bisulfite tags) */
103
110
  static int read_fails(const cm_config *cfg, const bam1_t *b) {
104
111
  if (cfg->req_flags && (b->core.flag & (uint32_t)cfg->req_flags) != (uint32_t)cfg->req_flags) return 1;
@@ -151,7 +158,7 @@ KHASH_INIT(rfc, rf_key, int, 1, rf_hash, rf_equal)
151
158
  * A pos/qlen/qname verify guards against a recycled buffer now holding a
152
159
  * different read. Unlike the RF_CAP-based cache this survives deep hotspots
153
160
  * (no global eviction: each mplp slot holds at most one read at a time). */
154
- typedef struct { int64_t pos; int qlen; char *qn; int pass; } expr_cc_t;
161
+ typedef struct { int64_t pos; int qlen; char *qn; int slot; } expr_cc_t;
155
162
  static inline khint_t pex_hash(uintptr_t p) {
156
163
  return (khint_t)(p >> 3) ^ ((khint_t)(p >> 13) & 0x0ff);
157
164
  }
@@ -166,8 +173,9 @@ void bed_destroy(void *_h);
166
173
  /* per-worker reusable state */
167
174
  typedef struct {
168
175
  khash_t(qn) *kh;
169
- int *sel, *mapq_a, *r1_a, *q_a, sel_cap;
170
- char *motif_buf;
176
+ int *sel, *mapq_a, *r1_a, *q_a, *g_a, sel_cap;
177
+ char *motif_buf; /* reference-forward motif window (per-site) */
178
+ char *motif_rc_buf; /* reverse-complemented copy for the minus-strand row */
171
179
  char *chr_seq; int chr_len, last_tid;
172
180
  BGZF *fp; hts_idx_t *idx; faidx_t *fai;
173
181
  void *inc_bed, *exc_bed;
@@ -181,9 +189,10 @@ static void worker_init(worker_t *w, const char *bam, const char *fa, int pad,
181
189
  const char *read_expr, const char *pile_expr,
182
190
  const char *output_expr) {
183
191
  w->kh = kh_init(qn);
184
- w->sel = w->mapq_a = w->r1_a = w->q_a = NULL;
192
+ w->sel = w->mapq_a = w->r1_a = w->q_a = w->g_a = NULL;
185
193
  w->sel_cap = 0;
186
194
  w->motif_buf = (char *)malloc(2 * pad + 2);
195
+ w->motif_rc_buf = (char *)malloc(2 * pad + 2);
187
196
  w->chr_seq = NULL; w->chr_len = 0; w->last_tid = -1;
188
197
  w->fp = bgzf_open(bam, "r");
189
198
  w->idx = bam_index_load(bam);
@@ -200,7 +209,9 @@ static void worker_free(worker_t *w) {
200
209
  if (w->mapq_a) free(w->mapq_a);
201
210
  if (w->r1_a) free(w->r1_a);
202
211
  if (w->q_a) free(w->q_a);
212
+ if (w->g_a) free(w->g_a);
203
213
  if (w->motif_buf) free(w->motif_buf);
214
+ if (w->motif_rc_buf) free(w->motif_rc_buf);
204
215
  if (w->chr_seq) free(w->chr_seq);
205
216
  kh_destroy(qn, w->kh);
206
217
  if (w->exc_bed) bed_destroy(w->exc_bed);
@@ -251,34 +262,38 @@ static int _read_fails_cached(worker_t *w, const cm_config *cfg, const bam1_t *b
251
262
  * the pileup engine evaluate it once per read instead of once per position
252
263
  * (the read-walk engine already does once per read). Per-base expressions are
253
264
  * evaluated at every aligned base, uncached. */
265
+ /* Route one aligned base: returns the category slot to count it into
266
+ * (-1 = drop, 0..CM_CAT_MAX-1 = slot). The router subsumes both the old -e
267
+ * keep/drop decision (nil/false -> -1) and the per-category tier assignment
268
+ * (a number -> that slot; true -> default slot 0). */
254
269
  static int expr_pass(worker_t *w, const bam1_t *b, const char *rname,
255
270
  const char *mrname, int s, int qpos, char ref_ch) {
256
271
  cm_expr *x = w->expr;
257
- if (x == NULL || !cm_expr_has_read(x)) return 1;
272
+ if (x == NULL || !cm_expr_has_read(x)) return 0; /* no -e -> default slot 0 */
258
273
  if (!cm_expr_read_constant(x))
259
- return cm_expr_read(x, b, rname, mrname, qpos, s ? -1 : 1, ref_ch);
274
+ return cm_expr_route(x, b, rname, mrname, qpos, s ? -1 : 1, ref_ch);
260
275
  /* read-constant: memoize by the pileup slot pointer (stable per read
261
276
  * across its span) with a pos/qlen/qname verify against recycling. */
262
- uintptr_t slot = (uintptr_t)(const void *)b;
263
- khint_t k = kh_get(pex, w->pexc, slot);
277
+ uintptr_t key = (uintptr_t)(const void *)b;
278
+ khint_t k = kh_get(pex, w->pexc, key);
264
279
  if (k != kh_end(w->pexc)) {
265
280
  expr_cc_t *cc = &kh_val(w->pexc, k);
266
281
  if (cc->pos == b->core.pos && cc->qlen == (int)b->core.l_qseq
267
282
  && cc->qn && strcmp(cc->qn, bam_get_qname(b)) == 0)
268
- return cc->pass;
283
+ return cc->slot;
269
284
  free(cc->qn); cc->qn = NULL;
270
285
  }
271
- int pass = cm_expr_read(x, b, rname, mrname, 0, s ? -1 : 1, 'N');
286
+ int slot = cm_expr_route(x, b, rname, mrname, 0, s ? -1 : 1, 'N');
272
287
  if (k == kh_end(w->pexc)) {
273
- int ret; k = kh_put(pex, w->pexc, slot, &ret);
288
+ int ret; k = kh_put(pex, w->pexc, key, &ret);
274
289
  memset(&kh_val(w->pexc, k), 0, sizeof(expr_cc_t)); /* fresh slots are uninitialized */
275
290
  }
276
291
  expr_cc_t *cc = &kh_val(w->pexc, k);
277
292
  if (cc->qn == NULL) cc->qn = strdup(bam_get_qname(b));
278
293
  cc->pos = b->core.pos;
279
294
  cc->qlen = (int)b->core.l_qseq;
280
- cc->pass = pass;
281
- return pass;
295
+ cc->slot = slot;
296
+ return slot;
282
297
  }
283
298
 
284
299
  typedef struct { int tid, beg, end; } region_t;
@@ -344,15 +359,12 @@ static void emit_site(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, FILE *f
344
359
  /* custom output template (-o): evaluate it per emitted strand and write
345
360
  * exactly what it returns; bypasses the built-in formats. */
346
361
  if (w != NULL && w->expr && cm_expr_has_output(w->expr)) {
347
- int cnt[5] = {0};
348
- int ins = 0, del = 0, rs = 0, fl = 0;
349
- for (int s = 0; s < 2; ++s) {
350
- for (int c = 0; c < 3; ++c)
351
- for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[s][c][b];
352
- ins += site->ins[s]; del += site->del[s]; rs += site->refskip[s]; fl += site->fail[s];
353
- }
362
+ /* Reference-forward motif window for {motif} (built whenever the
363
+ * sequence is available, not only in the legacy conversion view);
364
+ * the minus-strand row gets its reverse complement (parity with the
365
+ * bisulfite reformat bridge). */
354
366
  const char *motif = NULL;
355
- if (cfg->out == CM_OUT_CONVERSION && w->chr_len > 0) {
367
+ if (w->chr_len > 0) {
356
368
  int mlen = cfg->pad * 2 + 1;
357
369
  for (int k2 = (int)pos - cfg->pad; k2 < (int)pos + cfg->pad + 1; ++k2)
358
370
  w->motif_buf[k2 - ((int)pos - cfg->pad)] =
@@ -360,16 +372,29 @@ static void emit_site(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, FILE *f
360
372
  w->motif_buf[mlen] = 0;
361
373
  motif = w->motif_buf;
362
374
  }
363
- int refi = base_to_index((char)cfg->ref_base), muti = base_to_index((char)cfg->mut_base);
375
+ /* ref_base/mut_base are legacy/unused (always 0) -> -1 (unset), so the
376
+ * derived u/m/o fields stay off for template rows. */
377
+ int refi = cfg->ref_base ? base_to_index((char)cfg->ref_base) : -1;
378
+ int muti = cfg->mut_base ? base_to_index((char)cfg->mut_base) : -1;
379
+ /* Per-strand rows: counts are this strand only, per category slot. */
364
380
  for (int s = 0; s < 2; ++s) {
365
381
  if (s == 0 && !emit_plus) continue;
366
382
  if (s == 1 && !emit_minus) continue;
367
383
  int sdepth = site->ins[s] + site->del[s] + site->refskip[s] + site->fail[s];
368
- for (int c = 0; c < 3; ++c)
384
+ for (int c = 0; c < CM_CAT_MAX; ++c)
369
385
  for (int b = 0; b < 5; ++b) sdepth += site->cnt[s][c][b];
370
386
  if (sdepth == 0) continue; /* match the built-in formats: skip empty strands */
371
- cm_expr_output(w->expr, pos, ref_ch, motif, cnt, ins, del, rs, fl,
372
- refi, muti, s, fp);
387
+ const char *mot_s = motif;
388
+ if (motif && s == 1) { /* minus strand: reverse complement */
389
+ int mlen = cfg->pad * 2 + 1;
390
+ for (int k2 = 0; k2 < mlen; ++k2)
391
+ w->motif_rc_buf[k2] = rc_nt(motif[mlen - 1 - k2]);
392
+ w->motif_rc_buf[mlen] = 0;
393
+ mot_s = w->motif_rc_buf;
394
+ }
395
+ cm_expr_output(w->expr, hdr->target_name[tid], pos, ref_ch, mot_s,
396
+ site->cnt[s], site->ins[s], site->del[s],
397
+ site->refskip[s], site->fail[s], refi, muti, s, fp);
373
398
  }
374
399
  return;
375
400
  }
@@ -413,25 +438,29 @@ static void emit_site(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, FILE *f
413
438
  for (int s = 0; s < 2; ++s) {
414
439
  if (s == 0 && !emit_plus) continue;
415
440
  if (s == 1 && !emit_minus) continue;
416
- int dep = site->cnt[s][0][0]+site->cnt[s][0][1]+site->cnt[s][0][2]+site->cnt[s][0][3]+site->cnt[s][0][4];
441
+ int cnt[5] = {0};
442
+ for (int c = 0; c < CM_CAT_MAX; ++c)
443
+ for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[s][c][b];
444
+ int dep = cnt[0]+cnt[1]+cnt[2]+cnt[3]+cnt[4];
417
445
  int t_ins = site->ins[s], t_del = site->del[s], t_rs = site->refskip[s], t_fl = site->fail[s];
418
446
  if (dep + t_rs + t_del + t_ins + t_fl == 0) continue;
419
447
  if (cfg->min_depth > 0 && dep < cfg->min_depth) continue;
420
448
  fprintf(fp, "%s\t%d\t%c\t%c\t%d\t%d\t%d\t%d\t%d\t%d",
421
449
  hdr->target_name[tid], (int)pos + 1, s ? '-' : '+', ref_ch, dep,
422
- site->cnt[s][0][0], site->cnt[s][0][1], site->cnt[s][0][2],
423
- site->cnt[s][0][3], site->cnt[s][0][4]);
450
+ cnt[0], cnt[1], cnt[2], cnt[3], cnt[4]);
424
451
  if (cfg->count_indels) fprintf(fp, "\t%d\t%d\t%d\t%d", t_ins, t_del, t_rs, t_fl);
425
452
  fputc('\n', fp);
426
453
  }
427
454
  } else {
428
455
  int cnt[5] = {0}; int t_ins = 0, t_del = 0, t_rs = 0, t_fl = 0;
429
456
  if (emit_plus) {
430
- for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[0][0][b];
457
+ for (int c = 0; c < CM_CAT_MAX; ++c)
458
+ for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[0][c][b];
431
459
  t_ins += site->ins[0]; t_del += site->del[0]; t_rs += site->refskip[0]; t_fl += site->fail[0];
432
460
  }
433
461
  if (emit_minus) {
434
- for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[1][0][b];
462
+ for (int c = 0; c < CM_CAT_MAX; ++c)
463
+ for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[1][c][b];
435
464
  t_ins += site->ins[1]; t_del += site->del[1]; t_rs += site->refskip[1]; t_fl += site->fail[1];
436
465
  }
437
466
  int dep = cnt[0]+cnt[1]+cnt[2]+cnt[3]+cnt[4];
@@ -444,8 +473,12 @@ static void emit_site(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, FILE *f
444
473
  }
445
474
  } else { /* allele */
446
475
  int cnt[5] = {0};
447
- if (emit_plus) for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[0][0][b];
448
- if (emit_minus) for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[1][0][b];
476
+ if (emit_plus)
477
+ for (int c = 0; c < CM_CAT_MAX; ++c)
478
+ for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[0][c][b];
479
+ if (emit_minus)
480
+ for (int c = 0; c < CM_CAT_MAX; ++c)
481
+ for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[1][c][b];
449
482
  int dep = cnt[0]+cnt[1]+cnt[2]+cnt[3]+cnt[4];
450
483
  if (dep <= 0) return;
451
484
  if (cfg->min_depth > 0 && dep < cfg->min_depth) return;
@@ -469,16 +502,17 @@ static void emit_site(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, FILE *f
469
502
  * A/C/G/T/N totals (both strands, all quality tiers) plus indels, and the
470
503
  * reference window for mutation mode. Returns 1 = keep, 0 = omit. */
471
504
  static int expr_pile_apply(cm_expr *x, const cm_config *cfg, worker_t *w,
472
- const site_t *site, int64_t pos, char ref_ch) {
505
+ const char *chrom, const site_t *site, int64_t pos,
506
+ char ref_ch) {
473
507
  int cnt[5] = {0};
474
508
  int ins = 0, del = 0, rs = 0, fl = 0;
475
509
  for (int s = 0; s < 2; ++s) {
476
- for (int c = 0; c < 3; ++c)
510
+ for (int c = 0; c < CM_CAT_MAX; ++c)
477
511
  for (int b = 0; b < 5; ++b) cnt[b] += site->cnt[s][c][b];
478
512
  ins += site->ins[s]; del += site->del[s]; rs += site->refskip[s]; fl += site->fail[s];
479
513
  }
480
514
  const char *motif = NULL;
481
- if (cfg->out == CM_OUT_CONVERSION && w->chr_len > 0) {
515
+ if (w->chr_len > 0) { /* motif window for -p/-o expressions (ref-forward) */
482
516
  int mlen = cfg->pad * 2 + 1;
483
517
  for (int k2 = (int)pos - cfg->pad; k2 < (int)pos + cfg->pad + 1; ++k2) {
484
518
  w->motif_buf[k2 - ((int)pos - cfg->pad)] =
@@ -487,8 +521,10 @@ static int expr_pile_apply(cm_expr *x, const cm_config *cfg, worker_t *w,
487
521
  w->motif_buf[mlen] = 0;
488
522
  motif = w->motif_buf;
489
523
  }
490
- int refi = base_to_index((char)cfg->ref_base), muti = base_to_index((char)cfg->mut_base);
491
- return cm_expr_pile(x, pos, ref_ch, motif, cnt, ins, del, rs, fl, refi, muti);
524
+ int refi = cfg->ref_base ? base_to_index((char)cfg->ref_base) : -1;
525
+ int muti = cfg->mut_base ? base_to_index((char)cfg->mut_base) : -1;
526
+ return cm_expr_pile(x, chrom, pos, ref_ch, motif, cnt, ins, del, rs, fl,
527
+ refi, muti);
492
528
  }
493
529
 
494
530
  /* Fetch + uppercase the chromosome sequence once per tid (instead of calling
@@ -537,6 +573,7 @@ static void count_interval(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, FI
537
573
  w->mapq_a = (int *)realloc(w->mapq_a, n * sizeof(int));
538
574
  w->r1_a = (int *)realloc(w->r1_a, n * sizeof(int));
539
575
  w->q_a = (int *)realloc(w->q_a, n * sizeof(int));
576
+ w->g_a = (int *)realloc(w->g_a, n * sizeof(int));
540
577
  w->sel_cap = n;
541
578
  }
542
579
  site_zero(&site);
@@ -556,13 +593,15 @@ static void count_interval(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, FI
556
593
  if (s == 0) { if (qpos < cfg->trim_fragment_start || qlen - qpos <= cfg->trim_fragment_end) continue; }
557
594
  else { if (qpos < cfg->trim_fragment_end || qlen - qpos <= cfg->trim_fragment_start) continue; }
558
595
  if (read_trim_skip(b, qpos, cfg->trim_r1_end, cfg->trim_r2_start)) continue;
559
- /* -e read filter (once per read when read-constant via exprc memo,
560
- * else per aligned base; same spot as the Python engine) */
561
- if (!expr_pass(w, b, hdr->target_name[tid],
562
- (b->core.mtid >= 0 && b->core.mtid < hdr->n_targets)
563
- ? hdr->target_name[b->core.mtid] : "",
564
- s, qpos, ref_ch))
565
- continue;
596
+ /* -e read router: returns the category slot (-1 = drop). Evaluated
597
+ * once per read when read-constant via the exprc memo, else per
598
+ * aligned base (the same spot as the Python engine). */
599
+ int g = expr_pass(w, b, hdr->target_name[tid],
600
+ (b->core.mtid >= 0 && b->core.mtid < hdr->n_targets)
601
+ ? hdr->target_name[b->core.mtid] : "",
602
+ s, qpos, ref_ch);
603
+ if (g < 0) continue;
604
+ w->g_a[i] = g;
566
605
  int mapq = (int)b->core.qual;
567
606
  int r1 = (b->core.flag & BAM_FREAD1) ? 1 : 0;
568
607
  int qual = (int)bam_get_qual(b)[qpos];
@@ -590,12 +629,20 @@ static void count_interval(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, FI
590
629
  const bam1_t *b = p->b;
591
630
  int s = bio_strand(b);
592
631
  uint8_t nt = bam_seqi(bam_get_seq(b), p->qpos);
593
- int base_i = nt16_index(nt); /* stored SEQ is reference-forward */
632
+ int base_i = nt16_index(nt);
633
+ /* Minus reads: stored SEQ is 5'->3' (SAM spec) but qpos walks
634
+ * CIGAR order (left->right); complement the base into the
635
+ * reference frame (parity with countmut 0.0.x + pysam pairs). */
636
+ if (s == 1 && base_i < 4) base_i = 3 - base_i;
594
637
  int qual = (int)bam_get_qual(b)[p->qpos];
595
638
  if (cfg->out == CM_OUT_CONVERSION) {
596
639
  site.cnt[s][(qual >= cfg->min_baseq) ? 2 : 0][base_i]++;
597
640
  } else {
598
- site.cnt[s][0][base_i]++;
641
+ /* router-assigned category slot (0..CM_CAT_MAX-1); g_a set in
642
+ * the selection loop for every kept candidate. */
643
+ int cat = w->g_a[i];
644
+ if (cat < 0) cat = 0; /* defensive */
645
+ site.cnt[s][cat][base_i]++;
599
646
  }
600
647
  }
601
648
 
@@ -604,7 +651,8 @@ static void count_interval(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, FI
604
651
  const int emit_minus = cfg->strand_process != CM_STRAND_FORWARD;
605
652
  /* -p site filter */
606
653
  if (w->expr && cm_expr_has_pile(w->expr)
607
- && !expr_pile_apply(w->expr, cfg, w, &site, pos, ref_ch))
654
+ && !expr_pile_apply(w->expr, cfg, w, hdr->target_name[tid], &site,
655
+ pos, ref_ch))
608
656
  continue;
609
657
  emit_site(w, cfg, hdr, fp, tid, pos, ref_ch, &site, emit_plus, emit_minus);
610
658
  }
@@ -659,7 +707,7 @@ static inline khint_t posi_hash(khint64_t key) {
659
707
  KHASH_INIT(posi, khint64_t, int, 1, posi_hash, kh_int64_hash_equal)
660
708
 
661
709
  /* winner of a (pos,qname) dedup bucket */
662
- typedef struct { int mapq, r1, qual, strand, base; } rw_w;
710
+ typedef struct { int mapq, r1, qual, strand, base, slot; } rw_w;
663
711
 
664
712
  /* growable map pos -> site_t (one entry per visited reference position) */
665
713
  typedef struct {
@@ -762,19 +810,29 @@ static rw_w *rw_add_base(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, int
762
810
  if ((int)qpos < cfg->trim_fragment_end || (int)qlen - (int)qpos <= cfg->trim_fragment_start) return wins;
763
811
  }
764
812
  if (read_trim_skip(b, (int)qpos, cfg->trim_r1_end, cfg->trim_r2_start)) return wins;
765
- if (w->expr && cm_expr_has_read(w->expr) && !cm_expr_read_constant(w->expr)
766
- && !cm_expr_read(w->expr, b, hdr->target_name[tid],
767
- (b->core.mtid >= 0 && b->core.mtid < hdr->n_targets)
768
- ? hdr->target_name[b->core.mtid] : "",
769
- (int)qpos, s ? -1 : 1,
770
- (ref_pos >= 0 && ref_pos < w->chr_len) ? w->chr_seq[ref_pos] : 'N'))
771
- return wins;
813
+ /* -e router: category slot (-1 = drop). Read-constant -e is applied once
814
+ * per read by the caller (see count_interval_readwalk); here we route the
815
+ * per-base (non-read-constant) case only, which is what a bq/qpos-based
816
+ * router needs. Default slot is 0. */
817
+ int rslot = 0;
818
+ if (w->expr && cm_expr_has_read(w->expr) && !cm_expr_read_constant(w->expr)) {
819
+ rslot = cm_expr_route(w->expr, b, hdr->target_name[tid],
820
+ (b->core.mtid >= 0 && b->core.mtid < hdr->n_targets)
821
+ ? hdr->target_name[b->core.mtid] : "",
822
+ (int)qpos, s ? -1 : 1,
823
+ (ref_pos >= 0 && ref_pos < w->chr_len) ? w->chr_seq[ref_pos] : 'N');
824
+ if (rslot < 0) return wins;
825
+ }
772
826
  uint8_t nt = bam_seqi(bam_get_seq(b), qpos);
773
- int base_i = nt16_index(nt); /* stored SEQ is reference-forward */
827
+ int base_i = nt16_index(nt);
828
+ /* Minus reads: stored SEQ is 5'->3' (SAM spec) but qpos walks CIGAR order
829
+ * (left->right); complement the base into the reference frame (parity
830
+ * with countmut 0.0.x + pysam pairs). */
831
+ if (s == 1 && base_i < 4) base_i = 3 - base_i;
774
832
  int qual = (int)bam_get_qual(b)[qpos];
775
833
  if (direct) {
776
834
  int cat = (cfg->out == CM_OUT_CONVERSION)
777
- ? ((qual >= cfg->min_baseq) ? 2 : 0) : 0;
835
+ ? ((qual >= cfg->min_baseq) ? 2 : 0) : rslot;
778
836
  sm_get(sm, ref_pos)->cnt[s][cat][base_i]++;
779
837
  return wins;
780
838
  }
@@ -790,13 +848,13 @@ static rw_w *rw_add_base(worker_t *w, const cm_config *cfg, bam_hdr_t *hdr, int
790
848
  }
791
849
  int idx = (*wins_n)++;
792
850
  wins[idx].mapq = mapq; wins[idx].r1 = r1; wins[idx].qual = qual;
793
- wins[idx].strand = s; wins[idx].base = base_i;
851
+ wins[idx].strand = s; wins[idx].base = base_i; wins[idx].slot = rslot;
794
852
  kh_val(h, kh) = idx;
795
853
  } else {
796
854
  int j = kh_val(h, kh);
797
855
  if (better(mapq, r1, qual, wins[j].mapq, wins[j].r1, wins[j].qual)) {
798
856
  wins[j].mapq = mapq; wins[j].r1 = r1; wins[j].qual = qual;
799
- wins[j].strand = s; wins[j].base = base_i;
857
+ wins[j].strand = s; wins[j].base = base_i; wins[j].slot = rslot;
800
858
  }
801
859
  }
802
860
  return wins;
@@ -975,7 +1033,7 @@ static void count_interval_readwalk(worker_t *w, const cm_config *cfg, bam_hdr_t
975
1033
  int64_t pos = kh_key(h, k).pos;
976
1034
  const rw_w *win = &wins[kh_val(h, k)];
977
1035
  int cat = (cfg->out == CM_OUT_CONVERSION)
978
- ? ((win->qual >= cfg->min_baseq) ? 2 : 0) : 0;
1036
+ ? ((win->qual >= cfg->min_baseq) ? 2 : 0) : win->slot;
979
1037
  site_t *st = sm_get(&sm, pos);
980
1038
  st->cnt[win->strand][cat][win->base]++;
981
1039
  }
@@ -995,7 +1053,8 @@ static void count_interval_readwalk(worker_t *w, const cm_config *cfg, bam_hdr_t
995
1053
  if (w->exc_bed && bed_overlap(w->exc_bed, hdr->target_name[tid], (int)pos, (int)pos + 1)) continue;
996
1054
  /* -p site filter */
997
1055
  if (w->expr && cm_expr_has_pile(w->expr)
998
- && !expr_pile_apply(w->expr, cfg, w, &sm.st[ord[i].idx], pos, ref_ch))
1056
+ && !expr_pile_apply(w->expr, cfg, w, hdr->target_name[tid],
1057
+ &sm.st[ord[i].idx], pos, ref_ch))
999
1058
  continue;
1000
1059
  emit_site(w, cfg, hdr, fp, tid, (int)pos, ref_ch, &sm.st[ord[i].idx], emit_plus, emit_minus);
1001
1060
  }
@@ -20,6 +20,23 @@ extern "C" {
20
20
  #define CM_ENGINE_READWALK 1
21
21
  #define CM_ENGINE_PILEUP 2
22
22
 
23
+ /* -e read-expression status categories. Slot 0 is the DEFAULT category
24
+ * (used when no -e filter is set, or the filter returns true); slots 1..K
25
+ * are the user-declared --status names (in declaration order). site_t keeps
26
+ * one [strand][category][base] count matrix, so the -p / output expressions
27
+ * can split the per-site counts by category (e.g. legacy u0/u1/u2 columns).
28
+ * Bumped from 3 -> 4 to fit the default + the three legacy tiers
29
+ * (low quality / high conversion / insufficient conversion). */
30
+ #define CM_CAT_MAX 4
31
+ #define CM_CAT_NAME_MAX 16
32
+
33
+ /* per-site accumulator: counts per biological strand, per -e status
34
+ * category, per base (0=A,1=C,2=G,3=T,4=N) */
35
+ typedef struct {
36
+ int cnt[2][CM_CAT_MAX][5]; /* [strand][category][base] */
37
+ int ins[2], del[2], refskip[2], fail[2];
38
+ } site_t;
39
+
23
40
  /* Strand processing */
24
41
  #define CM_STRAND_BOTH 0
25
42
  #define CM_STRAND_FORWARD 1
@@ -37,7 +54,7 @@ typedef struct {
37
54
  int mut_base;
38
55
  int ref_base2; /* legacy/unused: always 0 */
39
56
  int mut_base2;
40
- int pad; /* legacy/unused */
57
+ int pad; /* {motif} reference window: 2*pad+1 bases (--motif-pad) */
41
58
  int save_rest; /* legacy/unused */
42
59
  const char *output_expr; /* -o output-row template (overrides the built-in format) */
43
60
  const char *fmt_header; /* header line for a custom output template ("" = none) */
@@ -77,6 +77,7 @@ static void usage(void) {
77
77
  " --trim-r1-end N --trim-r2-start N read R1 3'-end / R2 5'-start trim\n"
78
78
  " --min-allele-support N --min-allele-frac F --min-strand-support N\n"
79
79
  " --min-depth N --mean-depth N\n"
80
+ " --motif-pad N {motif} reference window: 2*N+1 bases (0 = the base only)\n"
80
81
  " --count-indels [--strandless]\n"
81
82
  " --strand S both | forward | reverse\n"
82
83
  " --read-expr EXPR -e Lua read filter (evaluated per base)\n"
@@ -121,6 +122,7 @@ int main(int argc, char **argv) {
121
122
  cfg.max_depth = 0; /* 0 = unlimited (count all reads) */
122
123
  cfg.threads = 1;
123
124
  cfg.flanking = 0;
125
+ cfg.pad = 0; /* {motif} window: 2*pad+1 ref bases (0 = the base only) */
124
126
  cfg.req_flags = 0;
125
127
  cfg.excl_flags = 1796; /* samtools default: UNMAP|SECONDARY|QCFAIL|DUP */
126
128
  cfg.bedfile = NULL;
@@ -161,6 +163,7 @@ int main(int argc, char **argv) {
161
163
  {"ff", required_argument, 0, 1101},
162
164
  {"input-fmt-option", required_argument, 0, 1102},
163
165
  {"mate-fix", no_argument, 0, 1016},
166
+ {"motif-pad", required_argument, 0, 1019},
164
167
  {"read-expr", required_argument, 0, 2002},
165
168
  {"pile-expr", required_argument, 0, 2003},
166
169
  {"output-expr", required_argument, 0, 2004},
@@ -209,6 +212,7 @@ int main(int argc, char **argv) {
209
212
  case 1013: cfg.strandless = 1; break;
210
213
  case 1014: cfg.max_depth = atoi(optarg); break;
211
214
  case 1015: cfg.flanking = atoi(optarg); break;
215
+ case 1019: cfg.pad = atoi(optarg); break;
212
216
  default: usage(); return 1;
213
217
  }
214
218
  }