archaeopteryx 3.1.0 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/archaeopteryx.js +2 -2
- package/forester.js +276 -76
- package/package.json +1 -1
package/archaeopteryx.js
CHANGED
package/forester.js
CHANGED
|
@@ -20,8 +20,8 @@
|
|
|
20
20
|
*
|
|
21
21
|
*/
|
|
22
22
|
|
|
23
|
-
// v 3.
|
|
24
|
-
// 2026-09-
|
|
23
|
+
// v 3.2.0
|
|
24
|
+
// 2026-09-10
|
|
25
25
|
//
|
|
26
26
|
// forester.js is a general suite for dealing with phylogenetic trees.
|
|
27
27
|
//
|
|
@@ -1159,18 +1159,33 @@
|
|
|
1159
1159
|
return style;
|
|
1160
1160
|
};
|
|
1161
1161
|
|
|
1162
|
-
// The
|
|
1163
|
-
//
|
|
1164
|
-
//
|
|
1165
|
-
//
|
|
1166
|
-
//
|
|
1167
|
-
//
|
|
1168
|
-
//
|
|
1169
|
-
//
|
|
1170
|
-
//
|
|
1171
|
-
//
|
|
1172
|
-
//
|
|
1173
|
-
|
|
1162
|
+
// The share of tips that must carry a prefix for it to count as boilerplate.
|
|
1163
|
+
// A strict longest-common-prefix (1.0) lets a handful of oddly-named tips
|
|
1164
|
+
// veto the strip for everyone else: on the BV-BRC influenza tree 13,096 of
|
|
1165
|
+
// 13,246 tips share a 51-character prefix, but the other 0.8% drag the LCP
|
|
1166
|
+
// down to "A" and nothing is stripped at all. 0.95 fixes that with real
|
|
1167
|
+
// margin, and still refuses splits that would leave two groups of tips
|
|
1168
|
+
// incomparable -- at 0.67 the same corpus strips the country code from 78%
|
|
1169
|
+
// of tips while 22% keep their full names. DESIGNED JOINTLY WITH THE
|
|
1170
|
+
// DESKTOP, which uses the identical value: this rule and its output are
|
|
1171
|
+
// byte-identical across both programs, so the threshold is not ours alone
|
|
1172
|
+
// to change.
|
|
1173
|
+
const COMMON_PREFIX_QUANTILE = 0.95;
|
|
1174
|
+
|
|
1175
|
+
// The boring part of every tip name. When most displayed names share a
|
|
1176
|
+
// long prefix ("Influenza A virus ..."), a shortener that keeps the first
|
|
1177
|
+
// characters keeps exactly the characters that carry no information. This
|
|
1178
|
+
// returns the longest prefix shared by at least COMMON_PREFIX_QUANTILE of
|
|
1179
|
+
// the displayed external names -- the label property's value where one is
|
|
1180
|
+
// in effect, the node name otherwise -- cut back to the last separator so
|
|
1181
|
+
// no word is split, and only when it is long enough to matter
|
|
1182
|
+
// (>= 6 characters). The Short Names rendering strips it before
|
|
1183
|
+
// truncating, so what survives is the part that tells the tips apart. The
|
|
1184
|
+
// comparison is case-insensitive -- "Influenza A virus" and "Influenza A
|
|
1185
|
+
// Virus" are the same boring prefix -- so callers must strip by LENGTH,
|
|
1186
|
+
// comparing case-insensitively, not by exact match. Tips that do NOT
|
|
1187
|
+
// carry the prefix keep their full names; the caller's startsWith test
|
|
1188
|
+
// handles that without needing to know about the quantile.
|
|
1174
1189
|
forester.commonNamePrefix = function (tree, labelProperty) {
|
|
1175
1190
|
let names = [];
|
|
1176
1191
|
let slot = labelProperty ? {kind: 'property', ref: labelProperty} : null;
|
|
@@ -1192,29 +1207,72 @@
|
|
|
1192
1207
|
if (names.length < 2) {
|
|
1193
1208
|
return '';
|
|
1194
1209
|
}
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1210
|
+
// Any k names sharing a prefix are CONTIGUOUS once sorted, so
|
|
1211
|
+
// comparing each sorted name with the one k-1 places later finds the
|
|
1212
|
+
// longest prefix shared by k of them exactly -- no approximation, and
|
|
1213
|
+
// O(n log n + n*L) rather than the quadratic scan the obvious reading
|
|
1214
|
+
// suggests. At q = 1.0, k = n and this reduces to the strict LCP.
|
|
1215
|
+
let lower = names.map(function (s) {
|
|
1216
|
+
return s.toLowerCase();
|
|
1217
|
+
}).sort();
|
|
1218
|
+
let n = lower.length;
|
|
1219
|
+
let k = Math.ceil(COMMON_PREFIX_QUANTILE * n);
|
|
1220
|
+
if (k < 2) {
|
|
1221
|
+
k = 2;
|
|
1222
|
+
}
|
|
1223
|
+
if (k > n) {
|
|
1224
|
+
k = n;
|
|
1225
|
+
}
|
|
1226
|
+
let bestLen = 0;
|
|
1227
|
+
let bestIdx = -1;
|
|
1228
|
+
for (let i = 0; i + k - 1 < n; ++i) {
|
|
1229
|
+
let a = lower[i];
|
|
1230
|
+
let b = lower[i + k - 1];
|
|
1199
1231
|
let max = Math.min(a.length, b.length);
|
|
1200
|
-
let
|
|
1201
|
-
while (
|
|
1202
|
-
++
|
|
1232
|
+
let j = 0;
|
|
1233
|
+
while (j < max && a.charCodeAt(j) === b.charCodeAt(j)) {
|
|
1234
|
+
++j;
|
|
1203
1235
|
}
|
|
1204
|
-
if (
|
|
1205
|
-
|
|
1236
|
+
if (j > bestLen) {
|
|
1237
|
+
bestLen = j;
|
|
1238
|
+
bestIdx = i;
|
|
1206
1239
|
}
|
|
1207
1240
|
}
|
|
1208
|
-
if (
|
|
1241
|
+
if (bestLen === 0) {
|
|
1242
|
+
return '';
|
|
1243
|
+
}
|
|
1244
|
+
let lowerPrefix = lower[bestIdx].substring(0, bestLen);
|
|
1245
|
+
// Casing comes from the first name in TRAVERSAL order that carries the
|
|
1246
|
+
// prefix. At q = 1.0 that is names[0], which is exactly what the
|
|
1247
|
+
// strict-LCP version returned, so the two agree character for
|
|
1248
|
+
// character -- and the desktop mirrors this traversal order
|
|
1249
|
+
// deliberately, so the two programs do too.
|
|
1250
|
+
let prefix = null;
|
|
1251
|
+
let carriers = [];
|
|
1252
|
+
for (let ci = 0; ci < names.length; ++ci) {
|
|
1253
|
+
let name = names[ci];
|
|
1254
|
+
if (name.length >= bestLen
|
|
1255
|
+
&& name.substring(0, bestLen).toLowerCase() === lowerPrefix) {
|
|
1256
|
+
if (prefix === null) {
|
|
1257
|
+
prefix = name.substring(0, bestLen);
|
|
1258
|
+
}
|
|
1259
|
+
carriers.push(name);
|
|
1260
|
+
}
|
|
1261
|
+
}
|
|
1262
|
+
if (prefix === null) {
|
|
1209
1263
|
return '';
|
|
1210
1264
|
}
|
|
1211
1265
|
// Trim back to the last separator ONLY when the prefix actually
|
|
1212
1266
|
// splits a word -- "ABC_ho" against "ABC_house"/"ABC_horse" does,
|
|
1213
1267
|
// "Influenza A virus" against "...virus A/x" and "...virus(A/y)"
|
|
1214
|
-
// does not, whatever character each name continues with.
|
|
1268
|
+
// does not, whatever character each name continues with. Only the
|
|
1269
|
+
// names that CARRY the prefix get a say: one that does not share it
|
|
1270
|
+
// says nothing about whether the prefix splits a word, and letting it
|
|
1271
|
+
// vote throws the prefix away -- 19 tips of "ABCDEFG_..."/"ABCDEFG-..."
|
|
1272
|
+
// plus one unrelated longer tip yields "" instead of "ABCDEFG".
|
|
1215
1273
|
let alnum = /[A-Za-z0-9]/;
|
|
1216
1274
|
let splitsWord = alnum.test(prefix.charAt(prefix.length - 1))
|
|
1217
|
-
&&
|
|
1275
|
+
&& carriers.some(function (name) {
|
|
1218
1276
|
return name.length > prefix.length && alnum.test(name.charAt(prefix.length));
|
|
1219
1277
|
});
|
|
1220
1278
|
if (splitsWord) {
|
|
@@ -1792,6 +1850,35 @@
|
|
|
1792
1850
|
return (lo !== null && hi !== null) ? [lo, hi] : null;
|
|
1793
1851
|
}
|
|
1794
1852
|
|
|
1853
|
+
// Nexus/Newick quoting: a quoted token is wrapped in a matching pair, and a
|
|
1854
|
+
// literal quote INSIDE it is written twice. Reading one back therefore
|
|
1855
|
+
// means removing one matching outer pair and un-doubling what is inside --
|
|
1856
|
+
// 'Seba''s bat' is the single label "Seba's bat", not "Sebas bat", which is
|
|
1857
|
+
// what stripping every quote gave. The doubling is per quote, so a token
|
|
1858
|
+
// holding two escapes in a row un-doubles to two literal quotes; a pass
|
|
1859
|
+
// that removes quotes wholesale loses both.
|
|
1860
|
+
//
|
|
1861
|
+
// A token that is NOT well-formed -- an odd number of quotes, no closing
|
|
1862
|
+
// quote, a quote in the middle of a bare word -- is not a quoted token at
|
|
1863
|
+
// all. Those keep the old lenient behaviour of dropping stray quotes rather
|
|
1864
|
+
// than throwing: a viewer that refuses to open a file teaches the user
|
|
1865
|
+
// nothing, and Nexus in the wild is written by many programs. Strict on
|
|
1866
|
+
// output, lenient on input. Matches the desktop (0.11.141+).
|
|
1867
|
+
function unquoteLabel(s) {
|
|
1868
|
+
if (s === null || s === undefined) {
|
|
1869
|
+
return s;
|
|
1870
|
+
}
|
|
1871
|
+
let t = String(s).trim();
|
|
1872
|
+
let len = t.length;
|
|
1873
|
+
if (len > 1) {
|
|
1874
|
+
let q = t.charAt(0);
|
|
1875
|
+
if ((q === "'" || q === '"') && t.charAt(len - 1) === q) {
|
|
1876
|
+
return t.substring(1, len - 1).split(q + q).join(q);
|
|
1877
|
+
}
|
|
1878
|
+
}
|
|
1879
|
+
return t.replace(/['"]+/g, '');
|
|
1880
|
+
}
|
|
1881
|
+
|
|
1795
1882
|
function stripValueQuotes(v) {
|
|
1796
1883
|
if (v.length >= 2
|
|
1797
1884
|
&& ((v.charAt(0) === '"' && v.charAt(v.length - 1) === '"')
|
|
@@ -1986,6 +2073,19 @@
|
|
|
1986
2073
|
let in_double_q = false;
|
|
1987
2074
|
let in_single_q = false;
|
|
1988
2075
|
let buffer = '';
|
|
2076
|
+
// In a quoted label a literal quote is DOUBLED. The tokenizer splits
|
|
2077
|
+
// on quotes, so a doubled one arrives as two quote elements with only
|
|
2078
|
+
// empty strings between them -- that is the signature. Seeing it means
|
|
2079
|
+
// 'emit one quote and stay inside the run', not 'close the run': the
|
|
2080
|
+
// run continuing is what keeps the label in one piece, and emitting
|
|
2081
|
+
// the character is what stops the apostrophe being swallowed.
|
|
2082
|
+
let doubledQuoteEnd = function (at, qch) {
|
|
2083
|
+
let j = at + 1;
|
|
2084
|
+
while (j < ssl && ss[j] === '') {
|
|
2085
|
+
++j;
|
|
2086
|
+
}
|
|
2087
|
+
return (j < ssl && ss[j] === qch) ? j : -1;
|
|
2088
|
+
};
|
|
1989
2089
|
for (let i = 0; i < ssl; ++i) {
|
|
1990
2090
|
let element = ss[i].replace(/\s+/g, '');
|
|
1991
2091
|
|
|
@@ -1993,25 +2093,37 @@
|
|
|
1993
2093
|
if (!in_double_q) {
|
|
1994
2094
|
in_double_q = true;
|
|
1995
2095
|
} else {
|
|
1996
|
-
|
|
1997
|
-
if (
|
|
1998
|
-
|
|
2096
|
+
let dq = doubledQuoteEnd(i, '"');
|
|
2097
|
+
if (dq > -1) {
|
|
2098
|
+
buffer += '"';
|
|
2099
|
+
i = dq;
|
|
1999
2100
|
} else {
|
|
2000
|
-
|
|
2101
|
+
in_double_q = false;
|
|
2102
|
+
if (x.name && x.name.length > 0) {
|
|
2103
|
+
x.name = x.name + buffer;
|
|
2104
|
+
} else {
|
|
2105
|
+
x.name = buffer;
|
|
2106
|
+
}
|
|
2107
|
+
buffer = '';
|
|
2001
2108
|
}
|
|
2002
|
-
buffer = '';
|
|
2003
2109
|
}
|
|
2004
2110
|
} else if (element === "'" && !in_double_q) {
|
|
2005
2111
|
if (!in_single_q) {
|
|
2006
2112
|
in_single_q = true;
|
|
2007
2113
|
} else {
|
|
2008
|
-
|
|
2009
|
-
if (
|
|
2010
|
-
|
|
2114
|
+
let dq = doubledQuoteEnd(i, "'");
|
|
2115
|
+
if (dq > -1) {
|
|
2116
|
+
buffer += "'";
|
|
2117
|
+
i = dq;
|
|
2011
2118
|
} else {
|
|
2012
|
-
|
|
2119
|
+
in_single_q = false;
|
|
2120
|
+
if (x.name && x.name.length > 0) {
|
|
2121
|
+
x.name = x.name + buffer;
|
|
2122
|
+
} else {
|
|
2123
|
+
x.name = buffer;
|
|
2124
|
+
}
|
|
2125
|
+
buffer = '';
|
|
2013
2126
|
}
|
|
2014
|
-
buffer = '';
|
|
2015
2127
|
}
|
|
2016
2128
|
} else {
|
|
2017
2129
|
if (in_double_q || in_single_q) {
|
|
@@ -2284,10 +2396,14 @@
|
|
|
2284
2396
|
throw new Error(NEXUS_FORMAT_ERR + 'ill-formatted translate table entry: "'
|
|
2285
2397
|
+ pair.trim() + '" -- is the Translate sub-command terminated with a ";"?');
|
|
2286
2398
|
}
|
|
2287
|
-
|
|
2399
|
+
// The sub-command terminator comes off BEFORE unquoting, or a
|
|
2400
|
+
// well-formed quoted value followed by ';' would not look
|
|
2401
|
+
// well-formed and would fall to the lenient path.
|
|
2402
|
+
let value = m[2].trim();
|
|
2288
2403
|
if (value.endsWith(';')) {
|
|
2289
|
-
value = value.slice(0, -1);
|
|
2404
|
+
value = value.slice(0, -1).trim();
|
|
2290
2405
|
}
|
|
2406
|
+
value = unquoteLabel(value);
|
|
2291
2407
|
translateMap[m[1]] = value;
|
|
2292
2408
|
});
|
|
2293
2409
|
}
|
|
@@ -2371,14 +2487,38 @@
|
|
|
2371
2487
|
for (let id in seqs) {
|
|
2372
2488
|
seqsByKey[joinKey(id)] = seqs[id];
|
|
2373
2489
|
}
|
|
2374
|
-
forester.getAllExternalNodes(phy)
|
|
2490
|
+
let externals = forester.getAllExternalNodes(phy);
|
|
2491
|
+
// A bare integer tip name counts as a TAXLABELS index only when
|
|
2492
|
+
// the WHOLE tree reads as index references: every tip a bare
|
|
2493
|
+
// integer AND every one of them in range. All-or-nothing, because
|
|
2494
|
+
// deciding it per tip fails silently and plausibly -- against six
|
|
2495
|
+
// labels, ((a,b,c),(1,2,3)) renamed just the three integers and
|
|
2496
|
+
// handed back a tree whose every tip was DUPLICATED, which still
|
|
2497
|
+
// parses and still renders; and a single out-of-range index left a
|
|
2498
|
+
// half-renamed tree behind for the same reason. Both are reachable
|
|
2499
|
+
// through our own writer: save as Nexus, reopen. So if any tip is
|
|
2500
|
+
// not an index, none of them are. Matches the desktop (0.11.140+).
|
|
2501
|
+
// A TRANSLATE entry still wins wherever it applies -- it is the
|
|
2502
|
+
// explicit mechanism, this is only the heuristic.
|
|
2503
|
+
let indexed = taxlabels.length > 0 && externals.every(function (node) {
|
|
2504
|
+
if (node.name && translateMap[node.name] !== undefined) {
|
|
2505
|
+
return true;
|
|
2506
|
+
}
|
|
2507
|
+
if (!node.name || !/^\d+$/.test(node.name)) {
|
|
2508
|
+
return false;
|
|
2509
|
+
}
|
|
2510
|
+
let i = parseInt(node.name, 10);
|
|
2511
|
+
return i > 0 && i <= taxlabels.length;
|
|
2512
|
+
});
|
|
2513
|
+
externals.forEach(function (node) {
|
|
2375
2514
|
if (node.name && translateMap[node.name] !== undefined) {
|
|
2376
2515
|
node.name = translateMap[node.name];
|
|
2377
|
-
} else if (
|
|
2378
|
-
|
|
2379
|
-
|
|
2380
|
-
|
|
2381
|
-
|
|
2516
|
+
} else if (indexed) {
|
|
2517
|
+
// The TAXLABELS tokenizer has already removed the outer
|
|
2518
|
+
// quotes and un-doubled what was inside, so the label is
|
|
2519
|
+
// used as-is: stripping quotes again here would undo the
|
|
2520
|
+
// un-doubling and drop the apostrophe a second time.
|
|
2521
|
+
node.name = taxlabels[parseInt(node.name, 10) - 1];
|
|
2382
2522
|
}
|
|
2383
2523
|
if (node.name) {
|
|
2384
2524
|
let s = seqsByKey[joinKey(node.name)];
|
|
@@ -2457,7 +2597,7 @@
|
|
|
2457
2597
|
inTree = true;
|
|
2458
2598
|
let nm = TREE_NAME_RE.exec(line);
|
|
2459
2599
|
if (nm) {
|
|
2460
|
-
name = nm[1]
|
|
2600
|
+
name = unquoteLabel(nm[1]);
|
|
2461
2601
|
}
|
|
2462
2602
|
let rm = ROOTEDNESS_RE.exec(line);
|
|
2463
2603
|
if (rm) {
|
|
@@ -2489,24 +2629,56 @@
|
|
|
2489
2629
|
// silently sheared such labels apart and shifted every
|
|
2490
2630
|
// numeric tip onto the wrong name. ';' (unquoted) ends
|
|
2491
2631
|
// the sub-command.
|
|
2632
|
+
//
|
|
2633
|
+
// A quote OPENS a run only at a TOKEN BOUNDARY. Inside a
|
|
2634
|
+
// word it is just a character -- an unquoted O'Neil must
|
|
2635
|
+
// not open a run and swallow the rest of the line, the
|
|
2636
|
+
// terminating ';' included, which is what the first
|
|
2637
|
+
// version of this tokenizer did: it merged every remaining
|
|
2638
|
+
// label into one and left the other tips as bare numbers.
|
|
2492
2639
|
let tok = '';
|
|
2493
2640
|
let q = null;
|
|
2641
|
+
let closed = false; // this token already held a quoted run
|
|
2494
2642
|
let push = function () {
|
|
2495
2643
|
if (tok.length > 0 && tok.toLowerCase() !== 'taxlabels') {
|
|
2496
2644
|
taxlabels.push(tok);
|
|
2497
2645
|
}
|
|
2498
2646
|
tok = '';
|
|
2647
|
+
closed = false;
|
|
2499
2648
|
};
|
|
2500
2649
|
for (let ci = 0; ci < line.length; ++ci) {
|
|
2501
2650
|
let ch = line.charAt(ci);
|
|
2502
2651
|
if (q) {
|
|
2503
2652
|
if (ch === q) {
|
|
2504
2653
|
q = null;
|
|
2654
|
+
closed = true;
|
|
2505
2655
|
} else {
|
|
2506
2656
|
tok += ch;
|
|
2507
2657
|
}
|
|
2508
2658
|
} else if (ch === "'" || ch === '"') {
|
|
2509
|
-
|
|
2659
|
+
if (tok.length === 0 && !closed) {
|
|
2660
|
+
q = ch;
|
|
2661
|
+
} else if (closed) {
|
|
2662
|
+
// A quote directly after a closing one is the
|
|
2663
|
+
// doubled Nexus escape: emit ONE literal quote
|
|
2664
|
+
// and let the run CONTINUE. The continuing is
|
|
2665
|
+
// what keeps a quoted label holding an
|
|
2666
|
+
// apostrophe in one piece instead of splitting
|
|
2667
|
+
// it at the space (that was N1); emitting the
|
|
2668
|
+
// character is the un-doubling half, deferred
|
|
2669
|
+
// until both programs could land it together.
|
|
2670
|
+
tok += ch;
|
|
2671
|
+
q = ch;
|
|
2672
|
+
} else {
|
|
2673
|
+
// A quote in the middle of a BARE word is not
|
|
2674
|
+
// an escape and not a delimiter -- an unquoted
|
|
2675
|
+
// token may not legally hold one at all. It is
|
|
2676
|
+
// dropped rather than kept, which is the older
|
|
2677
|
+
// lenient behaviour and is SHARED with the
|
|
2678
|
+
// desktop; keeping it here would be a new
|
|
2679
|
+
// divergence, not a fix.
|
|
2680
|
+
void ch;
|
|
2681
|
+
}
|
|
2510
2682
|
} else if (ch === ' ') {
|
|
2511
2683
|
push();
|
|
2512
2684
|
} else if (ch === ';') {
|
|
@@ -2943,16 +3115,55 @@
|
|
|
2943
3115
|
};
|
|
2944
3116
|
|
|
2945
3117
|
|
|
3118
|
+
// How a label is written into Newick or Nexus, ported from the desktop's
|
|
3119
|
+
// ForesterUtil.santitizeStringForNH so both programs emit the same token
|
|
3120
|
+
// for the same name. Quoting rather than transliterating is what makes a
|
|
3121
|
+
// save-and-reopen lossless: the previous rule mapped every quote, comma,
|
|
3122
|
+
// paren and space to '_', which no reader can undo, so a tip named
|
|
3123
|
+
// "Cooper's Hawk" came back "Cooper_s_Hawk".
|
|
3124
|
+
//
|
|
3125
|
+
// The one case that still loses information is a name carrying BOTH quote
|
|
3126
|
+
// styles: there is no quote character left to wrap it in, so the
|
|
3127
|
+
// apostrophes become backticks. The desktop does the same, deliberately;
|
|
3128
|
+
// that case is a JOINT open item and is NOT to be fixed on one side.
|
|
3129
|
+
//
|
|
3130
|
+
// Note the asymmetry with the READER: a doubled '' is how a quote is
|
|
3131
|
+
// escaped INSIDE a quoted token, and our reader un-doubles it, but neither
|
|
3132
|
+
// writer produces that form -- both sidestep it by switching quote style.
|
|
3133
|
+
// Reading a form you do not write is intentional here.
|
|
3134
|
+
function sanitizeLabelForNH(s) {
|
|
3135
|
+
let t = String(s).replace(/\s+/g, ' ').trim();
|
|
3136
|
+
let hasSingle = t.indexOf("'") > -1;
|
|
3137
|
+
let hasDouble = t.indexOf('"') > -1;
|
|
3138
|
+
if (hasSingle && hasDouble) {
|
|
3139
|
+
return "'" + t.replace(/'/g, '`') + "'";
|
|
3140
|
+
}
|
|
3141
|
+
if (hasSingle) {
|
|
3142
|
+
return '"' + t + '"';
|
|
3143
|
+
}
|
|
3144
|
+
if (hasDouble || /[\s,():;[\]]/.test(t)) {
|
|
3145
|
+
return "'" + t + "'";
|
|
3146
|
+
}
|
|
3147
|
+
return t;
|
|
3148
|
+
}
|
|
3149
|
+
|
|
2946
3150
|
/**
|
|
2947
3151
|
* To convert a phylogentic tree object to a New Hampshire (Newick) formatted string.
|
|
2948
3152
|
*
|
|
2949
3153
|
* @param phy - A phylogentic tree object.
|
|
2950
3154
|
* @param decPointsMax - Maximal number of decimal points for branch lengths (optional)
|
|
2951
|
-
* @param replaceChars -
|
|
3155
|
+
* @param replaceChars - RETIRED 2026-09-10 and ignored. It used to map
|
|
3156
|
+
* every space, comma, paren, colon, semicolon, bracket and quote to
|
|
3157
|
+
* '_', which no reader can undo: a tip named "Cooper's Hawk" was
|
|
3158
|
+
* written Cooper_s_Hawk and came back that way. Labels are now
|
|
3159
|
+
* always quoted instead, by the same rule the desktop uses, so a
|
|
3160
|
+
* save-and-reopen keeps the name. The parameter is still accepted
|
|
3161
|
+
* so positional callers keep working.
|
|
2952
3162
|
* @param writeConfidences - to write confidence values in brackets
|
|
2953
3163
|
* @returns {*} - a New Hampshire (Newick) formatted string.
|
|
2954
3164
|
*/
|
|
2955
3165
|
forester.toNewHampshire = function (phy, decPointsMax, replaceChars, writeConfidences) {
|
|
3166
|
+
void replaceChars; // retired: see the note above; labels are always quoted now
|
|
2956
3167
|
let nh = "";
|
|
2957
3168
|
if (phy.children && phy.children.length === 1) {
|
|
2958
3169
|
toNewHampshireHelper(phy.children[0], true);
|
|
@@ -2979,22 +3190,7 @@
|
|
|
2979
3190
|
nh += ")";
|
|
2980
3191
|
}
|
|
2981
3192
|
if (node.name && node.name.length > 0) {
|
|
2982
|
-
|
|
2983
|
-
nh += replaceUnsafeChars(node.name);
|
|
2984
|
-
} else {
|
|
2985
|
-
let myName = node.name.replace(/\s+/g, ' ');
|
|
2986
|
-
if (/[\s,():;'"[\]]/.test(myName)) {
|
|
2987
|
-
if ((myName.indexOf('"') > -1) && (myName.indexOf("'") > -1)) {
|
|
2988
|
-
nh += '"' + myName.replace(/"/g, "'") + '"';
|
|
2989
|
-
} else if (myName.indexOf('"') > -1) {
|
|
2990
|
-
nh += "'" + myName + "'";
|
|
2991
|
-
} else {
|
|
2992
|
-
nh += '"' + myName + '"';
|
|
2993
|
-
}
|
|
2994
|
-
} else {
|
|
2995
|
-
nh += myName;
|
|
2996
|
-
}
|
|
2997
|
-
}
|
|
3193
|
+
nh += sanitizeLabelForNH(node.name);
|
|
2998
3194
|
}
|
|
2999
3195
|
if (node.branch_length !== undefined && node.branch_length !== null) {
|
|
3000
3196
|
if (decPointsMax && decPointsMax > 0) {
|
|
@@ -3015,9 +3211,6 @@
|
|
|
3015
3211
|
}
|
|
3016
3212
|
}
|
|
3017
3213
|
|
|
3018
|
-
function replaceUnsafeChars(str) {
|
|
3019
|
-
return str.replace(/[\s,():;'"[\]]+/g, '_');
|
|
3020
|
-
}
|
|
3021
3214
|
};
|
|
3022
3215
|
|
|
3023
3216
|
// Writes a phylogeny as a Nexus-formatted string, ported from the
|
|
@@ -3029,11 +3222,16 @@
|
|
|
3029
3222
|
// carrying the tree and its alignment in one file is the point of Nexus,
|
|
3030
3223
|
// and parseNexus reads the alignment back onto the tips.
|
|
3031
3224
|
forester.toNexus = function (phy, decPointsMax, writeConfidences) {
|
|
3032
|
-
// the
|
|
3033
|
-
//
|
|
3034
|
-
|
|
3035
|
-
|
|
3036
|
-
|
|
3225
|
+
// The TaxLabels tokens, the Matrix row labels and the tree's tip
|
|
3226
|
+
// tokens must be byte-identical or nothing can join them back up, so
|
|
3227
|
+
// all three go through sanitizeLabelForNH -- the same helper
|
|
3228
|
+
// toNewHampshire writes the tree with.
|
|
3229
|
+
//
|
|
3230
|
+
// nexusLabel returns the label UNQUOTED, because it is also assigned
|
|
3231
|
+
// to node.name for a nameless tip and toNewHampshire quotes it again
|
|
3232
|
+
// on the way out. The old '_' substitution was idempotent so applying
|
|
3233
|
+
// it twice was harmless; quoting is not, and would emit "'a b'"
|
|
3234
|
+
// wrapped in quotes a second time.
|
|
3037
3235
|
|
|
3038
3236
|
// label preference as on the desktop: name, then taxonomy
|
|
3039
3237
|
// (code/scientific/common), then sequence (name/symbol/gene)
|
|
@@ -3051,7 +3249,7 @@
|
|
|
3051
3249
|
if (!s) {
|
|
3052
3250
|
s = 'node' + (i + 1); // an empty TaxLabels token would not parse back
|
|
3053
3251
|
}
|
|
3054
|
-
return
|
|
3252
|
+
return s;
|
|
3055
3253
|
}
|
|
3056
3254
|
|
|
3057
3255
|
let ext = forester.getAllExternalNodes(phy).reverse();
|
|
@@ -3071,7 +3269,7 @@
|
|
|
3071
3269
|
s += ' Dimensions NTax=' + ext.length + ';\n';
|
|
3072
3270
|
s += ' TaxLabels';
|
|
3073
3271
|
ext.forEach(function (node, i) {
|
|
3074
|
-
s += ' ' + nexusLabel(node, i);
|
|
3272
|
+
s += ' ' + sanitizeLabelForNH(nexusLabel(node, i));
|
|
3075
3273
|
});
|
|
3076
3274
|
s += ';\n';
|
|
3077
3275
|
s += 'End;\n';
|
|
@@ -3086,7 +3284,7 @@
|
|
|
3086
3284
|
for (let j = 0; j < node.sequences.length; ++j) {
|
|
3087
3285
|
let q = node.sequences[j];
|
|
3088
3286
|
if (q.mol_seq && q.mol_seq.is_aligned && q.mol_seq.value) {
|
|
3089
|
-
rows.push({label: nexusLabel(node, i), value: q.mol_seq.value});
|
|
3287
|
+
rows.push({label: sanitizeLabelForNH(nexusLabel(node, i)), value: q.mol_seq.value});
|
|
3090
3288
|
nchar = Math.max(nchar, q.mol_seq.value.length);
|
|
3091
3289
|
if (!datatype && (q.type === 'protein' || q.type === 'dna' || q.type === 'rna')) {
|
|
3092
3290
|
datatype = q.type;
|
|
@@ -3121,8 +3319,10 @@
|
|
|
3121
3319
|
}
|
|
3122
3320
|
|
|
3123
3321
|
s += 'Begin Trees;\n';
|
|
3124
|
-
|
|
3125
|
-
|
|
3322
|
+
// the tree name was stripped of its quotes for the same reason the tip
|
|
3323
|
+
// names were transliterated, and loses an apostrophe the same way
|
|
3324
|
+
let treeName = phy.name ? String(phy.name).trim() : '';
|
|
3325
|
+
s += ' Tree ' + (treeName ? sanitizeLabelForNH(treeName) : 'tree1') + '=';
|
|
3126
3326
|
s += (phy.rooted === false) ? '[&U]' : '[&R]';
|
|
3127
3327
|
let nh = forester.toNewHampshire(phy, decPointsMax, true, writeConfidences);
|
|
3128
3328
|
renamed.forEach(function (node) {
|
package/package.json
CHANGED