archaeopteryx 3.6.1 → 3.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +97 -5
- package/archaeopteryx.d.ts +6 -1
- package/archaeopteryx.js +654 -21
- package/forester.js +1191 -58
- package/package.json +1 -1
package/forester.js
CHANGED
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
*
|
|
21
21
|
*/
|
|
22
22
|
|
|
23
|
-
// v 3.
|
|
23
|
+
// v 3.8.0
|
|
24
24
|
// 2026-09-10
|
|
25
25
|
//
|
|
26
26
|
// forester.js is a general suite for dealing with phylogenetic trees.
|
|
@@ -781,6 +781,557 @@
|
|
|
781
781
|
});
|
|
782
782
|
};
|
|
783
783
|
|
|
784
|
+
// ---- representative tips ------------------------------------------------
|
|
785
|
+
//
|
|
786
|
+
// Tree-based dereplication, the desktop's "Select Representative Tips"
|
|
787
|
+
// (RepresentativeTipSelector) ported step for step, the order of its sums
|
|
788
|
+
// and its 1e-9 tolerance included: test/fixtures/rep-contract.tsv holds
|
|
789
|
+
// the desktop's own results and extractions (RepContract.java), and the
|
|
790
|
+
// tests hold ours to them.
|
|
791
|
+
//
|
|
792
|
+
// Tips are grouped into the maximal clades whose diameter -- the largest
|
|
793
|
+
// patristic distance between two of their tips -- is at most a cutoff, and
|
|
794
|
+
// each group keeps one representative. A clade's diameter only grows
|
|
795
|
+
// rootward, so the groups are one cut through the tree. Without branch
|
|
796
|
+
// lengths the distance is topological, one unit per edge.
|
|
797
|
+
|
|
798
|
+
forester.REPRESENTATIVE_MEDOID = 'medoid';
|
|
799
|
+
forester.REPRESENTATIVE_LONGEST_BRANCH = 'longest_branch';
|
|
800
|
+
|
|
801
|
+
const REPRESENTATIVE_EPS = 1e-9;
|
|
802
|
+
|
|
803
|
+
function repIsTip(n) {
|
|
804
|
+
return !n.children || n.children.length === 0;
|
|
805
|
+
}
|
|
806
|
+
|
|
807
|
+
// below `top`, parents before children, children in their order
|
|
808
|
+
function repPreorder(top) {
|
|
809
|
+
let out = [];
|
|
810
|
+
let stack = [top];
|
|
811
|
+
while (stack.length > 0) {
|
|
812
|
+
let n = stack.pop();
|
|
813
|
+
out.push(n);
|
|
814
|
+
if (n.children) {
|
|
815
|
+
for (let i = n.children.length - 1; i >= 0; --i) {
|
|
816
|
+
stack.push(n.children[i]);
|
|
817
|
+
}
|
|
818
|
+
}
|
|
819
|
+
}
|
|
820
|
+
return out;
|
|
821
|
+
}
|
|
822
|
+
|
|
823
|
+
// the branch above a node, a missing or negative length counting 0
|
|
824
|
+
function repEdge(n, topological) {
|
|
825
|
+
if (topological) {
|
|
826
|
+
return 1;
|
|
827
|
+
}
|
|
828
|
+
let d = n.branch_length;
|
|
829
|
+
return (typeof d === 'number' && d > 0) ? d : 0;
|
|
830
|
+
}
|
|
831
|
+
|
|
832
|
+
/**
|
|
833
|
+
* Whether any branch of the tree carries a length (0 included). Without
|
|
834
|
+
* one, representative tips are chosen by topological distance, and a
|
|
835
|
+
* distance cutoff means nothing.
|
|
836
|
+
*
|
|
837
|
+
* @param phy the tree
|
|
838
|
+
* @returns {boolean}
|
|
839
|
+
*/
|
|
840
|
+
forester.hasUsableBranchLengths = function (phy) {
|
|
841
|
+
let root = forester.getTreeRoot(phy);
|
|
842
|
+
return repPreorder(root).some(function (n) {
|
|
843
|
+
return n !== root && typeof n.branch_length === 'number';
|
|
844
|
+
});
|
|
845
|
+
};
|
|
846
|
+
|
|
847
|
+
// every node's diameter, in one pass from the tips up
|
|
848
|
+
function repDiameters(pre, topological) {
|
|
849
|
+
let height = new Map();
|
|
850
|
+
let diameter = new Map();
|
|
851
|
+
for (let i = pre.length - 1; i >= 0; --i) {
|
|
852
|
+
let n = pre[i];
|
|
853
|
+
if (repIsTip(n)) {
|
|
854
|
+
height.set(n, 0);
|
|
855
|
+
diameter.set(n, 0);
|
|
856
|
+
continue;
|
|
857
|
+
}
|
|
858
|
+
let best1 = 0; // the longest reach down through one child
|
|
859
|
+
let best2 = 0; // through another
|
|
860
|
+
let maxChildDiameter = 0;
|
|
861
|
+
for (let c = 0; c < n.children.length; ++c) {
|
|
862
|
+
let child = n.children[c];
|
|
863
|
+
let reach = height.get(child) + repEdge(child, topological);
|
|
864
|
+
if (reach >= best1) {
|
|
865
|
+
best2 = best1;
|
|
866
|
+
best1 = reach;
|
|
867
|
+
} else if (reach > best2) {
|
|
868
|
+
best2 = reach;
|
|
869
|
+
}
|
|
870
|
+
maxChildDiameter = Math.max(maxChildDiameter, diameter.get(child));
|
|
871
|
+
}
|
|
872
|
+
height.set(n, best1);
|
|
873
|
+
diameter.set(n, Math.max(maxChildDiameter, n.children.length >= 2 ? best1 + best2 : 0));
|
|
874
|
+
}
|
|
875
|
+
return diameter;
|
|
876
|
+
}
|
|
877
|
+
|
|
878
|
+
// the groups' clades: the highest nodes whose diameter is within the cutoff
|
|
879
|
+
function repGroupRoots(root, diameter, cutoff) {
|
|
880
|
+
let roots = [];
|
|
881
|
+
let stack = [root];
|
|
882
|
+
while (stack.length > 0) {
|
|
883
|
+
let n = stack.pop();
|
|
884
|
+
if (repIsTip(n) || diameter.get(n) <= cutoff + REPRESENTATIVE_EPS) {
|
|
885
|
+
roots.push(n);
|
|
886
|
+
} else {
|
|
887
|
+
for (let i = 0; i < n.children.length; ++i) {
|
|
888
|
+
stack.push(n.children[i]);
|
|
889
|
+
}
|
|
890
|
+
}
|
|
891
|
+
}
|
|
892
|
+
return roots;
|
|
893
|
+
}
|
|
894
|
+
|
|
895
|
+
// The cutoff whose group count comes closest to the target: the count
|
|
896
|
+
// changes only at clade diameters and never grows with the cutoff, so a
|
|
897
|
+
// binary search finds the smallest diameter giving at most the target,
|
|
898
|
+
// and its neighbour below gives more. A tie keeps more representatives;
|
|
899
|
+
// -1 stands for "below every diameter", every tip its own group.
|
|
900
|
+
function repCutoffForTarget(pre, root, diameter, target, tipCount) {
|
|
901
|
+
let values = [];
|
|
902
|
+
pre.forEach(function (n) {
|
|
903
|
+
if (!repIsTip(n)) {
|
|
904
|
+
values.push(diameter.get(n));
|
|
905
|
+
}
|
|
906
|
+
});
|
|
907
|
+
values.sort(function (a, b) { return a - b; });
|
|
908
|
+
let cand = [];
|
|
909
|
+
values.forEach(function (v) {
|
|
910
|
+
if (cand.length === 0 || v > cand[cand.length - 1] + REPRESENTATIVE_EPS) {
|
|
911
|
+
cand.push(v);
|
|
912
|
+
}
|
|
913
|
+
});
|
|
914
|
+
let count = function (cutoff) {
|
|
915
|
+
return repGroupRoots(root, diameter, cutoff).length;
|
|
916
|
+
};
|
|
917
|
+
let lo = 0;
|
|
918
|
+
let hi = cand.length - 1;
|
|
919
|
+
let boundary = cand.length - 1;
|
|
920
|
+
while (lo <= hi) {
|
|
921
|
+
let mid = (lo + hi) >>> 1;
|
|
922
|
+
if (count(cand[mid]) <= target) {
|
|
923
|
+
boundary = mid;
|
|
924
|
+
hi = mid - 1;
|
|
925
|
+
} else {
|
|
926
|
+
lo = mid + 1;
|
|
927
|
+
}
|
|
928
|
+
}
|
|
929
|
+
let tHigh = cand[boundary];
|
|
930
|
+
let cHigh = count(tHigh);
|
|
931
|
+
let tLow = boundary > 0 ? cand[boundary - 1] : -1;
|
|
932
|
+
let cLow = boundary > 0 ? count(tLow) : tipCount;
|
|
933
|
+
return (Math.abs(cLow - target) <= Math.abs(cHigh - target)) ? tLow : tHigh;
|
|
934
|
+
}
|
|
935
|
+
|
|
936
|
+
function repLongestBranch(members, topological) {
|
|
937
|
+
let best = members[0];
|
|
938
|
+
let bestLength = repEdge(best, topological);
|
|
939
|
+
for (let i = 1; i < members.length; ++i) {
|
|
940
|
+
let length = repEdge(members[i], topological);
|
|
941
|
+
if (length > bestLength + REPRESENTATIVE_EPS) {
|
|
942
|
+
best = members[i];
|
|
943
|
+
bestLength = length;
|
|
944
|
+
}
|
|
945
|
+
}
|
|
946
|
+
return best;
|
|
947
|
+
}
|
|
948
|
+
|
|
949
|
+
// The tip with the smallest summed distance to its group-mates, by the
|
|
950
|
+
// rerooting sum-of-distances recursion (linear, never all pairs): `down`
|
|
951
|
+
// sums the distances to the tips below a node, `up` to all the others.
|
|
952
|
+
function repMedoid(pre, members, topological) {
|
|
953
|
+
let tipsBelow = new Map();
|
|
954
|
+
let down = new Map();
|
|
955
|
+
for (let i = pre.length - 1; i >= 0; --i) {
|
|
956
|
+
let n = pre[i];
|
|
957
|
+
if (repIsTip(n)) {
|
|
958
|
+
tipsBelow.set(n, 1);
|
|
959
|
+
down.set(n, 0);
|
|
960
|
+
continue;
|
|
961
|
+
}
|
|
962
|
+
let s = 0;
|
|
963
|
+
let d = 0;
|
|
964
|
+
for (let c = 0; c < n.children.length; ++c) {
|
|
965
|
+
let child = n.children[c];
|
|
966
|
+
let e = repEdge(child, topological);
|
|
967
|
+
let cs = tipsBelow.get(child);
|
|
968
|
+
s += cs;
|
|
969
|
+
d += down.get(child) + (e * cs);
|
|
970
|
+
}
|
|
971
|
+
tipsBelow.set(n, s);
|
|
972
|
+
down.set(n, d);
|
|
973
|
+
}
|
|
974
|
+
let total = tipsBelow.get(pre[0]);
|
|
975
|
+
let up = new Map();
|
|
976
|
+
up.set(pre[0], 0);
|
|
977
|
+
pre.forEach(function (n) {
|
|
978
|
+
if (repIsTip(n)) {
|
|
979
|
+
return;
|
|
980
|
+
}
|
|
981
|
+
let un = up.get(n);
|
|
982
|
+
let dn = down.get(n);
|
|
983
|
+
for (let c = 0; c < n.children.length; ++c) {
|
|
984
|
+
let child = n.children[c];
|
|
985
|
+
let e = repEdge(child, topological);
|
|
986
|
+
let cs = tipsBelow.get(child);
|
|
987
|
+
up.set(child, un + dn - down.get(child) - (e * cs) + (e * (total - cs)));
|
|
988
|
+
}
|
|
989
|
+
});
|
|
990
|
+
let best = members[0];
|
|
991
|
+
let bestTotal = up.get(best);
|
|
992
|
+
for (let i = 1; i < members.length; ++i) {
|
|
993
|
+
let t = up.get(members[i]);
|
|
994
|
+
if (t < bestTotal - REPRESENTATIVE_EPS) {
|
|
995
|
+
best = members[i];
|
|
996
|
+
bestTotal = t;
|
|
997
|
+
}
|
|
998
|
+
}
|
|
999
|
+
return best;
|
|
1000
|
+
}
|
|
1001
|
+
|
|
1002
|
+
/**
|
|
1003
|
+
* Chooses representative tips: groups the tips into the maximal clades
|
|
1004
|
+
* whose members are all within a distance of each other, and keeps one
|
|
1005
|
+
* tip per group -- the desktop's Select Representative Tips.
|
|
1006
|
+
*
|
|
1007
|
+
* By `cutoff`, no two tips of a group are farther apart than it. By
|
|
1008
|
+
* `target`, the cutoff is the one whose group count comes closest to that
|
|
1009
|
+
* many (a tie keeps more); the count made is reported, since only certain
|
|
1010
|
+
* counts are possible. Without branch lengths the distance counts edges.
|
|
1011
|
+
*
|
|
1012
|
+
* A protected tip is never dropped: it stands in for its group's
|
|
1013
|
+
* representative, and a group with several keeps them all -- which can
|
|
1014
|
+
* keep more tips than the target.
|
|
1015
|
+
*
|
|
1016
|
+
* The tree is not changed.
|
|
1017
|
+
*
|
|
1018
|
+
* @param phy the tree
|
|
1019
|
+
* @param options {cutoff: number} or {target: integer}, with
|
|
1020
|
+
* pick: forester.REPRESENTATIVE_MEDOID (default, the most central
|
|
1021
|
+
* tip) or forester.REPRESENTATIVE_LONGEST_BRANCH (the most
|
|
1022
|
+
* divergent), and protectedTips: tips (nodes) never to drop
|
|
1023
|
+
* @returns {{groups: Array, keptTips: Array, keptCount: number,
|
|
1024
|
+
* protectedKeptCount: number, tipCount: number,
|
|
1025
|
+
* effectiveCutoff: number, topological: boolean, pick: string,
|
|
1026
|
+
* requestedTarget: number, summary: string}}
|
|
1027
|
+
* groups: {clade, members, kept}, in tree order of their first
|
|
1028
|
+
* kept tip; requestedTarget is -1 for a cutoff
|
|
1029
|
+
*/
|
|
1030
|
+
forester.selectRepresentativeTips = function (phy, options) {
|
|
1031
|
+
let opts = options || {};
|
|
1032
|
+
let root = phy ? forester.getTreeRoot(phy) : null;
|
|
1033
|
+
if (!root) {
|
|
1034
|
+
throw new Error('the tree is null or empty');
|
|
1035
|
+
}
|
|
1036
|
+
let byCutoff = opts.cutoff !== undefined;
|
|
1037
|
+
if (byCutoff === (opts.target !== undefined)) {
|
|
1038
|
+
throw new Error('give either a cutoff or a target');
|
|
1039
|
+
}
|
|
1040
|
+
if (byCutoff && (typeof opts.cutoff !== 'number' || !isFinite(opts.cutoff) || opts.cutoff < 0)) {
|
|
1041
|
+
throw new Error('cutoff must be a finite, non-negative number');
|
|
1042
|
+
}
|
|
1043
|
+
if (!byCutoff && !(Number.isInteger(opts.target) && opts.target >= 1)) {
|
|
1044
|
+
throw new Error('target number of representatives must be at least 1');
|
|
1045
|
+
}
|
|
1046
|
+
let pick = opts.pick === undefined ? forester.REPRESENTATIVE_MEDOID : opts.pick;
|
|
1047
|
+
if (pick !== forester.REPRESENTATIVE_MEDOID && pick !== forester.REPRESENTATIVE_LONGEST_BRANCH) {
|
|
1048
|
+
throw new Error('unknown representative pick: ' + pick);
|
|
1049
|
+
}
|
|
1050
|
+
let protectedTips = new Set(opts.protectedTips || []);
|
|
1051
|
+
|
|
1052
|
+
let pre = repPreorder(root);
|
|
1053
|
+
let order = new Map();
|
|
1054
|
+
pre.forEach(function (n, i) {
|
|
1055
|
+
order.set(n, i);
|
|
1056
|
+
});
|
|
1057
|
+
let tips = pre.filter(repIsTip);
|
|
1058
|
+
let topological = !forester.hasUsableBranchLengths(phy);
|
|
1059
|
+
let diameter = repDiameters(pre, topological);
|
|
1060
|
+
let groupRoots;
|
|
1061
|
+
let effectiveCutoff;
|
|
1062
|
+
if (byCutoff) {
|
|
1063
|
+
groupRoots = repGroupRoots(root, diameter, opts.cutoff);
|
|
1064
|
+
effectiveCutoff = opts.cutoff;
|
|
1065
|
+
} else if (opts.target >= tips.length) {
|
|
1066
|
+
groupRoots = tips;
|
|
1067
|
+
effectiveCutoff = 0;
|
|
1068
|
+
} else if (opts.target <= 1) {
|
|
1069
|
+
groupRoots = [root];
|
|
1070
|
+
effectiveCutoff = diameter.get(root);
|
|
1071
|
+
} else {
|
|
1072
|
+
let cutoff = repCutoffForTarget(pre, root, diameter, opts.target, tips.length);
|
|
1073
|
+
groupRoots = repGroupRoots(root, diameter, cutoff);
|
|
1074
|
+
effectiveCutoff = Math.max(0, cutoff);
|
|
1075
|
+
}
|
|
1076
|
+
|
|
1077
|
+
let protectedKeptCount = 0;
|
|
1078
|
+
let groups = groupRoots.map(function (clade) {
|
|
1079
|
+
let sub = repPreorder(clade);
|
|
1080
|
+
let members = sub.filter(repIsTip);
|
|
1081
|
+
let kept = members.filter(function (m) {
|
|
1082
|
+
return protectedTips.has(m);
|
|
1083
|
+
});
|
|
1084
|
+
if (kept.length > 0) {
|
|
1085
|
+
protectedKeptCount += kept.length;
|
|
1086
|
+
} else if (members.length === 1) {
|
|
1087
|
+
kept = [members[0]];
|
|
1088
|
+
} else if (pick === forester.REPRESENTATIVE_LONGEST_BRANCH) {
|
|
1089
|
+
kept = [repLongestBranch(members, topological)];
|
|
1090
|
+
} else {
|
|
1091
|
+
kept = [repMedoid(sub, members, topological)];
|
|
1092
|
+
}
|
|
1093
|
+
return {clade: clade, members: members, kept: kept};
|
|
1094
|
+
});
|
|
1095
|
+
groups.sort(function (a, b) {
|
|
1096
|
+
return order.get(a.kept[0]) - order.get(b.kept[0]);
|
|
1097
|
+
});
|
|
1098
|
+
let keptTips = [];
|
|
1099
|
+
groups.forEach(function (g) {
|
|
1100
|
+
g.kept.forEach(function (k) {
|
|
1101
|
+
keptTips.push(k);
|
|
1102
|
+
});
|
|
1103
|
+
});
|
|
1104
|
+
let result = {
|
|
1105
|
+
groups: groups,
|
|
1106
|
+
keptTips: keptTips,
|
|
1107
|
+
keptCount: keptTips.length,
|
|
1108
|
+
protectedKeptCount: protectedKeptCount,
|
|
1109
|
+
tipCount: tips.length,
|
|
1110
|
+
effectiveCutoff: effectiveCutoff,
|
|
1111
|
+
topological: topological,
|
|
1112
|
+
pick: pick,
|
|
1113
|
+
requestedTarget: byCutoff ? -1 : opts.target
|
|
1114
|
+
};
|
|
1115
|
+
result.summary = representativeSummary(result);
|
|
1116
|
+
return result;
|
|
1117
|
+
};
|
|
1118
|
+
|
|
1119
|
+
// Java's Double.toString, which the desktop's texts print numbers with:
|
|
1120
|
+
// the shortest digits that read back as the number (as JS's), but always
|
|
1121
|
+
// with a fraction ("1.0"), and in E notation outside [0.001, 10^7).
|
|
1122
|
+
function javaDoubleText(d) {
|
|
1123
|
+
if (d === 0) {
|
|
1124
|
+
return (1 / d < 0) ? '-0.0' : '0.0';
|
|
1125
|
+
}
|
|
1126
|
+
if (!isFinite(d)) {
|
|
1127
|
+
return String(d);
|
|
1128
|
+
}
|
|
1129
|
+
let a = Math.abs(d);
|
|
1130
|
+
if (a >= 1e-3 && a < 1e7) {
|
|
1131
|
+
let s = String(d);
|
|
1132
|
+
return s.indexOf('.') < 0 ? s + '.0' : s;
|
|
1133
|
+
}
|
|
1134
|
+
let e = d.toExponential();
|
|
1135
|
+
let i = e.indexOf('e');
|
|
1136
|
+
let mantissa = e.slice(0, i);
|
|
1137
|
+
return (mantissa.indexOf('.') < 0 ? mantissa + '.0' : mantissa) + 'E' + e.slice(i + 1).replace('+', '');
|
|
1138
|
+
}
|
|
1139
|
+
|
|
1140
|
+
// a distance to five decimals, a whole number without its fraction
|
|
1141
|
+
function representativeDistanceText(d) {
|
|
1142
|
+
let r = Math.round(d * 1e5) / 1e5;
|
|
1143
|
+
return (r === Math.round(r)) ? String(r) : javaDoubleText(r);
|
|
1144
|
+
}
|
|
1145
|
+
|
|
1146
|
+
function representativeSummary(result) {
|
|
1147
|
+
let groups = result.groups.length;
|
|
1148
|
+
let s = 'Grouped ' + result.tipCount + (result.tipCount === 1 ? ' tip' : ' tips')
|
|
1149
|
+
+ ' into ' + groups + (groups === 1 ? ' group.' : ' groups.');
|
|
1150
|
+
if (result.requestedTarget > 0 && groups !== result.requestedTarget) {
|
|
1151
|
+
s += ' (requested ' + result.requestedTarget + ')';
|
|
1152
|
+
}
|
|
1153
|
+
s += '\nEach group\'s tips are within a distance of ' + representativeDistanceText(result.effectiveCutoff)
|
|
1154
|
+
+ ' of each other';
|
|
1155
|
+
if (result.topological) {
|
|
1156
|
+
s += ' (topological distance — the tree has no branch lengths)';
|
|
1157
|
+
}
|
|
1158
|
+
s += '.';
|
|
1159
|
+
if (result.protectedKeptCount > 0) {
|
|
1160
|
+
s += '\nKeeping ' + result.keptCount + (result.keptCount === 1 ? ' tip, including ' : ' tips, including ')
|
|
1161
|
+
+ result.protectedKeptCount
|
|
1162
|
+
+ (result.protectedKeptCount === 1 ? ' selected tip protected from removal.'
|
|
1163
|
+
: ' selected tips protected from removal.');
|
|
1164
|
+
}
|
|
1165
|
+
s += '\nRepresentative per group: '
|
|
1166
|
+
+ (result.pick === forester.REPRESENTATIVE_LONGEST_BRANCH ? 'most divergent (longest branch)'
|
|
1167
|
+
: 'most central (medoid)') + '.';
|
|
1168
|
+
return s;
|
|
1169
|
+
}
|
|
1170
|
+
|
|
1171
|
+
// a copy of a node's own data (never its children or parent link)
|
|
1172
|
+
function repCopyData(v) {
|
|
1173
|
+
if (Array.isArray(v)) {
|
|
1174
|
+
return v.map(repCopyData);
|
|
1175
|
+
}
|
|
1176
|
+
if (v !== null && typeof v === 'object') {
|
|
1177
|
+
let o = {};
|
|
1178
|
+
Object.keys(v).forEach(function (k) {
|
|
1179
|
+
if (k !== 'parent' && k !== 'children') {
|
|
1180
|
+
o[k] = repCopyData(v[k]);
|
|
1181
|
+
}
|
|
1182
|
+
});
|
|
1183
|
+
return o;
|
|
1184
|
+
}
|
|
1185
|
+
return v;
|
|
1186
|
+
}
|
|
1187
|
+
|
|
1188
|
+
/**
|
|
1189
|
+
* A copy of the tree holding only the given tips, pruned as the desktop
|
|
1190
|
+
* prunes (Phylogeny.deleteSubtree): a node left with one child is
|
|
1191
|
+
* replaced by that child, whose branch gains the node's length (a missing
|
|
1192
|
+
* or negative length adds nothing; two of them leave the length missing);
|
|
1193
|
+
* a root left with one child is replaced by it. The new root keeps the
|
|
1194
|
+
* original root's own branch length (normally none), where the desktop's
|
|
1195
|
+
* depends on the order it deletes in. The tree is not changed.
|
|
1196
|
+
*
|
|
1197
|
+
* @param phy the tree
|
|
1198
|
+
* @param keep the tips to keep (nodes of phy), at least one
|
|
1199
|
+
* @returns the copy, with every node's data copied
|
|
1200
|
+
*/
|
|
1201
|
+
forester.copyTreeKeepingTips = function (phy, keep) {
|
|
1202
|
+
let keepSet = new Set(keep);
|
|
1203
|
+
let top = phy.children && phy.children.length === 1 && !phy.parent ? phy : {children: [phy]};
|
|
1204
|
+
let copies = new Map();
|
|
1205
|
+
let copyTop = repCopyData(top === phy ? phy : {});
|
|
1206
|
+
copies.set(top, copyTop);
|
|
1207
|
+
let stack = [top];
|
|
1208
|
+
while (stack.length > 0) {
|
|
1209
|
+
let n = stack.pop();
|
|
1210
|
+
if (n.children) {
|
|
1211
|
+
copies.get(n).children = n.children.map(function (child) {
|
|
1212
|
+
let cc = repCopyData(child);
|
|
1213
|
+
copies.set(child, cc);
|
|
1214
|
+
stack.push(child);
|
|
1215
|
+
return cc;
|
|
1216
|
+
});
|
|
1217
|
+
}
|
|
1218
|
+
}
|
|
1219
|
+
let tips = repPreorder(top).filter(function (n) {
|
|
1220
|
+
return n !== top && repIsTip(n);
|
|
1221
|
+
});
|
|
1222
|
+
let dropped = tips.filter(function (n) {
|
|
1223
|
+
return !keepSet.has(n);
|
|
1224
|
+
});
|
|
1225
|
+
if (dropped.length === tips.length) {
|
|
1226
|
+
throw new Error('at least one tip must be kept');
|
|
1227
|
+
}
|
|
1228
|
+
let parentOf = new Map();
|
|
1229
|
+
stack = [copyTop];
|
|
1230
|
+
while (stack.length > 0) {
|
|
1231
|
+
let n = stack.pop();
|
|
1232
|
+
(n.children || []).forEach(function (child) {
|
|
1233
|
+
parentOf.set(child, n);
|
|
1234
|
+
stack.push(child);
|
|
1235
|
+
});
|
|
1236
|
+
}
|
|
1237
|
+
let add = function (a, b) {
|
|
1238
|
+
let okA = typeof a === 'number' && a >= 0;
|
|
1239
|
+
let okB = typeof b === 'number' && b >= 0;
|
|
1240
|
+
return (okA && okB) ? a + b : (okA ? a : (okB ? b : undefined));
|
|
1241
|
+
};
|
|
1242
|
+
dropped.forEach(function (tip) {
|
|
1243
|
+
let t = copies.get(tip);
|
|
1244
|
+
let p = parentOf.get(t);
|
|
1245
|
+
let i = p.children.indexOf(t);
|
|
1246
|
+
if (parentOf.get(p) === copyTop) {
|
|
1247
|
+
if (p.children.length === 2) {
|
|
1248
|
+
let other = p.children[1 - i];
|
|
1249
|
+
copyTop.children[0] = other;
|
|
1250
|
+
parentOf.set(other, copyTop);
|
|
1251
|
+
} else {
|
|
1252
|
+
p.children.splice(i, 1);
|
|
1253
|
+
}
|
|
1254
|
+
} else {
|
|
1255
|
+
let pp = parentOf.get(p);
|
|
1256
|
+
if (p.children.length === 2) {
|
|
1257
|
+
let other = p.children[1 - i];
|
|
1258
|
+
let length = add(p.branch_length, other.branch_length);
|
|
1259
|
+
if (length === undefined) {
|
|
1260
|
+
delete other.branch_length;
|
|
1261
|
+
} else {
|
|
1262
|
+
other.branch_length = length;
|
|
1263
|
+
}
|
|
1264
|
+
pp.children[pp.children.indexOf(p)] = other;
|
|
1265
|
+
parentOf.set(other, pp);
|
|
1266
|
+
} else {
|
|
1267
|
+
p.children.splice(i, 1);
|
|
1268
|
+
}
|
|
1269
|
+
}
|
|
1270
|
+
if (p.children.length === 0) {
|
|
1271
|
+
delete p.children;
|
|
1272
|
+
}
|
|
1273
|
+
});
|
|
1274
|
+
// The root keeps the original root's own branch length, normally
|
|
1275
|
+
// none. What the desktop's pruning leaves there depends on the order
|
|
1276
|
+
// it deletes tips in -- that is, on node ids -- so the same selection
|
|
1277
|
+
// gave 0.05 in one session and 0.25 in another (Christian, 2026-09-15).
|
|
1278
|
+
let rootLength = top.children[0].branch_length;
|
|
1279
|
+
if (typeof rootLength === 'number') {
|
|
1280
|
+
copyTop.children[0].branch_length = rootLength;
|
|
1281
|
+
} else {
|
|
1282
|
+
delete copyTop.children[0].branch_length;
|
|
1283
|
+
}
|
|
1284
|
+
return copyTop;
|
|
1285
|
+
};
|
|
1286
|
+
|
|
1287
|
+
/**
|
|
1288
|
+
* Strips a file-type suffix -- a dot and 1 to 5 other characters, such as
|
|
1289
|
+
* .xml or .nexus -- from a tree name, as the desktop does.
|
|
1290
|
+
*
|
|
1291
|
+
* @param name
|
|
1292
|
+
* @returns {string|null}
|
|
1293
|
+
*/
|
|
1294
|
+
forester.stripShortExtension = function (name) {
|
|
1295
|
+
return (name === null || name === undefined) ? null : String(name).replace(/\.[^.]{1,5}$/, '');
|
|
1296
|
+
};
|
|
1297
|
+
|
|
1298
|
+
/**
|
|
1299
|
+
* The desktop's name for a tree of representative tips: the parent's name
|
|
1300
|
+
* without its file suffix, then the count -- mammals_233reps, _1rep --
|
|
1301
|
+
* or "tree" for an unnamed parent.
|
|
1302
|
+
*
|
|
1303
|
+
* @param parentName
|
|
1304
|
+
* @param count
|
|
1305
|
+
* @returns {string}
|
|
1306
|
+
*/
|
|
1307
|
+
forester.representativeTreeName = function (parentName, count) {
|
|
1308
|
+
let stripped = forester.stripShortExtension(parentName);
|
|
1309
|
+
return (stripped ? stripped : 'tree') + '_' + count + (count === 1 ? 'rep' : 'reps');
|
|
1310
|
+
};
|
|
1311
|
+
|
|
1312
|
+
/**
|
|
1313
|
+
* The desktop's provenance sentence for a tree of representative tips,
|
|
1314
|
+
* which it adds to the tree's description.
|
|
1315
|
+
*
|
|
1316
|
+
* @param byCutoff true for a cutoff, false for a target
|
|
1317
|
+
* @param cutoff
|
|
1318
|
+
* @param target
|
|
1319
|
+
* @param pick forester.REPRESENTATIVE_MEDOID or _LONGEST_BRANCH
|
|
1320
|
+
* @param count tips kept
|
|
1321
|
+
* @param parentName
|
|
1322
|
+
* @param parentTipCount
|
|
1323
|
+
* @returns {string}
|
|
1324
|
+
*/
|
|
1325
|
+
forester.representativeTreeDescription = function (byCutoff, cutoff, target, pick, count, parentName, parentTipCount) {
|
|
1326
|
+
let pickText = pick === forester.REPRESENTATIVE_LONGEST_BRANCH ? 'longest-branch' : 'medoid';
|
|
1327
|
+
let algorithm = byCutoff
|
|
1328
|
+
? 'distance-cutoff (maximum distance ' + javaDoubleText(cutoff) + ', ' + pickText + ' representative)'
|
|
1329
|
+
: 'target-count (target ' + target + ', ' + pickText + ' representative)';
|
|
1330
|
+
return 'Used the ' + algorithm + ' algorithm to select ' + count + ' representative '
|
|
1331
|
+
+ (count === 1 ? 'tip' : 'tips') + ' from tree named "' + (parentName ? parentName : 'tree') + '" with '
|
|
1332
|
+
+ parentTipCount + (parentTipCount === 1 ? ' tip.' : ' tips.');
|
|
1333
|
+
};
|
|
1334
|
+
|
|
784
1335
|
/**
|
|
785
1336
|
* Whether a node carries data about the node itself, the kind a
|
|
786
1337
|
* re-rooting can take the meaning away from: a name, taxonomy, sequence
|
|
@@ -2526,6 +3077,57 @@
|
|
|
2526
3077
|
// starting with '&' (a [95] confidence) is left untouched, as is any
|
|
2527
3078
|
// bracket inside a quoted label. Quotes and nested brackets inside an
|
|
2528
3079
|
// annotation are honoured when finding its end.
|
|
3080
|
+
//
|
|
3081
|
+
// A QUOTE CHARACTER INSIDE A BLOB IS DATA unless it opens a quoted VALUE.
|
|
3082
|
+
// Auspice writes values bare -- country=Côte d'Ivoire -- and treating that
|
|
3083
|
+
// apostrophe as the start of a quoted string was a real bug, in both of
|
|
3084
|
+
// its forms: one such tip and the quote never closed, so the file was
|
|
3085
|
+
// refused over its "unbalanced parentheses"; two and the apostrophes
|
|
3086
|
+
// paired up ACROSS the tips, no error at all, the second tip gone and the
|
|
3087
|
+
// first one's country reading "Côte d'Ivoire],B:1[&country=Côte d'Ivoire".
|
|
3088
|
+
// (Real file: nextstrain_chikv_global_timetree.nexus, 16 apostrophes, all
|
|
3089
|
+
// of them that one country.) Matches the desktop's scanner rule.
|
|
3090
|
+
//
|
|
3091
|
+
// So a quote opens a run only where a value can START -- straight after
|
|
3092
|
+
// '=', or after '{', '[' or ',' inside a set -- and only if it is closed
|
|
3093
|
+
// by the same character standing where a value can END: before ',', '}',
|
|
3094
|
+
// ']' or the end. Anything else is a character like any other.
|
|
3095
|
+
function opensBlobQuote(s, p, from) {
|
|
3096
|
+
let m = p - 1;
|
|
3097
|
+
while (m >= from && /\s/.test(s.charAt(m))) {
|
|
3098
|
+
--m;
|
|
3099
|
+
}
|
|
3100
|
+
return m >= from && '={[,'.indexOf(s.charAt(m)) >= 0;
|
|
3101
|
+
}
|
|
3102
|
+
|
|
3103
|
+
// The index of the quote closing the run opened at p, or -1. `bounded` is
|
|
3104
|
+
// for the extraction pass, which does not yet know where the blob ends:
|
|
3105
|
+
// there the search gives up at a ']' that is followed by Newick structure,
|
|
3106
|
+
// so a bare value that merely BEGINS with an apostrophe ('s-Hertogenbosch)
|
|
3107
|
+
// cannot reach into the next node's blob for its partner.
|
|
3108
|
+
function blobQuoteClose(s, p, bounded) {
|
|
3109
|
+
let q = s.charAt(p);
|
|
3110
|
+
for (let k = p + 1; k < s.length; ++k) {
|
|
3111
|
+
let c = s.charAt(k);
|
|
3112
|
+
if (c !== q && !(bounded && c === ']')) {
|
|
3113
|
+
continue;
|
|
3114
|
+
}
|
|
3115
|
+
let m = k + 1;
|
|
3116
|
+
while (m < s.length && /\s/.test(s.charAt(m))) {
|
|
3117
|
+
++m;
|
|
3118
|
+
}
|
|
3119
|
+
let next = m < s.length ? s.charAt(m) : '';
|
|
3120
|
+
if (c === q) {
|
|
3121
|
+
if (next === '' || ',}]'.indexOf(next) >= 0) {
|
|
3122
|
+
return k;
|
|
3123
|
+
}
|
|
3124
|
+
} else if (next === '' || ',):;(['.indexOf(next) >= 0) {
|
|
3125
|
+
return -1;
|
|
3126
|
+
}
|
|
3127
|
+
}
|
|
3128
|
+
return -1;
|
|
3129
|
+
}
|
|
3130
|
+
|
|
2529
3131
|
function extractBracketAnnotations(str) {
|
|
2530
3132
|
if (str.indexOf('[') < 0) {
|
|
2531
3133
|
return {text: str, blobs: []};
|
|
@@ -2551,16 +3153,16 @@
|
|
|
2551
3153
|
} else if (c === '[') {
|
|
2552
3154
|
let j = i + 1;
|
|
2553
3155
|
let depth = 1;
|
|
2554
|
-
let q = null;
|
|
2555
3156
|
while (j < str.length && depth > 0) {
|
|
2556
3157
|
let cj = str.charAt(j);
|
|
2557
|
-
if (
|
|
2558
|
-
|
|
2559
|
-
|
|
3158
|
+
if ((cj === "'" || cj === '"') && opensBlobQuote(str, j, i + 1)) {
|
|
3159
|
+
let close = blobQuoteClose(str, j, true);
|
|
3160
|
+
if (close > -1) {
|
|
3161
|
+
j = close + 1; // a quoted value: its brackets are data
|
|
3162
|
+
continue;
|
|
2560
3163
|
}
|
|
2561
|
-
}
|
|
2562
|
-
|
|
2563
|
-
} else if (cj === '[') {
|
|
3164
|
+
}
|
|
3165
|
+
if (cj === '[') {
|
|
2564
3166
|
++depth;
|
|
2565
3167
|
} else if (cj === ']') {
|
|
2566
3168
|
--depth;
|
|
@@ -2610,21 +3212,37 @@
|
|
|
2610
3212
|
}
|
|
2611
3213
|
|
|
2612
3214
|
// Split on TOP-LEVEL commas only: a comma inside {...}/[...] sets or
|
|
2613
|
-
// inside
|
|
2614
|
-
// stay one token).
|
|
2615
|
-
|
|
3215
|
+
// inside a quoted VALUE is data, not a separator (height_95%_HPD={1.4,1.5}
|
|
3216
|
+
// must stay one token). A quote that does not open a value is itself data
|
|
3217
|
+
// (opensBlobQuote): country=Côte d'Ivoire,region=Africa is two fields.
|
|
3218
|
+
//
|
|
3219
|
+
// TWO quoting rules live here, because two grammars do. In a Nexus
|
|
3220
|
+
// TRANSLATE table the things between the commas are LABELS, and a quote
|
|
3221
|
+
// opens one wherever it stands (1 'Korea, Republic of'). In a [&...] blob
|
|
3222
|
+
// they are key=value fields, and a quote opens only a VALUE. Giving the
|
|
3223
|
+
// table the blob's rule split 'Korea, Republic of' in two, which a test
|
|
3224
|
+
// caught the moment it was tried; `blob` says which grammar this is.
|
|
3225
|
+
function splitTopLevelCommas(s, blob) {
|
|
2616
3226
|
let out = [];
|
|
2617
3227
|
let depth = 0;
|
|
2618
3228
|
let q = null;
|
|
2619
3229
|
let cur = '';
|
|
2620
3230
|
for (let i = 0; i < s.length; ++i) {
|
|
2621
3231
|
let c = s.charAt(i);
|
|
3232
|
+
if (blob && (c === "'" || c === '"') && opensBlobQuote(s, i, 0)) {
|
|
3233
|
+
let close = blobQuoteClose(s, i, false);
|
|
3234
|
+
if (close > -1) {
|
|
3235
|
+
cur += s.substring(i, close + 1); // a quoted value, its commas data
|
|
3236
|
+
i = close;
|
|
3237
|
+
continue;
|
|
3238
|
+
}
|
|
3239
|
+
}
|
|
2622
3240
|
if (q) {
|
|
2623
3241
|
if (c === q) {
|
|
2624
3242
|
q = null;
|
|
2625
3243
|
}
|
|
2626
3244
|
cur += c;
|
|
2627
|
-
} else if (c === "'" || c === '"') {
|
|
3245
|
+
} else if (!blob && (c === "'" || c === '"')) {
|
|
2628
3246
|
q = c;
|
|
2629
3247
|
cur += c;
|
|
2630
3248
|
} else if (c === '{' || c === '[') {
|
|
@@ -2648,9 +3266,44 @@
|
|
|
2648
3266
|
return out;
|
|
2649
3267
|
}
|
|
2650
3268
|
|
|
3269
|
+
// A NUMBER is a plain decimal with an optional exponent, and nothing else
|
|
3270
|
+
// -- the desktop's grammar (Christian, 2026-09-16). The test used to be
|
|
3271
|
+
// "parseFloat is finite AND Number is finite", and the two read different
|
|
3272
|
+
// languages: Number() understands 0x1A, 0b101 and 0o17, parseFloat() stops
|
|
3273
|
+
// at the letter and answers 0. So a trait that merely LOOKED like a hex
|
|
3274
|
+
// literal was typed numeric, and a height written that way dated its node
|
|
3275
|
+
// at 0 -- a wrong answer where a refusal was due. One pattern now decides,
|
|
3276
|
+
// for a value read as a number and for a property's datatype alike.
|
|
3277
|
+
const PLAIN_DECIMAL_RE = /^[+-]?(\d+\.?\d*|\.\d+)([eE][+-]?\d+)?$/;
|
|
3278
|
+
|
|
2651
3279
|
function parseBeastNumber(v) {
|
|
2652
|
-
let
|
|
2653
|
-
|
|
3280
|
+
let t = String(v).trim();
|
|
3281
|
+
if (!PLAIN_DECIMAL_RE.test(t)) {
|
|
3282
|
+
return null;
|
|
3283
|
+
}
|
|
3284
|
+
let d = parseFloat(t);
|
|
3285
|
+
return isFinite(d) ? d : null; // 1e400 is well-formed and still not a number we can use
|
|
3286
|
+
}
|
|
3287
|
+
|
|
3288
|
+
// FigTree's !color value: #rrggbb, or Java's SIGNED Color.getRGB() int,
|
|
3289
|
+
// which FigTree writes whenever the colour came from AWT (real files: every
|
|
3290
|
+
// tag in test_trees/influenza.tree is #-8381639, never hex). Each '>>>'
|
|
3291
|
+
// coerces to an unsigned 32-bit value first, so the low three bytes come
|
|
3292
|
+
// out as RGB regardless of sign; the alpha byte is discarded. Null when it
|
|
3293
|
+
// is neither.
|
|
3294
|
+
function parseFigTreeColor(value) {
|
|
3295
|
+
if (/^#[0-9a-f]{6}$/i.test(value)) {
|
|
3296
|
+
return {
|
|
3297
|
+
red: parseInt(value.substring(1, 3), 16),
|
|
3298
|
+
green: parseInt(value.substring(3, 5), 16),
|
|
3299
|
+
blue: parseInt(value.substring(5, 7), 16)
|
|
3300
|
+
};
|
|
3301
|
+
}
|
|
3302
|
+
if (/^#-?[0-9]{1,10}$/.test(value)) {
|
|
3303
|
+
let argb = Number(value.substring(1));
|
|
3304
|
+
return {red: (argb >>> 16) & 0xff, green: (argb >>> 8) & 0xff, blue: argb & 0xff};
|
|
3305
|
+
}
|
|
3306
|
+
return null;
|
|
2654
3307
|
}
|
|
2655
3308
|
|
|
2656
3309
|
// A two-value BEAST set {lo,hi} (or [lo,hi]) as [lo,hi] numbers, or null.
|
|
@@ -2659,7 +3312,7 @@
|
|
|
2659
3312
|
if (s.length < 3 || (s.charAt(0) !== '{' && s.charAt(0) !== '[')) {
|
|
2660
3313
|
return null;
|
|
2661
3314
|
}
|
|
2662
|
-
let parts = splitTopLevelCommas(s.substring(1, s.length - 1));
|
|
3315
|
+
let parts = splitTopLevelCommas(s.substring(1, s.length - 1), true);
|
|
2663
3316
|
if (parts.length !== 2) {
|
|
2664
3317
|
return null;
|
|
2665
3318
|
}
|
|
@@ -2721,6 +3374,15 @@
|
|
|
2721
3374
|
// - node age height/height_mean/height_median + height_95%_HPD (or
|
|
2722
3375
|
// height_range) + date -> node.date value/min/max/desc (the node-age
|
|
2723
3376
|
// HPD bars draw the interval);
|
|
3377
|
+
// - Auspice's "download Nexus" vocabulary lands exactly where
|
|
3378
|
+
// parseAuspiceJson puts the same dataset, so one Nextstrain build opens
|
|
3379
|
+
// the same way whichever format it was saved in: num_date -> the date
|
|
3380
|
+
// VALUE with unit "year", plus a nextstrain:num_date property;
|
|
3381
|
+
// num_date_CI={lo,hi} -> that date's minimum/maximum, on a tip too
|
|
3382
|
+
// (there it is the sampling-date uncertainty); div -> a
|
|
3383
|
+
// nextstrain:div property. A num_date outranks every height* (it is a
|
|
3384
|
+
// calendar year, a height is an age before present), it alone carries
|
|
3385
|
+
// the unit, and it never borrows the height's HPD as its interval;
|
|
2724
3386
|
// - FigTree !color=#rrggbb -> the branch color;
|
|
2725
3387
|
// - every other field (rate, length_*, traits, location, ...) -> a
|
|
2726
3388
|
// beast:<key> node property (numeric -> xsd:decimal, so Color-by
|
|
@@ -2734,9 +3396,14 @@
|
|
|
2734
3396
|
let hpd = null;
|
|
2735
3397
|
let range = null;
|
|
2736
3398
|
let dateDesc = null;
|
|
3399
|
+
let hpdText = null;
|
|
3400
|
+
let rangeText = null;
|
|
3401
|
+
let numDate = null;
|
|
3402
|
+
let numDateCi = null;
|
|
3403
|
+
let numDateCiKey = null;
|
|
2737
3404
|
let prob = null;
|
|
2738
3405
|
let probSd = null;
|
|
2739
|
-
splitTopLevelCommas(blob).forEach(function (token) {
|
|
3406
|
+
splitTopLevelCommas(blob, true).forEach(function (token) {
|
|
2740
3407
|
let eq = token.indexOf('=');
|
|
2741
3408
|
if (eq <= 0) {
|
|
2742
3409
|
return;
|
|
@@ -2761,13 +3428,8 @@
|
|
|
2761
3428
|
if (b !== null) {
|
|
2762
3429
|
pushConfidence(node, b, 'bootstrap');
|
|
2763
3430
|
}
|
|
2764
|
-
} else if ((kl === '!color' || kl === '!colour')
|
|
2765
|
-
|
|
2766
|
-
node.color = {
|
|
2767
|
-
red: parseInt(value.substring(1, 3), 16),
|
|
2768
|
-
green: parseInt(value.substring(3, 5), 16),
|
|
2769
|
-
blue: parseInt(value.substring(5, 7), 16)
|
|
2770
|
-
};
|
|
3431
|
+
} else if ((kl === '!color' || kl === '!colour') && parseFigTreeColor(value) !== null) {
|
|
3432
|
+
node.color = parseFigTreeColor(value); // in the tree string it is the BRANCH colour
|
|
2771
3433
|
} else if (kl === 'height_median') {
|
|
2772
3434
|
heightMedian = value;
|
|
2773
3435
|
} else if (kl === 'height_mean') {
|
|
@@ -2776,10 +3438,37 @@
|
|
|
2776
3438
|
height = value;
|
|
2777
3439
|
} else if (kl === 'height_95%_hpd') {
|
|
2778
3440
|
hpd = parseBeastInterval(value);
|
|
3441
|
+
hpdText = value;
|
|
2779
3442
|
} else if (kl === 'height_range') {
|
|
2780
3443
|
range = parseBeastInterval(value);
|
|
3444
|
+
rangeText = value;
|
|
2781
3445
|
} else if (kl === 'date') {
|
|
2782
3446
|
dateDesc = value;
|
|
3447
|
+
} else if (kl === 'num_date') {
|
|
3448
|
+
numDate = value;
|
|
3449
|
+
} else if (kl === 'num_date_ci') {
|
|
3450
|
+
numDateCi = value;
|
|
3451
|
+
numDateCiKey = beastRefKey(key);
|
|
3452
|
+
} else if (kl === 'div' && parseBeastNumber(value) !== null) {
|
|
3453
|
+
addNodeProperty(node, NEXTSTRAIN_PREFIX + 'div', value);
|
|
3454
|
+
} else if (key.charAt(0) === '!') {
|
|
3455
|
+
// A key that starts with '!' is one of FigTree's display
|
|
3456
|
+
// DIRECTIVES -- !color, !rotate, !collapse, !hilight, !name --
|
|
3457
|
+
// never a measurement, so it is never typed numeric. It
|
|
3458
|
+
// matters for the forms we refuse as a colour: !color=-8381639
|
|
3459
|
+
// (no '#') used to land as beast:_color typed xsd:decimal, and
|
|
3460
|
+
// Color-by offered FigTree's paint as a gradient. It is still
|
|
3461
|
+
// KEPT -- nothing in this reader throws data away -- under the
|
|
3462
|
+
// same ref, as text. A user's own trait called "color" has no
|
|
3463
|
+
// '!' and stays an ordinary trait. (The desktop's rule;
|
|
3464
|
+
// Christian, 2026-09-17: "do the same".)
|
|
3465
|
+
addNodeProperty(node, 'beast:' + beastRefKey(key), value, 'xsd:string');
|
|
3466
|
+
} else if (kl === 'mutations' || kl === 'mcc') {
|
|
3467
|
+
// TEXT, whatever it looks like (the desktop forces the same):
|
|
3468
|
+
// a list of mutations that happens to hold one number, or a
|
|
3469
|
+
// clade label that happens to be "3", is not a measurement,
|
|
3470
|
+
// and typing it decimal offers it to Color-by as a gradient
|
|
3471
|
+
addNodeProperty(node, 'beast:' + beastRefKey(key), value, 'xsd:string');
|
|
2783
3472
|
} else {
|
|
2784
3473
|
addNodeProperty(node, 'beast:' + beastRefKey(key), value);
|
|
2785
3474
|
}
|
|
@@ -2787,28 +3476,60 @@
|
|
|
2787
3476
|
if (prob !== null) {
|
|
2788
3477
|
pushConfidence(node, prob, 'posterior probability', probSd);
|
|
2789
3478
|
}
|
|
2790
|
-
//
|
|
2791
|
-
//
|
|
2792
|
-
//
|
|
2793
|
-
let
|
|
2794
|
-
|
|
2795
|
-
|
|
2796
|
-
|
|
2797
|
-
if (dv === null && !interval && dateDesc === null) {
|
|
2798
|
-
return;
|
|
3479
|
+
// A num_date that parses is the node's date, and nothing about a
|
|
3480
|
+
// height may touch it. One that does not parse is just a field: it and
|
|
3481
|
+
// its interval fall back to plain text, as any unknown key does.
|
|
3482
|
+
let year = (numDate !== null) ? parseBeastNumber(numDate) : null;
|
|
3483
|
+
let yearCi = (numDateCi !== null) ? parseBeastInterval(numDateCi) : null;
|
|
3484
|
+
if (numDate !== null) {
|
|
3485
|
+
addNodeProperty(node, (year !== null ? NEXTSTRAIN_PREFIX : 'beast:') + 'num_date', numDate);
|
|
2799
3486
|
}
|
|
2800
|
-
|
|
2801
|
-
|
|
2802
|
-
|
|
3487
|
+
if (numDateCi !== null && (year === null || yearCi === null)) {
|
|
3488
|
+
// an interval with no date to bracket, or not an interval at all
|
|
3489
|
+
addNodeProperty(node, (yearCi !== null ? NEXTSTRAIN_PREFIX : 'beast:') + numDateCiKey, numDateCi);
|
|
2803
3490
|
}
|
|
2804
|
-
|
|
2805
|
-
|
|
2806
|
-
date.
|
|
3491
|
+
let date = {};
|
|
3492
|
+
if (year !== null) {
|
|
3493
|
+
date.value = year;
|
|
3494
|
+
date.unit = 'year';
|
|
3495
|
+
if (yearCi !== null) {
|
|
3496
|
+
date.minimum = yearCi[0];
|
|
3497
|
+
date.maximum = yearCi[1];
|
|
3498
|
+
}
|
|
3499
|
+
// provisional until the whole tree has been read: settleNumDates
|
|
3500
|
+
// decides whether this tree is time-scaled at all
|
|
3501
|
+
node._numDate = {ciKey: numDateCiKey, ciText: yearCi !== null ? numDateCi : null};
|
|
3502
|
+
// the heights it outranked are kept as what they were written as,
|
|
3503
|
+
// rather than dropped: no real file carries both, so nothing here
|
|
3504
|
+
// is lost to a guess
|
|
3505
|
+
[['height_median', heightMedian], ['height_mean', heightMean], ['height', height],
|
|
3506
|
+
['height_95_HPD', hpdText], ['height_range', rangeText]].forEach(function (h) {
|
|
3507
|
+
if (h[1] !== null) {
|
|
3508
|
+
addNodeProperty(node, 'beast:' + h[0], h[1]);
|
|
3509
|
+
}
|
|
3510
|
+
});
|
|
3511
|
+
} else {
|
|
3512
|
+
// age preference: median, then mean, then height -- and each piece
|
|
3513
|
+
// parsed independently, so an unparseable point value never
|
|
3514
|
+
// discards a valid {lo,hi} interval
|
|
3515
|
+
let v = heightMedian !== null ? heightMedian
|
|
3516
|
+
: (heightMean !== null ? heightMean : height);
|
|
3517
|
+
let dv = (v !== null) ? parseBeastNumber(v) : null;
|
|
3518
|
+
let interval = hpd || range;
|
|
3519
|
+
if (dv !== null) {
|
|
3520
|
+
date.value = dv;
|
|
3521
|
+
}
|
|
3522
|
+
if (interval) {
|
|
3523
|
+
date.minimum = interval[0];
|
|
3524
|
+
date.maximum = interval[1];
|
|
3525
|
+
}
|
|
2807
3526
|
}
|
|
2808
3527
|
if (dateDesc !== null) {
|
|
2809
3528
|
date.desc = dateDesc;
|
|
2810
3529
|
}
|
|
2811
|
-
|
|
3530
|
+
if (Object.keys(date).length > 0) {
|
|
3531
|
+
node.date = date;
|
|
3532
|
+
}
|
|
2812
3533
|
}
|
|
2813
3534
|
|
|
2814
3535
|
// The classic NHX tag set, as the desktop maps it: S= taxonomy
|
|
@@ -2816,8 +3537,8 @@
|
|
|
2816
3537
|
// duplication (Y/T) / speciation (N/F) / undecided (?) event, GN=
|
|
2817
3538
|
// sequence name, AC= sequence accession, C= an nh:comment property.
|
|
2818
3539
|
// Unknown tags (and DS= domain structures) are ignored.
|
|
2819
|
-
function applyNhxTags(node,
|
|
2820
|
-
|
|
3540
|
+
function applyNhxTags(node, fields) {
|
|
3541
|
+
fields.forEach(function (tag) {
|
|
2821
3542
|
let t = tag.trim();
|
|
2822
3543
|
if (t.length < 3) {
|
|
2823
3544
|
return;
|
|
@@ -2850,14 +3571,327 @@
|
|
|
2850
3571
|
});
|
|
2851
3572
|
}
|
|
2852
3573
|
|
|
3574
|
+
// Inside a legacy [&&NHX:...] tag the desktop reads by its LABEL rule, and
|
|
3575
|
+
// so do we (Christian, 2026-09-16: "do what desktop does"):
|
|
3576
|
+
// - UNQUOTED whitespace is formatting noise and is squeezed out -- its
|
|
3577
|
+
// Test.testNHXParsingQuotes pins "[\t&\t&\n N\tH\tX:S=mo\tnkey !]" as
|
|
3578
|
+
// S=monkey!, and S=Homo sapiens reads as Homosapiens;
|
|
3579
|
+
// - a QUOTED run, either style, keeps what is inside it -- S="homo sapiens"
|
|
3580
|
+
// is homo sapiens, the one way to put a two-word species into an NHX tag
|
|
3581
|
+
// -- with a run of whitespace collapsed to one space, and may carry the
|
|
3582
|
+
// ':' that would otherwise end the tag (S="a:b c":D=Y);
|
|
3583
|
+
// - the quote characters themselves are never part of the value.
|
|
3584
|
+
// My first version squeezed quotes and ALL whitespace out of the blob, from
|
|
3585
|
+
// an inference off that one pinned case; the desktop then MEASURED its own
|
|
3586
|
+
// behaviour and the quoted forms differed (homosapiens here, homo sapiens
|
|
3587
|
+
// there). The opposite of a single-& blob, where quotes and spaces are
|
|
3588
|
+
// data -- so none of this runs until the blob has shown itself to be NHX.
|
|
3589
|
+
function nhxFields(blob) {
|
|
3590
|
+
let out = [];
|
|
3591
|
+
let cur = '';
|
|
3592
|
+
let q = null;
|
|
3593
|
+
let spaced = false;
|
|
3594
|
+
for (let i = 0; i < blob.length; ++i) {
|
|
3595
|
+
let c = blob.charAt(i);
|
|
3596
|
+
if (q) {
|
|
3597
|
+
if (c === q) {
|
|
3598
|
+
q = null;
|
|
3599
|
+
} else if (/\s/.test(c)) {
|
|
3600
|
+
if (!spaced) {
|
|
3601
|
+
cur += ' ';
|
|
3602
|
+
spaced = true;
|
|
3603
|
+
}
|
|
3604
|
+
} else {
|
|
3605
|
+
cur += c;
|
|
3606
|
+
spaced = false;
|
|
3607
|
+
}
|
|
3608
|
+
} else if (c === "'" || c === '"') {
|
|
3609
|
+
q = c;
|
|
3610
|
+
spaced = false;
|
|
3611
|
+
} else if (c === ':') {
|
|
3612
|
+
out.push(cur);
|
|
3613
|
+
cur = '';
|
|
3614
|
+
} else if (!/\s/.test(c)) {
|
|
3615
|
+
cur += c;
|
|
3616
|
+
}
|
|
3617
|
+
}
|
|
3618
|
+
out.push(cur);
|
|
3619
|
+
return out;
|
|
3620
|
+
}
|
|
3621
|
+
|
|
2853
3622
|
function applyExtendedAnnotations(node, blob) {
|
|
2854
|
-
|
|
2855
|
-
|
|
3623
|
+
let fields = nhxFields(blob);
|
|
3624
|
+
if (/^&&NHX$/i.test(fields[0])) {
|
|
3625
|
+
applyNhxTags(node, fields.slice(1));
|
|
2856
3626
|
} else {
|
|
2857
3627
|
applyBeastAnnotations(node, blob.replace(/^&/, ''));
|
|
2858
3628
|
}
|
|
2859
3629
|
}
|
|
2860
3630
|
|
|
3631
|
+
// ---------------------------------------------------------------
|
|
3632
|
+
// A bare numeric date= as a node date VALUE (TreeTime)
|
|
3633
|
+
// ---------------------------------------------------------------
|
|
3634
|
+
//
|
|
3635
|
+
// applyBeastAnnotations files date= as a date DESC and nothing else, which
|
|
3636
|
+
// is what the desktop's BeastAnnotationParser does and stays that way --
|
|
3637
|
+
// in BEAST output the age lives in height*, and date= is a decoration.
|
|
3638
|
+
//
|
|
3639
|
+
// TreeTime has no height at all: it writes "[&mutations=...,date=2003.84]"
|
|
3640
|
+
// and the decimal year IS the node's position in time. Left as a desc the
|
|
3641
|
+
// tree carries no date value, so isTimeTree is false and the calendar axis
|
|
3642
|
+
// never appears -- a time tree that does not look like one.
|
|
3643
|
+
//
|
|
3644
|
+
// The catch is that TreeTime writes that SAME comment on both trees it
|
|
3645
|
+
// emits: timetree.nexus, whose branch lengths are years, and
|
|
3646
|
+
// divergence_tree.nexus, whose branch lengths are substitutions. No single
|
|
3647
|
+
// annotation says which file it came from, and promoting blindly would put
|
|
3648
|
+
// a calendar axis (which maps one branch-length unit to one year) on a
|
|
3649
|
+
// divergence tree and silently disable re-rooting for it.
|
|
3650
|
+
//
|
|
3651
|
+
// So the TREE is asked rather than the annotation sniffed: a numeric date
|
|
3652
|
+
// becomes a value only where the parent-to-child date differences actually
|
|
3653
|
+
// reproduce the branch lengths. That needs no format detection, it is
|
|
3654
|
+
// self-validating on any input, and it separates TreeTime's two files
|
|
3655
|
+
// exactly. The desc is left in place either way, so nothing is lost.
|
|
3656
|
+
const NUMERIC_DATE_ABS_TOL = 0.02; // date= is written to 2 decimals, so a
|
|
3657
|
+
const NUMERIC_DATE_REL_TOL = 0.01; // difference of two carries ~0.01 error
|
|
3658
|
+
|
|
3659
|
+
function numericDateDesc(n) {
|
|
3660
|
+
if (!n.date || n.date.value !== undefined || typeof n.date.desc !== 'string') {
|
|
3661
|
+
return null;
|
|
3662
|
+
}
|
|
3663
|
+
return parseBeastNumber(n.date.desc);
|
|
3664
|
+
}
|
|
3665
|
+
|
|
3666
|
+
// THE PAIR COUNT, shared by the date= promotion below and by
|
|
3667
|
+
// settleNumDates, so the two can never drift: a PAIR is a dated node, its
|
|
3668
|
+
// dated DIRECT parent and a branch length between them, and it AGREES
|
|
3669
|
+
// when the year difference reproduces that length within the tolerances.
|
|
3670
|
+
//
|
|
3671
|
+
// A pair must also be INFORMATIVE: max(|dYear|, |length|) > the absolute
|
|
3672
|
+
// tolerance. One that is not cannot tell years from substitutions at all
|
|
3673
|
+
// -- both numbers sit inside the tolerance, so it "agrees" whatever the
|
|
3674
|
+
// tree is measured in -- and it is left out of BOTH counts. That is not a
|
|
3675
|
+
// nicety. The 0.02 tolerance is larger than a dense tree's substitution
|
|
3676
|
+
// lengths, so on a densely sampled DIVERGENCE tree such pairs pile up as
|
|
3677
|
+
// agreement. Measured by rebuilding real Auspice time-tree exports as
|
|
3678
|
+
// their divergence trees (length = the div difference, same num_dates):
|
|
3679
|
+
// agreeing, every pair informative pairs only
|
|
3680
|
+
// H5N1 (2 y) 3620 of 9205 (39.3%) 1 of 5586
|
|
3681
|
+
// chikungunya 672 of 2645 (25.4%) 0 of 1973
|
|
3682
|
+
// measles 915 of 5388 (17.0%) 0 of 4473
|
|
3683
|
+
// while every time tree stays at 100% either way (H5N1 5586 of 5586).
|
|
3684
|
+
// 39% is under the majority, so nothing visible changed on those files;
|
|
3685
|
+
// it is headroom -- a denser build would have crossed it and been dated
|
|
3686
|
+
// on substitutions. Found by a review on the desktop, reproduced here to
|
|
3687
|
+
// the pair, and a JOINT RULE (Christian, 2026-09-17, both sessions). The
|
|
3688
|
+
// burden of proof is untouched: with no informative pair at all a date=
|
|
3689
|
+
// is still not promoted and a num_date still stands.
|
|
3690
|
+
function countDatePairs(root, yearOf) {
|
|
3691
|
+
let dated = [];
|
|
3692
|
+
let agree = 0;
|
|
3693
|
+
let pairs = 0;
|
|
3694
|
+
let stack = [[root, null]];
|
|
3695
|
+
while (stack.length > 0) {
|
|
3696
|
+
let top = stack.pop();
|
|
3697
|
+
let n = top[0];
|
|
3698
|
+
let year = yearOf(n);
|
|
3699
|
+
if (year !== null) {
|
|
3700
|
+
dated.push([n, year]);
|
|
3701
|
+
let parentYear = top[1];
|
|
3702
|
+
if (parentYear !== null && typeof n.branch_length === 'number'
|
|
3703
|
+
&& isFinite(n.branch_length)) {
|
|
3704
|
+
let dYear = year - parentYear;
|
|
3705
|
+
if (Math.max(Math.abs(dYear), Math.abs(n.branch_length)) > NUMERIC_DATE_ABS_TOL) {
|
|
3706
|
+
++pairs;
|
|
3707
|
+
let tol = NUMERIC_DATE_ABS_TOL
|
|
3708
|
+
+ NUMERIC_DATE_REL_TOL * Math.abs(n.branch_length);
|
|
3709
|
+
if (Math.abs(dYear - n.branch_length) <= tol) {
|
|
3710
|
+
++agree;
|
|
3711
|
+
}
|
|
3712
|
+
}
|
|
3713
|
+
}
|
|
3714
|
+
}
|
|
3715
|
+
if (n.children) {
|
|
3716
|
+
for (let i = 0; i < n.children.length; ++i) {
|
|
3717
|
+
stack.push([n.children[i], year]);
|
|
3718
|
+
}
|
|
3719
|
+
}
|
|
3720
|
+
}
|
|
3721
|
+
return {dated: dated, pairs: pairs, agree: agree};
|
|
3722
|
+
}
|
|
3723
|
+
|
|
3724
|
+
function promoteTimeScaledDates(phy) {
|
|
3725
|
+
let root = forester.getTreeRoot(phy);
|
|
3726
|
+
if (!root) {
|
|
3727
|
+
return;
|
|
3728
|
+
}
|
|
3729
|
+
let counted = countDatePairs(root, numericDateDesc);
|
|
3730
|
+
let dated = counted.dated;
|
|
3731
|
+
let pairs = counted.pairs;
|
|
3732
|
+
let agree = counted.agree;
|
|
3733
|
+
if (pairs < 2 || agree * 2 <= pairs) {
|
|
3734
|
+
return;
|
|
3735
|
+
}
|
|
3736
|
+
dated.forEach(function (d) {
|
|
3737
|
+
d[0].date.value = d[1];
|
|
3738
|
+
d[0].date.unit = 'year';
|
|
3739
|
+
});
|
|
3740
|
+
}
|
|
3741
|
+
|
|
3742
|
+
// ---------------------------------------------------------------
|
|
3743
|
+
// An Auspice num_date is a date VALUE only on a time-scaled tree
|
|
3744
|
+
// ---------------------------------------------------------------
|
|
3745
|
+
//
|
|
3746
|
+
// Auspice's "download Nexus" offers the SAME annotations on two trees: the
|
|
3747
|
+
// time tree (…_timetree.nexus, branch lengths in years) and the divergence
|
|
3748
|
+
// tree (…_tree.nexus, branch lengths in substitutions). A date value is
|
|
3749
|
+
// what makes a tree a time tree here -- isTimeTree counts them -- and the
|
|
3750
|
+
// calendar axis maps one branch-length unit to one year, so dating the
|
|
3751
|
+
// divergence export would hang that axis on a tree measured in
|
|
3752
|
+
// substitutions and refuse its re-rooting. Measured on real exports, the
|
|
3753
|
+
// parent-to-child num_date differences reproduce the branch lengths on
|
|
3754
|
+
// measles timetree 5388 of 5388 pairs
|
|
3755
|
+
// chikv timetree 2645 of 2645 pairs
|
|
3756
|
+
// lassa_gpc tree 169 of 2295 pairs (7.4%)
|
|
3757
|
+
// so this is the question promoteTimeScaledDates already asks of a
|
|
3758
|
+
// TreeTime date=, with the same pinned tolerances, put to the tree rather
|
|
3759
|
+
// than guessed from a file name. The difference is the burden of proof: a
|
|
3760
|
+
// date= may not be a value at all, so it needs evidence FOR; a num_date is
|
|
3761
|
+
// a date by its very name, so it stands unless there is evidence AGAINST
|
|
3762
|
+
// -- two comparable pairs or more, and no strict majority agreeing. A tree
|
|
3763
|
+
// too small to say anything keeps its dates.
|
|
3764
|
+
//
|
|
3765
|
+
// A JOINT RULE with the desktop (Christian, 2026-09-16), chosen over
|
|
3766
|
+
// keeping it here alone, over dating unconditionally, and over moving the
|
|
3767
|
+
// question up into isTimeTree -- which would have changed that answer for
|
|
3768
|
+
// every dated input, phyloXML and BEAST included. Do not retune it alone.
|
|
3769
|
+
//
|
|
3770
|
+
// Where the tree is not time-scaled nothing is lost: the year stays on
|
|
3771
|
+
// the node as nextstrain:num_date (numeric, so Color-by and search have
|
|
3772
|
+
// it) and its interval as nextstrain:num_date_CI, exactly what an
|
|
3773
|
+
// interval with no date to bracket becomes anyway.
|
|
3774
|
+
function settleNumDates(phy) {
|
|
3775
|
+
let root = forester.getTreeRoot(phy);
|
|
3776
|
+
if (!root) {
|
|
3777
|
+
return;
|
|
3778
|
+
}
|
|
3779
|
+
let counted = countDatePairs(root, function (n) {
|
|
3780
|
+
return n._numDate ? n.date.value : null;
|
|
3781
|
+
});
|
|
3782
|
+
let marked = counted.dated.map(function (d) {
|
|
3783
|
+
return d[0];
|
|
3784
|
+
});
|
|
3785
|
+
let pairs = counted.pairs;
|
|
3786
|
+
let agree = counted.agree;
|
|
3787
|
+
let notTimeScaled = pairs >= 2 && agree * 2 <= pairs;
|
|
3788
|
+
marked.forEach(function (n) {
|
|
3789
|
+
let m = n._numDate;
|
|
3790
|
+
delete n._numDate;
|
|
3791
|
+
if (!notTimeScaled) {
|
|
3792
|
+
// A TIP keeps its interval too. It used to be dropped here
|
|
3793
|
+
// and in parseAuspiceJson, for a DISPLAY reason -- a bar on a
|
|
3794
|
+
// tip read as a fossil range -- and that threw real data away:
|
|
3795
|
+
// a sample dated only to its month or year. Counted on real
|
|
3796
|
+
// exports: dengue 2347 of 3863 tips carry a genuine interval
|
|
3797
|
+
// (median 0.78 y), measles 1390 of 2985, enterovirus 715 of
|
|
3798
|
+
// 1600. The display now tells a sampled tip from a fossil
|
|
3799
|
+
// instead (drawTimeAxis); Christian, 2026-09-17, both programs.
|
|
3800
|
+
return;
|
|
3801
|
+
}
|
|
3802
|
+
delete n.date.value;
|
|
3803
|
+
delete n.date.unit;
|
|
3804
|
+
delete n.date.minimum;
|
|
3805
|
+
delete n.date.maximum;
|
|
3806
|
+
if (Object.keys(n.date).length === 0) {
|
|
3807
|
+
delete n.date;
|
|
3808
|
+
}
|
|
3809
|
+
if (m.ciText !== null) {
|
|
3810
|
+
addNodeProperty(n, NEXTSTRAIN_PREFIX + m.ciKey, m.ciText);
|
|
3811
|
+
}
|
|
3812
|
+
});
|
|
3813
|
+
}
|
|
3814
|
+
|
|
3815
|
+
// ---------------------------------------------------------------
|
|
3816
|
+
// TreeTime's own namespace
|
|
3817
|
+
// ---------------------------------------------------------------
|
|
3818
|
+
//
|
|
3819
|
+
// TreeTime's annotations arrive through the BEAST path, so they were
|
|
3820
|
+
// landing as beast:<key> -- accurate about the syntax, wrong about the
|
|
3821
|
+
// producer, and confusing next to a real BEAST run. Christian asked for a
|
|
3822
|
+
// namespace of their own (2026-09-16).
|
|
3823
|
+
//
|
|
3824
|
+
// The producer is recognised on the TREE, not the file: a TreeTime tree
|
|
3825
|
+
// carries mutations= and no node age at all, where every BEAST/MrBayes
|
|
3826
|
+
// run states an age (height, height_mean, height_median, and the
|
|
3827
|
+
// height_95%_HPD / height_range intervals). So the test is "mutations
|
|
3828
|
+
// present, age absent", which cannot fire on a BEAST file and leaves that
|
|
3829
|
+
// shared contract with the desktop untouched.
|
|
3830
|
+
//
|
|
3831
|
+
// TreeTime's mugration output is a bare user-named trait -- [®ion="x"]
|
|
3832
|
+
// and nothing else -- which no rule could attribute to any producer. It
|
|
3833
|
+
// keeps the generic namespace, correctly.
|
|
3834
|
+
const TREETIME_PREFIX = 'treetime:';
|
|
3835
|
+
const BEAST_PREFIX = 'beast:';
|
|
3836
|
+
|
|
3837
|
+
// Must run BEFORE promoteTimeScaledDates: until then a date VALUE can
|
|
3838
|
+
// only have come from a BEAST height field or an Auspice num_date, and
|
|
3839
|
+
// either one says this is not TreeTime's own Nexus.
|
|
3840
|
+
function renameTreeTimeProperties(phy) {
|
|
3841
|
+
let nodes = forester.getAllNodes(phy);
|
|
3842
|
+
let mutations = false;
|
|
3843
|
+
for (let i = 0; i < nodes.length; ++i) {
|
|
3844
|
+
let d = nodes[i].date;
|
|
3845
|
+
if (d && (d.value !== undefined || d.minimum !== undefined
|
|
3846
|
+
|| d.maximum !== undefined)) {
|
|
3847
|
+
return;
|
|
3848
|
+
}
|
|
3849
|
+
if (!mutations && nodes[i].properties) {
|
|
3850
|
+
mutations = nodes[i].properties.some(function (p) {
|
|
3851
|
+
return p.ref === BEAST_PREFIX + 'mutations';
|
|
3852
|
+
});
|
|
3853
|
+
}
|
|
3854
|
+
}
|
|
3855
|
+
if (!mutations) {
|
|
3856
|
+
return;
|
|
3857
|
+
}
|
|
3858
|
+
nodes.forEach(function (n) {
|
|
3859
|
+
if (n.properties) {
|
|
3860
|
+
n.properties.forEach(function (p) {
|
|
3861
|
+
if (typeof p.ref === 'string' && p.ref.indexOf(BEAST_PREFIX) === 0) {
|
|
3862
|
+
p.ref = TREETIME_PREFIX + p.ref.substring(BEAST_PREFIX.length);
|
|
3863
|
+
}
|
|
3864
|
+
});
|
|
3865
|
+
}
|
|
3866
|
+
});
|
|
3867
|
+
}
|
|
3868
|
+
|
|
3869
|
+
// The namespace this tree's bracket annotations ended up in, for anything
|
|
3870
|
+
// added to it AFTER the pass above has run -- the Nexus reader hangs a
|
|
3871
|
+
// taxon's refused colour on its tip once the tree string is parsed, and
|
|
3872
|
+
// filed it under beast: on a tree the pass had just renamed: one tree, two
|
|
3873
|
+
// namespaces. The rename is all-or-nothing, so the tree already says which
|
|
3874
|
+
// it took; and asking it, rather than running the pass a second time, is
|
|
3875
|
+
// deliberate -- after promoteTimeScaledDates a TreeTime tree HAS date
|
|
3876
|
+
// values, and a second run would read that as "not TreeTime's own".
|
|
3877
|
+
// (Sound only while nothing read from a taxon can sway the decision; we
|
|
3878
|
+
// read !color alone there. The desktop reads the whole blob, and so has to
|
|
3879
|
+
// run its pass after the taxlabels instead.)
|
|
3880
|
+
function annotationPrefix(phy) {
|
|
3881
|
+
let nodes = forester.getAllNodes(phy);
|
|
3882
|
+
for (let i = 0; i < nodes.length; ++i) {
|
|
3883
|
+
let props = nodes[i].properties;
|
|
3884
|
+
if (props) {
|
|
3885
|
+
for (let j = 0; j < props.length; ++j) {
|
|
3886
|
+
if (typeof props[j].ref === 'string' && props[j].ref.indexOf(TREETIME_PREFIX) === 0) {
|
|
3887
|
+
return TREETIME_PREFIX;
|
|
3888
|
+
}
|
|
3889
|
+
}
|
|
3890
|
+
}
|
|
3891
|
+
}
|
|
3892
|
+
return BEAST_PREFIX;
|
|
3893
|
+
}
|
|
3894
|
+
|
|
2861
3895
|
forester.parseNewHampshire = function (nhStr, confidenceValuesInBrackets, confidenceValuesAsInternalNames) {
|
|
2862
3896
|
|
|
2863
3897
|
let NH_FORMAT_ERR_OPEN_PARENS = NH_FORMAT_ERR + 'likely cause: number of open parentheses is larger than number of close parentheses';
|
|
@@ -3087,6 +4121,10 @@
|
|
|
3087
4121
|
moveInternalNodeNamesToConfidenceValues(phy);
|
|
3088
4122
|
}
|
|
3089
4123
|
|
|
4124
|
+
renameTreeTimeProperties(phy); // first: a provisional num_date still says "not TreeTime's own"
|
|
4125
|
+
settleNumDates(phy);
|
|
4126
|
+
promoteTimeScaledDates(phy);
|
|
4127
|
+
|
|
3090
4128
|
return phy;
|
|
3091
4129
|
|
|
3092
4130
|
function addConfidence(x, element) {
|
|
@@ -3222,6 +4260,11 @@
|
|
|
3222
4260
|
|
|
3223
4261
|
let trees = [];
|
|
3224
4262
|
let taxlabels = [];
|
|
4263
|
+
let taxlabelColors = Object.create(null); // label -> #rrggbb, from 'name'[&!color=...]
|
|
4264
|
+
let taxlabelColorsByKey = Object.create(null); // the same, under the Nexus join key
|
|
4265
|
+
let taxlabelRefused = Object.create(null); // label -> a !color value we could not read, kept as text
|
|
4266
|
+
let taxlabelRefusedByKey = Object.create(null);
|
|
4267
|
+
let taxlabelKeyCount = Object.create(null); // join key -> how many taxlabels share it, annotated or not
|
|
3225
4268
|
// null-prototype maps: a taxon named "__proto__" must stay data
|
|
3226
4269
|
let translateMap = Object.create(null);
|
|
3227
4270
|
let seqs = Object.create(null);
|
|
@@ -3386,6 +4429,7 @@
|
|
|
3386
4429
|
seqsByKey[joinKey(id)] = seqs[id];
|
|
3387
4430
|
}
|
|
3388
4431
|
let externals = forester.getAllExternalNodes(phy);
|
|
4432
|
+
let annotationNs = null;
|
|
3389
4433
|
// A bare integer tip name counts as a TAXLABELS index only when
|
|
3390
4434
|
// the WHOLE tree reads as index references: every tip a bare
|
|
3391
4435
|
// integer AND every one of them in range. All-or-nothing, because
|
|
@@ -3418,6 +4462,41 @@
|
|
|
3418
4462
|
// un-doubling and drop the apostrophe a second time.
|
|
3419
4463
|
node.name = taxlabels[parseInt(node.name, 10) - 1];
|
|
3420
4464
|
}
|
|
4465
|
+
// A colour FigTree gave the TAXON is the colour of its LABEL
|
|
4466
|
+
// -- FigTree's own meaning, and the desktop's (Christian,
|
|
4467
|
+
// 2026-09-16) -- where a !color in the tree string is the
|
|
4468
|
+
// BRANCH's. It lands as the desktop's style:font_color
|
|
4469
|
+
// property, which is what Visual Styles already draws and what
|
|
4470
|
+
// phyloXML already carries, so nothing downstream is new.
|
|
4471
|
+
// The taxon is found by its exact name, else by the Nexus join
|
|
4472
|
+
// key (case-insensitive, '_' for ' ') -- but by the key ONLY
|
|
4473
|
+
// when it names exactly ONE taxlabel, counting every taxlabel,
|
|
4474
|
+
// annotated or not. Without that, Taxon_A[&!color=red] beside a
|
|
4475
|
+
// plain taxon_a coloured BOTH tips: taxon_a has no annotation
|
|
4476
|
+
// of its own, so it fell through to a key it shares. And a tip
|
|
4477
|
+
// matching two labels by key alone is ambiguous: no colour,
|
|
4478
|
+
// rather than whichever was written last. (Found by a review on
|
|
4479
|
+
// the desktop, which had the same fallback; its rule.)
|
|
4480
|
+
let loneKey = !!node.name && taxlabelKeyCount[joinKey(node.name)] === 1;
|
|
4481
|
+
let labelColor = !node.name ? undefined
|
|
4482
|
+
: (taxlabelColors[node.name] !== undefined ? taxlabelColors[node.name]
|
|
4483
|
+
: (loneKey ? taxlabelColorsByKey[joinKey(node.name)] : undefined));
|
|
4484
|
+
if (labelColor !== undefined) {
|
|
4485
|
+
if (!node.properties) {
|
|
4486
|
+
node.properties = [];
|
|
4487
|
+
}
|
|
4488
|
+
node.properties.push({ref: 'style:font_color', value: labelColor,
|
|
4489
|
+
datatype: 'xsd:token', applies_to: 'node'});
|
|
4490
|
+
}
|
|
4491
|
+
let refusedColor = !node.name ? undefined
|
|
4492
|
+
: (taxlabelRefused[node.name] !== undefined ? taxlabelRefused[node.name]
|
|
4493
|
+
: (loneKey ? taxlabelRefusedByKey[joinKey(node.name)] : undefined));
|
|
4494
|
+
if (refusedColor !== undefined) {
|
|
4495
|
+
if (annotationNs === null) {
|
|
4496
|
+
annotationNs = annotationPrefix(phy); // once per tree, and only if needed
|
|
4497
|
+
}
|
|
4498
|
+
addNodeProperty(node, annotationNs + '_color', refusedColor, 'xsd:string');
|
|
4499
|
+
}
|
|
3421
4500
|
if (node.name) {
|
|
3422
4501
|
let s = seqsByKey[joinKey(node.name)];
|
|
3423
4502
|
if (s) {
|
|
@@ -3540,6 +4619,7 @@
|
|
|
3540
4619
|
let push = function () {
|
|
3541
4620
|
if (tok.length > 0 && tok.toLowerCase() !== 'taxlabels') {
|
|
3542
4621
|
taxlabels.push(tok);
|
|
4622
|
+
taxlabelKeyCount[joinKey(tok)] = (taxlabelKeyCount[joinKey(tok)] || 0) + 1;
|
|
3543
4623
|
}
|
|
3544
4624
|
tok = '';
|
|
3545
4625
|
closed = false;
|
|
@@ -3577,6 +4657,48 @@
|
|
|
3577
4657
|
// divergence, not a fix.
|
|
3578
4658
|
void ch;
|
|
3579
4659
|
}
|
|
4660
|
+
} else if (ch === '[') {
|
|
4661
|
+
// A bracket after a label is not part of it. FigTree
|
|
4662
|
+
// hangs the taxon's colour here --
|
|
4663
|
+
// 'NewYork_454_1999.05'[&!color=#-8381639] -- and it
|
|
4664
|
+
// used to be glued onto the label, which then named
|
|
4665
|
+
// no tip (invisible while a tree spells its tips
|
|
4666
|
+
// out, wrong as soon as it refers to them by number).
|
|
4667
|
+
// A [&...] belongs to the label just read; any other
|
|
4668
|
+
// bracket is an ordinary Nexus comment.
|
|
4669
|
+
//
|
|
4670
|
+
// Only !color is read from it, and that is DECIDED
|
|
4671
|
+
// (Christian, 2026-09-17: "keep it as it is"), a named
|
|
4672
|
+
// difference from the desktop, which keeps every field
|
|
4673
|
+
// of a taxon's annotation. No producer we know writes
|
|
4674
|
+
// anything else here. Widening it is not a small edit:
|
|
4675
|
+
// a posterior, a height or mutations arriving by this
|
|
4676
|
+
// road could sway the per-tree pass, which would then
|
|
4677
|
+
// have to run after the taxlabels (see annotationPrefix).
|
|
4678
|
+
let end = line.indexOf(']', ci);
|
|
4679
|
+
let inside = line.substring(ci + 1, end < 0 ? line.length : end).trim();
|
|
4680
|
+
if (inside.charAt(0) === '&' && tok.length > 0) {
|
|
4681
|
+
splitTopLevelCommas(inside.substring(1), true).forEach(function (field) {
|
|
4682
|
+
let eq = field.indexOf('=');
|
|
4683
|
+
let key = eq > 0 ? field.substring(0, eq).trim().toLowerCase() : '';
|
|
4684
|
+
let isColor = key === '!color' || key === '!colour';
|
|
4685
|
+
let raw = isColor ? stripValueQuotes(field.substring(eq + 1).trim()) : '';
|
|
4686
|
+
let rgb = isColor ? parseFigTreeColor(raw) : null;
|
|
4687
|
+
if (isColor && !rgb && raw.length > 0) {
|
|
4688
|
+
// not a colour we read (no '#', say): kept as
|
|
4689
|
+
// text on the tip, as in the tree string
|
|
4690
|
+
taxlabelRefused[tok] = raw;
|
|
4691
|
+
taxlabelRefusedByKey[joinKey(tok)] = raw;
|
|
4692
|
+
}
|
|
4693
|
+
if (rgb) {
|
|
4694
|
+
taxlabelColors[tok] = '#' + [rgb.red, rgb.green, rgb.blue].map(function (c) {
|
|
4695
|
+
return (c < 16 ? '0' : '') + c.toString(16);
|
|
4696
|
+
}).join('');
|
|
4697
|
+
taxlabelColorsByKey[joinKey(tok)] = taxlabelColors[tok];
|
|
4698
|
+
}
|
|
4699
|
+
});
|
|
4700
|
+
}
|
|
4701
|
+
ci = end < 0 ? line.length : end;
|
|
3580
4702
|
} else if (ch === ' ') {
|
|
3581
4703
|
push();
|
|
3582
4704
|
} else if (ch === ';') {
|
|
@@ -3689,7 +4811,9 @@
|
|
|
3689
4811
|
return d.toFixed(20).replace(/0+$/, '').replace(/\.$/, '');
|
|
3690
4812
|
}
|
|
3691
4813
|
|
|
3692
|
-
|
|
4814
|
+
// `datatype` is for the few values whose type is decided by what they ARE
|
|
4815
|
+
// rather than by what they look like (see mutations / mcc).
|
|
4816
|
+
function addNodeProperty(node, ref, value, datatype) {
|
|
3693
4817
|
if (value === undefined || value === null || String(value).length === 0) {
|
|
3694
4818
|
return;
|
|
3695
4819
|
}
|
|
@@ -3700,7 +4824,7 @@
|
|
|
3700
4824
|
node.properties.push({
|
|
3701
4825
|
ref: ref,
|
|
3702
4826
|
value: v,
|
|
3703
|
-
datatype:
|
|
4827
|
+
datatype: datatype || (parseBeastNumber(v) !== null ? 'xsd:decimal' : 'xsd:string'),
|
|
3704
4828
|
applies_to: 'node'
|
|
3705
4829
|
});
|
|
3706
4830
|
}
|
|
@@ -3708,7 +4832,8 @@
|
|
|
3708
4832
|
// Parses an Auspice / Nextstrain v2 dataset.json (string or already-parsed
|
|
3709
4833
|
// object) into ONE tree object, mapping its per-node data onto the native
|
|
3710
4834
|
// phyloXML shape so the existing overlays light it up -- ported from the
|
|
3711
|
-
// desktop's AuspiceJsonParser
|
|
4835
|
+
// desktop's AuspiceJsonParser. TreeTime's own auspice_tree.json is read
|
|
4836
|
+
// here too (see the version check below):
|
|
3712
4837
|
// - node_attrs.num_date.value -> node.date value (decimal year) -> the
|
|
3713
4838
|
// calendar time axis; its .confidence [lo,hi] -> date minimum/maximum
|
|
3714
4839
|
// -> the node-age (HPD) bars;
|
|
@@ -3727,8 +4852,21 @@
|
|
|
3727
4852
|
if (!doc || typeof doc !== 'object' || Array.isArray(doc)) {
|
|
3728
4853
|
throw new Error('not an Auspice dataset (the JSON root is not an object)');
|
|
3729
4854
|
}
|
|
3730
|
-
if (
|
|
3731
|
-
|
|
4855
|
+
if (!doc.tree || typeof doc.tree !== 'object' || Array.isArray(doc.tree)) {
|
|
4856
|
+
throw new Error('not an Auspice v2 dataset (expected "version":"v2" and a "tree" object)');
|
|
4857
|
+
}
|
|
4858
|
+
// The version stamp is taken as PRESENT-OR-IMPLIED: TreeTime writes a
|
|
4859
|
+
// fully valid v2 dataset ({meta, tree} with node_attrs.num_date,
|
|
4860
|
+
// branch_attrs, children) and simply never writes "version":"v2", so
|
|
4861
|
+
// demanding the stamp rejected the richest file TreeTime produces --
|
|
4862
|
+
// the only one carrying full-precision dates (its .nexus rounds them
|
|
4863
|
+
// to two decimals). The shape is still checked, so arbitrary JSON
|
|
4864
|
+
// keeps getting the clear error rather than a confusing parse.
|
|
4865
|
+
if (doc.version !== 'v2'
|
|
4866
|
+
&& !(doc.meta && typeof doc.meta === 'object' && !Array.isArray(doc.meta)
|
|
4867
|
+
&& (typeof doc.tree.name === 'string'
|
|
4868
|
+
|| (doc.tree.node_attrs && typeof doc.tree.node_attrs === 'object')
|
|
4869
|
+
|| Array.isArray(doc.tree.children)))) {
|
|
3732
4870
|
throw new Error('not an Auspice v2 dataset (expected "version":"v2" and a "tree" object)');
|
|
3733
4871
|
}
|
|
3734
4872
|
|
|
@@ -3836,16 +4974,11 @@
|
|
|
3836
4974
|
// keep the layout meaningful instead of a cladogram
|
|
3837
4975
|
setDeltaBranchLengths(root, null, auspiceNodeDiv);
|
|
3838
4976
|
}
|
|
3839
|
-
// A
|
|
3840
|
-
//
|
|
3841
|
-
//
|
|
3842
|
-
//
|
|
3843
|
-
|
|
3844
|
-
if (!n.children && n.date
|
|
3845
|
-
&& (n.date.minimum !== undefined || n.date.maximum !== undefined)) {
|
|
3846
|
-
n.date = {value: n.date.value, unit: n.date.unit};
|
|
3847
|
-
}
|
|
3848
|
-
});
|
|
4977
|
+
// A tip keeps its date INTERVAL: on a Nextstrain build it is the
|
|
4978
|
+
// sampling-date uncertainty of a sample dated only to its month or
|
|
4979
|
+
// year, which is data. (It was dropped here until 2026-09-17 because
|
|
4980
|
+
// the time axis drew every tip interval as a fossil range; the axis
|
|
4981
|
+
// now tells the two apart -- see settleNumDates and drawTimeAxis.)
|
|
3849
4982
|
forester.addParents(phy);
|
|
3850
4983
|
return phy;
|
|
3851
4984
|
};
|