unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,606 @@
|
|
|
1
|
+
"""Campaign-scale checking of chemical class definitions against molecules.
|
|
2
|
+
|
|
3
|
+
:func:`~unicode_logic_kit.eval.datasets.c3po.score_definition` answers "how good
|
|
4
|
+
is THIS definition" and returns a confusion matrix. This module answers the
|
|
5
|
+
other question a campaign asks — "run these K definitions over these N
|
|
6
|
+
molecules and write down everything that happened" — and its contract is
|
|
7
|
+
shaped by the two ways such a run fails in practice.
|
|
8
|
+
|
|
9
|
+
**A run never stops because of data.** One unparseable SMILES in a set of
|
|
10
|
+
200 000, or one definition mentioning a predicate no structure interprets,
|
|
11
|
+
must not cost the other 199 999 rows. Every such failure becomes a ROW with a
|
|
12
|
+
``status``, never an exception. The dividing line is exactly: is this a
|
|
13
|
+
property of the input data (row) or of the configuration (raise, immediately,
|
|
14
|
+
before the first molecule)? RDKit not installed, an unknown ``naming``, an
|
|
15
|
+
unwritable results path — those are configuration, and they fail loudly up
|
|
16
|
+
front rather than as 200 000 identical error rows.
|
|
17
|
+
|
|
18
|
+
**Partial results survive an interrupt.** Rows are flushed to JSONL after each
|
|
19
|
+
DEFINITION rather than at the end, so a run killed at 90 % keeps 90 %. With
|
|
20
|
+
``resume=True`` a re-run reads back what is already there and skips those
|
|
21
|
+
``(definition, molecule)`` pairs — the campaign's own restart mechanism, not
|
|
22
|
+
an afterthought.
|
|
23
|
+
|
|
24
|
+
**The structure is built once per molecule, not once per check.** That is what
|
|
25
|
+
:class:`~unicode_logic_kit.chem.cache.StructureCache` is for, and why this
|
|
26
|
+
module loops definition-outer/molecule-inner: the inner loop hits the cache.
|
|
27
|
+
Measured speed-up on this kit: 1.6× to 59× depending on how fast the formula
|
|
28
|
+
short-circuits, around 3–8× for a realistic mix (see the cache module's own
|
|
29
|
+
docstring for the numbers and the method).
|
|
30
|
+
|
|
31
|
+
Statuses a row can carry:
|
|
32
|
+
|
|
33
|
+
``ok``
|
|
34
|
+
The formula was evaluated; ``holds`` is ``True``/``False``.
|
|
35
|
+
``exhausted``
|
|
36
|
+
The evaluation budget ran out. ``holds`` is ``None`` — NEVER ``False``.
|
|
37
|
+
A separate status because counting an undecided check as a negative is
|
|
38
|
+
how an evaluation quietly flatters itself.
|
|
39
|
+
``structure_error``
|
|
40
|
+
RDKit refused the SMILES. Cached, so it costs one attempt per molecule
|
|
41
|
+
for the whole run, not one per definition.
|
|
42
|
+
``parse_error`` / ``eval_error``
|
|
43
|
+
The definition did not parse (recorded ONCE, on a synthetic row with
|
|
44
|
+
``smiles=None``, and the definition is then skipped), or the evaluator
|
|
45
|
+
refused this structure/formula pair (per row — typically an
|
|
46
|
+
:class:`~unicode_logic_kit.semantics.model_eval.UninterpretedSymbol` for a
|
|
47
|
+
class predicate that only exists in another definition).
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
import io
|
|
51
|
+
import json
|
|
52
|
+
import os
|
|
53
|
+
import time
|
|
54
|
+
from dataclasses import dataclass, field
|
|
55
|
+
from typing import (
|
|
56
|
+
Any, Callable, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple,
|
|
57
|
+
Union,
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
from ..chem.cache import StructureBuildError, StructureCache
|
|
61
|
+
from ..fol.nodes import Atom, Node
|
|
62
|
+
from ..semantics.model_eval import (
|
|
63
|
+
evaluate_detailed, UninterpretedSymbol, UnsupportedNode,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
__all__ = ["ChemBatchResult", "check_definitions"]
|
|
67
|
+
|
|
68
|
+
_STATUSES = ("ok", "exhausted", "structure_error", "parse_error", "eval_error")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
#: :meth:`ChemBatchResult.to_markdown`/:meth:`~ChemBatchResult.to_html`
|
|
72
|
+
#: default cap on how many non-``ok`` rows are rendered — see their
|
|
73
|
+
#: docstrings for why a campaign-scale result must never dump every row.
|
|
74
|
+
#: Private (not re-exported): a caller who wants a different cap just passes
|
|
75
|
+
#: ``max_rows=`` explicitly rather than importing this default.
|
|
76
|
+
_DEFAULT_MAX_ROWS = 25
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@dataclass(frozen=True)
|
|
80
|
+
class ChemBatchResult:
|
|
81
|
+
"""Outcome of :func:`check_definitions`.
|
|
82
|
+
|
|
83
|
+
``rows`` is every row produced BY THIS CALL — rows skipped because a
|
|
84
|
+
resume found them already written are not re-reported, so ``counts`` and
|
|
85
|
+
``rows`` describe the work this call did, while the JSONL file describes
|
|
86
|
+
the run as a whole. ``skipped`` says how many pairs the resume passed
|
|
87
|
+
over, so the difference is never silent.
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
rows: Tuple[dict, ...]
|
|
91
|
+
counts: Dict[str, int]
|
|
92
|
+
cache_stats: Dict[str, Any]
|
|
93
|
+
seconds: float
|
|
94
|
+
skipped: int = 0
|
|
95
|
+
|
|
96
|
+
def to_dict(self) -> dict:
|
|
97
|
+
return {"rows": list(self.rows), "counts": dict(self.counts),
|
|
98
|
+
"cache_stats": dict(self.cache_stats),
|
|
99
|
+
"seconds": self.seconds, "skipped": self.skipped}
|
|
100
|
+
|
|
101
|
+
def to_markdown(self, max_rows: int = _DEFAULT_MAX_ROWS) -> str:
|
|
102
|
+
"""Render a summary as plain-formatted Markdown — NOT a row dump.
|
|
103
|
+
|
|
104
|
+
A campaign run is hundreds of thousands of rows; a renderer that
|
|
105
|
+
printed every one would produce multi-hundred-MB output and defeat
|
|
106
|
+
the point of a summary. This renders ``counts``/``cache_stats``/
|
|
107
|
+
``seconds``/``skipped`` as summary tables, then an explicit, capped
|
|
108
|
+
sample of at most ``max_rows`` non-``ok`` rows (their
|
|
109
|
+
``error_msg``/``unknown_predicates``/``witness``), and always closes
|
|
110
|
+
with an honest "showing N of M" note — never a silent truncation.
|
|
111
|
+
|
|
112
|
+
Every rendered field goes through :func:`_md_cell`, so a hostile
|
|
113
|
+
SMILES/error message (one containing ``|`` or a newline) cannot
|
|
114
|
+
corrupt the table structure. Calling this twice on the same result
|
|
115
|
+
returns the identical string.
|
|
116
|
+
"""
|
|
117
|
+
return "\n".join(_chem_markdown_lines(self, max_rows))
|
|
118
|
+
|
|
119
|
+
def to_html(self, title: str = "Chem batch result",
|
|
120
|
+
max_rows: int = _DEFAULT_MAX_ROWS) -> str:
|
|
121
|
+
"""Render as a self-contained, theme-aware HTML page — same summary
|
|
122
|
+
content and the same ``max_rows`` cap as :meth:`to_markdown`, in the
|
|
123
|
+
idiom :meth:`unicode_logic_kit.fol.derivation.CCGDerivation.to_html`
|
|
124
|
+
established. Every user-supplied string (SMILES, error message,
|
|
125
|
+
witness) is HTML-escaped.
|
|
126
|
+
"""
|
|
127
|
+
return _html_page(title, _chem_html_body(self, max_rows), _CHEM_HTML_CSS)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _resolve(formula: Union[Node, str], dialect: str) -> Node:
|
|
131
|
+
"""Definition text -> a ChemLog-spelled :class:`Node`.
|
|
132
|
+
|
|
133
|
+
Delegates to :func:`unicode_logic_kit.eval.datasets.c3po._resolve_formula`
|
|
134
|
+
rather than reimplementing the dialect routing and the ChemLog rename:
|
|
135
|
+
two implementations of "which parser, then which rename" is exactly how
|
|
136
|
+
the unicode path silently stopped matching structures once before.
|
|
137
|
+
"""
|
|
138
|
+
from .datasets.c3po import _resolve_formula
|
|
139
|
+
|
|
140
|
+
return _resolve_formula(formula, dialect)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _unknown_predicates(formula: Node, structure) -> List[str]:
|
|
144
|
+
"""Predicates the formula uses that this structure does not interpret.
|
|
145
|
+
|
|
146
|
+
Informational, never fatal. A ChEBI class definition legitimately names
|
|
147
|
+
OTHER class predicates (``carboxylicAcid``, ``lipid``) that no molecule
|
|
148
|
+
structure can decide — they have to be unfolded first — and refusing the
|
|
149
|
+
definition for it would reject most of a real corpus. Reporting them once
|
|
150
|
+
per definition turns a wall of identical per-row ``eval_error``s into one
|
|
151
|
+
actionable line.
|
|
152
|
+
"""
|
|
153
|
+
unknown = set()
|
|
154
|
+
for part in formula.walk():
|
|
155
|
+
if isinstance(part, Atom) and part.predicate not in ("=", "≠"):
|
|
156
|
+
if not structure.interprets(part.predicate, len(part.args)):
|
|
157
|
+
unknown.add(f"{part.predicate}/{len(part.args)}")
|
|
158
|
+
return sorted(unknown)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _existing_pairs(path: str) -> set:
|
|
162
|
+
"""``(def_id, smiles)`` pairs already in the results file.
|
|
163
|
+
|
|
164
|
+
A truncated final line (the normal shape of an interrupted run) is
|
|
165
|
+
ignored rather than raising: the pair it described simply gets redone.
|
|
166
|
+
"""
|
|
167
|
+
pairs = set()
|
|
168
|
+
if not os.path.exists(path):
|
|
169
|
+
return pairs
|
|
170
|
+
with io.open(path, encoding="utf-8") as handle:
|
|
171
|
+
for line in handle:
|
|
172
|
+
line = line.strip()
|
|
173
|
+
if not line:
|
|
174
|
+
continue
|
|
175
|
+
try:
|
|
176
|
+
row = json.loads(line)
|
|
177
|
+
except ValueError:
|
|
178
|
+
continue
|
|
179
|
+
pairs.add((row.get("def_id"), row.get("smiles")))
|
|
180
|
+
return pairs
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def check_definitions(
|
|
184
|
+
definitions: Iterable[Mapping[str, Any]],
|
|
185
|
+
molecules: Sequence[str],
|
|
186
|
+
*,
|
|
187
|
+
results_path: Optional[str] = None,
|
|
188
|
+
dialect: str = "tptp",
|
|
189
|
+
naming: str = "chemlog",
|
|
190
|
+
aromatic: bool = True,
|
|
191
|
+
computed: bool = True,
|
|
192
|
+
all_different: bool = True,
|
|
193
|
+
budget: Optional[int] = None,
|
|
194
|
+
cache: Optional[StructureCache] = None,
|
|
195
|
+
resume: bool = True,
|
|
196
|
+
progress: Optional[Callable[[str, int, int], None]] = None,
|
|
197
|
+
) -> ChemBatchResult:
|
|
198
|
+
"""Evaluate every definition against every molecule.
|
|
199
|
+
|
|
200
|
+
Args:
|
|
201
|
+
definitions: mappings with ``"id"`` and ``"formula"`` (text in
|
|
202
|
+
``dialect``, or an already-parsed
|
|
203
|
+
:class:`~unicode_logic_kit.fol.nodes.Node`). Validated up front —
|
|
204
|
+
a missing key is a caller bug, not data, and raises.
|
|
205
|
+
molecules: SMILES strings. Only strings: an already-parsed
|
|
206
|
+
``rdkit.Chem.Mol`` cannot be a cache key (object identity would
|
|
207
|
+
defeat the cache and the key would be mutable).
|
|
208
|
+
results_path: JSONL to append to, flushed after each definition.
|
|
209
|
+
``None`` keeps everything in memory only.
|
|
210
|
+
dialect: ``"tptp"`` (the ChemLog format) or ``"unicode"``.
|
|
211
|
+
naming / aromatic / computed: structure-building options, forwarded to
|
|
212
|
+
:func:`~unicode_logic_kit.chem.mol_to_structure` AND part of the
|
|
213
|
+
cache key — see :mod:`unicode_logic_kit.chem.cache`.
|
|
214
|
+
all_different: ChemLog's convention that separately introduced
|
|
215
|
+
existential variables denote distinct individuals. Defaults to
|
|
216
|
+
``True`` here, matching every other chemical entry point in the
|
|
217
|
+
kit (and unlike the underlying evaluator's plain-FOL default).
|
|
218
|
+
budget: evaluation step budget per check. ``None`` is unbounded; a
|
|
219
|
+
campaign should set one, and an exhausted check is reported as
|
|
220
|
+
``status="exhausted"`` with ``holds=None``.
|
|
221
|
+
cache: a shared :class:`~unicode_logic_kit.chem.cache.StructureCache`.
|
|
222
|
+
One is created if omitted; pass your own to share it across calls
|
|
223
|
+
(that is where the speed-up lives).
|
|
224
|
+
resume: skip ``(def_id, smiles)`` pairs already present in
|
|
225
|
+
``results_path``.
|
|
226
|
+
progress: called as ``(def_id, index, total)`` after each definition.
|
|
227
|
+
|
|
228
|
+
Returns:
|
|
229
|
+
A :class:`ChemBatchResult`.
|
|
230
|
+
|
|
231
|
+
Raises:
|
|
232
|
+
ValueError: a definition without ``id``/``formula``, a non-string
|
|
233
|
+
SMILES, an unknown ``naming``/``dialect``, or an unwritable
|
|
234
|
+
``results_path`` — all configuration, all detected before the
|
|
235
|
+
first molecule is built.
|
|
236
|
+
ImportError: RDKit is not installed.
|
|
237
|
+
"""
|
|
238
|
+
started = time.perf_counter()
|
|
239
|
+
if naming not in ("chemlog", "paper"):
|
|
240
|
+
raise ValueError(f"check_definitions: unknown naming={naming!r}")
|
|
241
|
+
if dialect not in ("tptp", "unicode"):
|
|
242
|
+
raise ValueError(f"check_definitions: unknown dialect={dialect!r}")
|
|
243
|
+
|
|
244
|
+
prepared: List[Tuple[str, Any]] = []
|
|
245
|
+
for entry in definitions:
|
|
246
|
+
if "id" not in entry or "formula" not in entry:
|
|
247
|
+
raise ValueError(
|
|
248
|
+
"check_definitions: every definition needs 'id' and 'formula'; "
|
|
249
|
+
f"got keys {sorted(entry)}")
|
|
250
|
+
prepared.append((str(entry["id"]), entry["formula"]))
|
|
251
|
+
for smiles in molecules:
|
|
252
|
+
if not isinstance(smiles, str):
|
|
253
|
+
raise ValueError(
|
|
254
|
+
"check_definitions: molecules must be SMILES strings, got "
|
|
255
|
+
f"{type(smiles).__name__}")
|
|
256
|
+
|
|
257
|
+
if results_path:
|
|
258
|
+
directory = os.path.dirname(os.path.abspath(results_path))
|
|
259
|
+
if directory and not os.path.isdir(directory):
|
|
260
|
+
raise ValueError(
|
|
261
|
+
f"check_definitions: results_path directory does not exist: "
|
|
262
|
+
f"{directory}")
|
|
263
|
+
|
|
264
|
+
cache = cache if cache is not None else StructureCache()
|
|
265
|
+
# Fail on a missing RDKit here, once, rather than per molecule: it is the
|
|
266
|
+
# environment, not the data. Deliberately NOT through the cache — a
|
|
267
|
+
# pre-warm would count as a hit on the first row and make the reported
|
|
268
|
+
# hit rate flatter than the run actually achieved, which is the one
|
|
269
|
+
# number M2 exists to measure.
|
|
270
|
+
if molecules:
|
|
271
|
+
from ..chem import mol_to_structure
|
|
272
|
+
|
|
273
|
+
try:
|
|
274
|
+
mol_to_structure(molecules[0], naming=naming, aromatic=aromatic,
|
|
275
|
+
computed=computed)
|
|
276
|
+
except ValueError:
|
|
277
|
+
pass # a bad FIRST molecule is data; it gets its own row below
|
|
278
|
+
|
|
279
|
+
done = _existing_pairs(results_path) if (resume and results_path) else set()
|
|
280
|
+
rows: List[dict] = []
|
|
281
|
+
counts: Dict[str, int] = {status: 0 for status in _STATUSES}
|
|
282
|
+
skipped = 0
|
|
283
|
+
|
|
284
|
+
def record(row: dict) -> None:
|
|
285
|
+
rows.append(row)
|
|
286
|
+
counts[row["status"]] = counts.get(row["status"], 0) + 1
|
|
287
|
+
|
|
288
|
+
for index, (def_id, raw_formula) in enumerate(prepared, 1):
|
|
289
|
+
slab: List[dict] = []
|
|
290
|
+
try:
|
|
291
|
+
formula = _resolve(raw_formula, dialect)
|
|
292
|
+
except Exception as exc: # noqa: BLE001 — every parser has its own
|
|
293
|
+
row = {"def_id": def_id, "smiles": None, "status": "parse_error",
|
|
294
|
+
"holds": None, "steps": None, "exhausted": None,
|
|
295
|
+
"witness": None, "error_kind": type(exc).__name__,
|
|
296
|
+
"error_msg": str(exc)[:400], "ms": None, "cached": None}
|
|
297
|
+
record(row)
|
|
298
|
+
slab.append(row)
|
|
299
|
+
_flush(results_path, slab)
|
|
300
|
+
if progress:
|
|
301
|
+
progress(def_id, index, len(prepared))
|
|
302
|
+
continue
|
|
303
|
+
|
|
304
|
+
reported_unknown = False
|
|
305
|
+
for smiles in molecules:
|
|
306
|
+
if (def_id, smiles) in done:
|
|
307
|
+
skipped += 1
|
|
308
|
+
continue
|
|
309
|
+
cell_started = time.perf_counter()
|
|
310
|
+
before = cache.misses
|
|
311
|
+
structure = cache.structure_for(smiles, naming=naming,
|
|
312
|
+
aromatic=aromatic, computed=computed)
|
|
313
|
+
was_cached = cache.misses == before
|
|
314
|
+
row = {"def_id": def_id, "smiles": smiles, "status": "ok",
|
|
315
|
+
"holds": None, "steps": None, "exhausted": False,
|
|
316
|
+
"witness": None, "error_kind": None, "error_msg": None,
|
|
317
|
+
"ms": None, "cached": was_cached}
|
|
318
|
+
|
|
319
|
+
if isinstance(structure, StructureBuildError):
|
|
320
|
+
row.update(status="structure_error", exhausted=None,
|
|
321
|
+
error_kind="StructureBuildError",
|
|
322
|
+
error_msg=structure.message)
|
|
323
|
+
else:
|
|
324
|
+
if not reported_unknown:
|
|
325
|
+
unknown = _unknown_predicates(formula, structure)
|
|
326
|
+
reported_unknown = True
|
|
327
|
+
if unknown:
|
|
328
|
+
row["unknown_predicates"] = unknown
|
|
329
|
+
try:
|
|
330
|
+
result = evaluate_detailed(
|
|
331
|
+
formula, structure, all_different=all_different,
|
|
332
|
+
budget=budget)
|
|
333
|
+
except (UninterpretedSymbol, UnsupportedNode, ValueError) as exc:
|
|
334
|
+
row.update(status="eval_error", exhausted=None,
|
|
335
|
+
error_kind=type(exc).__name__,
|
|
336
|
+
error_msg=str(exc)[:400])
|
|
337
|
+
else:
|
|
338
|
+
row["steps"] = result.steps
|
|
339
|
+
row["exhausted"] = result.exhausted
|
|
340
|
+
if result.exhausted:
|
|
341
|
+
row["status"] = "exhausted"
|
|
342
|
+
else:
|
|
343
|
+
row["holds"] = result.holds
|
|
344
|
+
row["witness"] = (dict(result.witness)
|
|
345
|
+
if result.witness else None)
|
|
346
|
+
row["ms"] = round((time.perf_counter() - cell_started) * 1000, 4)
|
|
347
|
+
record(row)
|
|
348
|
+
slab.append(row)
|
|
349
|
+
|
|
350
|
+
_flush(results_path, slab)
|
|
351
|
+
if progress:
|
|
352
|
+
progress(def_id, index, len(prepared))
|
|
353
|
+
|
|
354
|
+
return ChemBatchResult(
|
|
355
|
+
rows=tuple(rows), counts=counts, cache_stats=cache.stats(),
|
|
356
|
+
seconds=round(time.perf_counter() - started, 3), skipped=skipped)
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def _flush(results_path: Optional[str], slab: List[dict]) -> None:
|
|
360
|
+
"""Append one definition's rows and fsync-free close.
|
|
361
|
+
|
|
362
|
+
Append mode, one open/close per definition: a campaign run is minutes to
|
|
363
|
+
hours long, and holding one handle open for its whole duration means an
|
|
364
|
+
interrupt loses whatever the OS had buffered.
|
|
365
|
+
"""
|
|
366
|
+
if not results_path or not slab:
|
|
367
|
+
return
|
|
368
|
+
with io.open(results_path, "a", encoding="utf-8", newline="\n") as handle:
|
|
369
|
+
for row in slab:
|
|
370
|
+
handle.write(json.dumps(row, ensure_ascii=False) + "\n")
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
# ---------------------------------------------------------------------------
|
|
374
|
+
# ChemBatchResult.to_markdown() / to_html() — a capped SUMMARY, never a row
|
|
375
|
+
# dump (see both docstrings): display only, no new decision-procedure logic,
|
|
376
|
+
# so no soundness risk. Pure formatting over already-computed rows/counts.
|
|
377
|
+
# ---------------------------------------------------------------------------
|
|
378
|
+
|
|
379
|
+
def _md_cell(value) -> str:
|
|
380
|
+
"""Escape a value for safe embedding in one Markdown table cell.
|
|
381
|
+
|
|
382
|
+
A bare ``|`` would be read as a new column and an embedded newline would
|
|
383
|
+
split the row across lines — both routinely occur in campaign data (a
|
|
384
|
+
SMILES cannot contain either, but an ``error_msg`` free-text string can)
|
|
385
|
+
— so both are neutralised here. Mirrors
|
|
386
|
+
:func:`unicode_logic_kit.eval.theory_check._md_cell` exactly (a stable,
|
|
387
|
+
three-line function); duplicated rather than imported the way
|
|
388
|
+
:mod:`unicode_logic_kit.atp._html` documents duplicating ``esc_html`` —
|
|
389
|
+
see :func:`_esc_html` below for the matching HTML-side duplicate and why
|
|
390
|
+
this module does not import ``atp._html`` itself.
|
|
391
|
+
"""
|
|
392
|
+
text = _cell(value)
|
|
393
|
+
text = text.replace("|", "\\|")
|
|
394
|
+
return text.replace("\r\n", " ").replace("\n", " ").replace("\r", " ")
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def _cell(value) -> str:
|
|
398
|
+
"""None-safe stringification shared by the Markdown and HTML table
|
|
399
|
+
builders: a row field genuinely IS ``None`` for every ``parse_error``
|
|
400
|
+
row (:func:`check_definitions` always sets ``smiles=None`` there), and a
|
|
401
|
+
bare ``str(None)`` would render the misleading literal text ``"None"``
|
|
402
|
+
in the cell rather than an empty one.
|
|
403
|
+
"""
|
|
404
|
+
return "" if value is None else str(value)
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def _gloss_chem_witness(witness: Optional[Mapping[str, Any]]) -> str:
|
|
408
|
+
"""A short, honest gloss of a row's ``witness``: which FOL variable was
|
|
409
|
+
bound to which individual, per :func:`~unicode_logic_kit.semantics.model_eval
|
|
410
|
+
.evaluate_detailed`'s own existential search (``EvalResult.witness``).
|
|
411
|
+
|
|
412
|
+
Deliberately NOT routed through
|
|
413
|
+
:func:`unicode_logic_kit.eval.explain.explain_countermodel`: that
|
|
414
|
+
function's bare-``{name: value}``-dict branch is documented, and its
|
|
415
|
+
generated text is hardcoded, as a Z3 SMT model ("Z3 found a model ...
|
|
416
|
+
the two sides differ") — a chem row's witness never touched Z3 and
|
|
417
|
+
proves a formula IS satisfied by this structure, not that two sides of
|
|
418
|
+
an implication differ, so routing it through that text would misstate
|
|
419
|
+
both the source and the claim. (Contrast
|
|
420
|
+
:mod:`unicode_logic_kit.eval.theory_check`'s own renderer, where a
|
|
421
|
+
``SatisfiabilityResult``/``SubsumptionResult`` witness genuinely IS
|
|
422
|
+
shaped like — and produced by the same backend chain as —
|
|
423
|
+
``Verdict.countermodel``, so reusing ``explain_countermodel`` there is a
|
|
424
|
+
faithful reuse of that function's actual contract.)
|
|
425
|
+
"""
|
|
426
|
+
if not witness:
|
|
427
|
+
return ""
|
|
428
|
+
items = sorted(witness.items(), key=lambda kv: str(kv[0]))
|
|
429
|
+
return "witness " + ", ".join(f"{k}={v}" for k, v in items)
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def _row_detail(row: Mapping[str, Any]) -> str:
|
|
433
|
+
"""The one-line "why" for a non-``ok`` row: its error message, any
|
|
434
|
+
unknown predicates, and a gloss of its witness — whichever are present.
|
|
435
|
+
|
|
436
|
+
Note: :func:`check_definitions` only ever populates ``row["witness"]``
|
|
437
|
+
inside the branch that leaves ``status="ok"`` (mirroring
|
|
438
|
+
:func:`~unicode_logic_kit.semantics.model_eval.evaluate_detailed`'s own
|
|
439
|
+
``EvalResult.witness``, documented "only ever set when ``holds`` is
|
|
440
|
+
``True``"). :meth:`ChemBatchResult.to_markdown`/:meth:`to_html` only ever
|
|
441
|
+
sample non-``ok`` rows (see :func:`_sample_rows`), so on genuine
|
|
442
|
+
``check_definitions`` output the witness-gloss branch below is currently
|
|
443
|
+
unreachable from the sample — this function still handles it (a witness
|
|
444
|
+
on a non-``ok`` row is not a contract violation, just not something
|
|
445
|
+
today's row-construction produces) rather than assuming it can't happen.
|
|
446
|
+
"""
|
|
447
|
+
parts = []
|
|
448
|
+
error_msg = row.get("error_msg")
|
|
449
|
+
if error_msg:
|
|
450
|
+
parts.append(str(error_msg))
|
|
451
|
+
unknown = row.get("unknown_predicates")
|
|
452
|
+
if unknown:
|
|
453
|
+
parts.append("unknown predicates: " + ", ".join(unknown))
|
|
454
|
+
gloss = _gloss_chem_witness(row.get("witness"))
|
|
455
|
+
if gloss:
|
|
456
|
+
parts.append(gloss)
|
|
457
|
+
return "; ".join(parts)
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
def _summary_rows(result: ChemBatchResult) -> List[Tuple[str, str]]:
|
|
461
|
+
rows = [("rows", str(len(result.rows))), ("seconds", str(result.seconds)),
|
|
462
|
+
("skipped", str(result.skipped))]
|
|
463
|
+
for key in sorted(result.cache_stats):
|
|
464
|
+
rows.append((f"cache.{key}", str(result.cache_stats[key])))
|
|
465
|
+
return rows
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def _status_rows(result: ChemBatchResult) -> List[Tuple[str, str]]:
|
|
469
|
+
return [(status, str(result.counts[status]))
|
|
470
|
+
for status in sorted(result.counts)]
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
def _sample_rows(result: ChemBatchResult, max_rows: int) -> Tuple[List[dict], int]:
|
|
474
|
+
"""The first ``max_rows`` non-``ok`` rows (row order preserved, so this
|
|
475
|
+
is deterministic), and the total non-``ok`` count they are a sample of.
|
|
476
|
+
|
|
477
|
+
``max_rows`` is clamped to 0 rather than passed straight into a slice: a
|
|
478
|
+
negative value (e.g. computed by a caller as ``limit - offset``) would
|
|
479
|
+
otherwise hit Python's negative-slice semantics, which drop elements
|
|
480
|
+
from the END of the list instead of capping the sample to zero — the
|
|
481
|
+
opposite of what a cap parameter should do, while still letting the
|
|
482
|
+
closing "showing N of M" line look plausible instead of clearly wrong.
|
|
483
|
+
"""
|
|
484
|
+
non_ok = [row for row in result.rows if row.get("status") != "ok"]
|
|
485
|
+
max_rows = max(0, max_rows)
|
|
486
|
+
return non_ok[:max_rows], len(non_ok)
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def _chem_markdown_lines(result: ChemBatchResult, max_rows: int) -> List[str]:
|
|
490
|
+
lines: List[str] = ["# Chem batch result", ""]
|
|
491
|
+
lines.append(f"**{len(result.rows)}** row(s) in **{result.seconds}s**, "
|
|
492
|
+
f"**{result.skipped}** skipped.")
|
|
493
|
+
lines.append("")
|
|
494
|
+
|
|
495
|
+
lines.append("## Summary")
|
|
496
|
+
lines.append("")
|
|
497
|
+
lines.append("| metric | value |")
|
|
498
|
+
lines.append("|---|---|")
|
|
499
|
+
for metric, value in _summary_rows(result):
|
|
500
|
+
lines.append(f"| {_md_cell(metric)} | {_md_cell(value)} |")
|
|
501
|
+
lines.append("")
|
|
502
|
+
|
|
503
|
+
lines.append("## Status breakdown")
|
|
504
|
+
lines.append("")
|
|
505
|
+
lines.append("| status | count |")
|
|
506
|
+
lines.append("|---|---|")
|
|
507
|
+
for status, count in _status_rows(result):
|
|
508
|
+
lines.append(f"| {_md_cell(status)} | {_md_cell(count)} |")
|
|
509
|
+
lines.append("")
|
|
510
|
+
|
|
511
|
+
lines.append("## Sample of non-ok rows")
|
|
512
|
+
lines.append("")
|
|
513
|
+
sample, total = _sample_rows(result, max_rows)
|
|
514
|
+
lines.append("| def_id | smiles | status | detail |")
|
|
515
|
+
lines.append("|---|---|---|---|")
|
|
516
|
+
for row in sample:
|
|
517
|
+
lines.append(f"| {_md_cell(row.get('def_id'))} | {_md_cell(row.get('smiles'))} "
|
|
518
|
+
f"| {_md_cell(row.get('status'))} | {_md_cell(_row_detail(row))} |")
|
|
519
|
+
lines.append("")
|
|
520
|
+
lines.append(f"showing {len(sample)} of {total} non-ok row(s); full data in "
|
|
521
|
+
"`rows`/the JSONL results file.")
|
|
522
|
+
lines.append("")
|
|
523
|
+
|
|
524
|
+
while len(lines) >= 2 and lines[-1] == "" and lines[-2] == "":
|
|
525
|
+
lines.pop()
|
|
526
|
+
return lines
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
# ---------------------------------------------------------------------------
|
|
530
|
+
# HTML: a small, self-contained page-wrapper local to this module. See
|
|
531
|
+
# _md_cell's docstring — this duplicates unicode_logic_kit.atp._html's
|
|
532
|
+
# esc_html/html_page idiom rather than importing it, because (unlike
|
|
533
|
+
# eval.theory_check, which already imports unicode_logic_kit.atp at module
|
|
534
|
+
# scope) this module does not otherwise reach into atp, and this batch's own
|
|
535
|
+
# working notes ask for a local helper rather than establishing that import
|
|
536
|
+
# direction here.
|
|
537
|
+
# ---------------------------------------------------------------------------
|
|
538
|
+
|
|
539
|
+
def _esc_html(s: str) -> str:
|
|
540
|
+
return s.replace("&", "&").replace("<", "<").replace(">", ">")
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
_CHEM_HTML_CSS = """
|
|
544
|
+
:root{--pg:#faf9f6;--ink:#23232a;--muted:#6b6b76;--bar:#3f3f47;--accent:#1c46b0}
|
|
545
|
+
@media(prefers-color-scheme:dark){:root{--pg:#15151a;--ink:#e9e8e4;--muted:#a6a6b0;--bar:#c3c3cd;--accent:#84a6ff}}
|
|
546
|
+
:root[data-theme=light]{--pg:#faf9f6;--ink:#23232a;--muted:#6b6b76;--bar:#3f3f47;--accent:#1c46b0}
|
|
547
|
+
:root[data-theme=dark]{--pg:#15151a;--ink:#e9e8e4;--muted:#a6a6b0;--bar:#c3c3cd;--accent:#84a6ff}
|
|
548
|
+
body{margin:0;background:var(--pg);color:var(--ink)}
|
|
549
|
+
.rpt{max-width:900px;margin:0 auto;padding:26px 16px;
|
|
550
|
+
font-family:ui-sans-serif,system-ui,"Segoe UI",Arial,sans-serif;
|
|
551
|
+
font-size:14px;line-height:1.5}
|
|
552
|
+
.rpt h1{font-size:20px;margin:0 0 8px}
|
|
553
|
+
.rpt h2{font-size:16px;margin:22px 0 6px;border-bottom:1.3px solid var(--bar);padding-bottom:3px}
|
|
554
|
+
.rpt table{border-collapse:collapse;width:100%;margin:4px 0 10px}
|
|
555
|
+
.rpt th,.rpt td{border:1px solid var(--bar);padding:4px 8px;text-align:left;
|
|
556
|
+
vertical-align:top}
|
|
557
|
+
.rpt th{color:var(--muted);font-weight:600}
|
|
558
|
+
.rpt .note{color:var(--muted);font-size:12.5px}
|
|
559
|
+
"""
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def _html_page(title: str, body_html: str, extra_css: str) -> str:
|
|
563
|
+
"""Wrap ``body_html`` in a self-contained, theme-aware HTML page — the
|
|
564
|
+
same skeleton :meth:`unicode_logic_kit.fol.derivation.CCGDerivation.to_html`
|
|
565
|
+
and :mod:`unicode_logic_kit.atp._html`'s ``html_page`` use."""
|
|
566
|
+
return (
|
|
567
|
+
"<!doctype html>\n<html><head><meta charset=\"utf-8\">\n"
|
|
568
|
+
"<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n"
|
|
569
|
+
"<title>%s</title>\n<style>%s</style></head>\n<body>\n%s\n</body></html>\n"
|
|
570
|
+
% (_esc_html(title), extra_css, body_html)
|
|
571
|
+
)
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
def _html_table(headers: Tuple[str, ...], rows: List[Tuple[str, ...]]) -> str:
|
|
575
|
+
head = "".join("<th>%s</th>" % _esc_html(h) for h in headers)
|
|
576
|
+
body = "".join(
|
|
577
|
+
"<tr>%s</tr>" % "".join("<td>%s</td>" % _esc_html(str(cell)) for cell in row)
|
|
578
|
+
for row in rows
|
|
579
|
+
)
|
|
580
|
+
return "<table><tr>%s</tr>%s</table>" % (head, body)
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def _chem_html_body(result: ChemBatchResult, max_rows: int) -> str:
|
|
584
|
+
parts: List[str] = ['<div class="rpt">', "<h1>Chem batch result</h1>",
|
|
585
|
+
"<p>%d row(s) in %ss, %d skipped.</p>"
|
|
586
|
+
% (len(result.rows), result.seconds, result.skipped)]
|
|
587
|
+
|
|
588
|
+
parts.append("<h2>Summary</h2>")
|
|
589
|
+
parts.append(_html_table(("metric", "value"), _summary_rows(result)))
|
|
590
|
+
|
|
591
|
+
parts.append("<h2>Status breakdown</h2>")
|
|
592
|
+
parts.append(_html_table(("status", "count"), _status_rows(result)))
|
|
593
|
+
|
|
594
|
+
parts.append("<h2>Sample of non-ok rows</h2>")
|
|
595
|
+
sample, total = _sample_rows(result, max_rows)
|
|
596
|
+
table_rows = [
|
|
597
|
+
(_cell(row.get("def_id")), _cell(row.get("smiles")), _cell(row.get("status")),
|
|
598
|
+
_row_detail(row))
|
|
599
|
+
for row in sample
|
|
600
|
+
]
|
|
601
|
+
parts.append(_html_table(("def_id", "smiles", "status", "detail"), table_rows))
|
|
602
|
+
parts.append('<p class="note">showing %d of %d non-ok row(s); full data in '
|
|
603
|
+
"<code>rows</code>/the JSONL results file.</p>" % (len(sample), total))
|
|
604
|
+
|
|
605
|
+
parts.append("</div>")
|
|
606
|
+
return "".join(parts)
|