unicode-logic-kit 0.31.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- unicode_logic_kit/__init__.py +385 -0
- unicode_logic_kit/__main__.py +520 -0
- unicode_logic_kit/_deadline.py +219 -0
- unicode_logic_kit/ace/__init__.py +126 -0
- unicode_logic_kit/ace/_align.py +135 -0
- unicode_logic_kit/ace/chem_lexicon.py +128 -0
- unicode_logic_kit/ace/drs_reader.py +570 -0
- unicode_logic_kit/ace/mapping.py +666 -0
- unicode_logic_kit/ace/reverse_modal.py +138 -0
- unicode_logic_kit/ace/runner.py +551 -0
- unicode_logic_kit/ace/translate.py +452 -0
- unicode_logic_kit/ace/verbalize.py +1070 -0
- unicode_logic_kit/api.py +1284 -0
- unicode_logic_kit/atp/__init__.py +177 -0
- unicode_logic_kit/atp/_ascii_names.py +113 -0
- unicode_logic_kit/atp/_html.py +72 -0
- unicode_logic_kit/atp/_substructural_input.py +228 -0
- unicode_logic_kit/atp/_tff_problem.py +715 -0
- unicode_logic_kit/atp/_tptp_problem.py +1111 -0
- unicode_logic_kit/atp/_writer_support.py +289 -0
- unicode_logic_kit/atp/clingo_backend.py +1180 -0
- unicode_logic_kit/atp/cvc5_backend.py +1385 -0
- unicode_logic_kit/atp/eprover_backend.py +732 -0
- unicode_logic_kit/atp/finite_domain.py +1055 -0
- unicode_logic_kit/atp/fitch.py +1547 -0
- unicode_logic_kit/atp/fitch_search.py +551 -0
- unicode_logic_kit/atp/hets_backend.py +339 -0
- unicode_logic_kit/atp/hybrid_down.py +120 -0
- unicode_logic_kit/atp/incremental.py +250 -0
- unicode_logic_kit/atp/kripke_enum.py +741 -0
- unicode_logic_kit/atp/lambek.py +436 -0
- unicode_logic_kit/atp/leo3_backend.py +332 -0
- unicode_logic_kit/atp/linear.py +738 -0
- unicode_logic_kit/atp/lj.py +705 -0
- unicode_logic_kit/atp/logic_backends.py +566 -0
- unicode_logic_kit/atp/ltl_tableau.py +1084 -0
- unicode_logic_kit/atp/minizinc_backend.py +1402 -0
- unicode_logic_kit/atp/modal_tableau.py +1382 -0
- unicode_logic_kit/atp/nanocop_backend.py +410 -0
- unicode_logic_kit/atp/portfolio.py +489 -0
- unicode_logic_kit/atp/protocol.py +1803 -0
- unicode_logic_kit/atp/prover9_entailment.py +1153 -0
- unicode_logic_kit/atp/resolution.py +1376 -0
- unicode_logic_kit/atp/resolution_check.py +1114 -0
- unicode_logic_kit/atp/sequent.py +1050 -0
- unicode_logic_kit/atp/tableau.py +921 -0
- unicode_logic_kit/atp/tableau_check.py +543 -0
- unicode_logic_kit/atp/tptp_ncl.py +811 -0
- unicode_logic_kit/atp/tptp_tff.py +1546 -0
- unicode_logic_kit/atp/tstp.py +1333 -0
- unicode_logic_kit/atp/tstp_check.py +1096 -0
- unicode_logic_kit/atp/twee_backend.py +236 -0
- unicode_logic_kit/atp/twee_check.py +711 -0
- unicode_logic_kit/atp/twee_entailment.py +953 -0
- unicode_logic_kit/atp/vampire_entailment.py +540 -0
- unicode_logic_kit/atp/z3_arith.py +470 -0
- unicode_logic_kit/atp/z3_equivalence.py +36 -0
- unicode_logic_kit/atp/z3_fuzzy.py +362 -0
- unicode_logic_kit/atp/z3_input.py +500 -0
- unicode_logic_kit/atp/z3_models.py +208 -0
- unicode_logic_kit/chem/__init__.py +88 -0
- unicode_logic_kit/chem/_naming.py +284 -0
- unicode_logic_kit/chem/cache.py +185 -0
- unicode_logic_kit/chem/interop.py +244 -0
- unicode_logic_kit/chem/mol.py +525 -0
- unicode_logic_kit/chem/signature.py +112 -0
- unicode_logic_kit/comorphism.py +497 -0
- unicode_logic_kit/dl/__init__.py +384 -0
- unicode_logic_kit/dl/classification.py +227 -0
- unicode_logic_kit/dl/concepts.py +632 -0
- unicode_logic_kit/dl/datatypes.py +818 -0
- unicode_logic_kit/dl/owl_functional.py +2433 -0
- unicode_logic_kit/dl/owl_manchester.py +1637 -0
- unicode_logic_kit/dl/owl_reasoner.py +790 -0
- unicode_logic_kit/dl/parser.py +391 -0
- unicode_logic_kit/dl/tableau.py +4048 -0
- unicode_logic_kit/dl/translate.py +2704 -0
- unicode_logic_kit/drt/__init__.py +94 -0
- unicode_logic_kit/drt/export.py +179 -0
- unicode_logic_kit/drt/nodes.py +506 -0
- unicode_logic_kit/drt/parser.py +965 -0
- unicode_logic_kit/drt/resolve.py +195 -0
- unicode_logic_kit/drt/reverse.py +175 -0
- unicode_logic_kit/eval/__init__.py +106 -0
- unicode_logic_kit/eval/batch.py +382 -0
- unicode_logic_kit/eval/canonical.py +663 -0
- unicode_logic_kit/eval/chem_batch.py +606 -0
- unicode_logic_kit/eval/converses.py +200 -0
- unicode_logic_kit/eval/datasets/__init__.py +136 -0
- unicode_logic_kit/eval/datasets/_base.py +263 -0
- unicode_logic_kit/eval/datasets/_proofwriter_proof.py +422 -0
- unicode_logic_kit/eval/datasets/c3po.py +678 -0
- unicode_logic_kit/eval/datasets/folio.py +158 -0
- unicode_logic_kit/eval/datasets/fracas.py +418 -0
- unicode_logic_kit/eval/datasets/groves.py +191 -0
- unicode_logic_kit/eval/datasets/logicbench.py +467 -0
- unicode_logic_kit/eval/datasets/logicnli.py +303 -0
- unicode_logic_kit/eval/datasets/malls.py +133 -0
- unicode_logic_kit/eval/datasets/pfolio.py +594 -0
- unicode_logic_kit/eval/datasets/pmb.py +242 -0
- unicode_logic_kit/eval/datasets/prontoqa.py +611 -0
- unicode_logic_kit/eval/datasets/proofwriter.py +1431 -0
- unicode_logic_kit/eval/datasets/proverqa.py +674 -0
- unicode_logic_kit/eval/datasets/willow.py +478 -0
- unicode_logic_kit/eval/equivalence.py +466 -0
- unicode_logic_kit/eval/exercise_gen.py +533 -0
- unicode_logic_kit/eval/explain.py +791 -0
- unicode_logic_kit/eval/generality.py +750 -0
- unicode_logic_kit/eval/metric_hf.py +458 -0
- unicode_logic_kit/eval/predicate_match.py +343 -0
- unicode_logic_kit/eval/theory_check.py +1170 -0
- unicode_logic_kit/eval/validate.py +306 -0
- unicode_logic_kit/fol/__init__.py +177 -0
- unicode_logic_kit/fol/_atom_keys.py +510 -0
- unicode_logic_kit/fol/_fol_nodes.py +3586 -0
- unicode_logic_kit/fol/_free_parameters.py +105 -0
- unicode_logic_kit/fol/_ho_nodes.py +448 -0
- unicode_logic_kit/fol/_hybrid_nodes.py +308 -0
- unicode_logic_kit/fol/_identifiers.py +1091 -0
- unicode_logic_kit/fol/_lambek_nodes.py +112 -0
- unicode_logic_kit/fol/_linear_nodes.py +352 -0
- unicode_logic_kit/fol/_modal_nodes.py +1467 -0
- unicode_logic_kit/fol/_msfl_nodes.py +2196 -0
- unicode_logic_kit/fol/_numeral_symbols.py +231 -0
- unicode_logic_kit/fol/_so_nodes.py +200 -0
- unicode_logic_kit/fol/_symbol_names.py +81 -0
- unicode_logic_kit/fol/_team_nodes.py +181 -0
- unicode_logic_kit/fol/_tptp_symbols.py +551 -0
- unicode_logic_kit/fol/_truth_constants.py +117 -0
- unicode_logic_kit/fol/casl_export.py +1135 -0
- unicode_logic_kit/fol/casl_import.py +929 -0
- unicode_logic_kit/fol/derivation.py +367 -0
- unicode_logic_kit/fol/dialect_detect.py +70 -0
- unicode_logic_kit/fol/dialect_repair.py +537 -0
- unicode_logic_kit/fol/frames.py +637 -0
- unicode_logic_kit/fol/grammars/terminals.lark +31 -0
- unicode_logic_kit/fol/lambda_tools.py +297 -0
- unicode_logic_kit/fol/latex_input.py +429 -0
- unicode_logic_kit/fol/modal_translation.py +944 -0
- unicode_logic_kit/fol/msflparser.py +1033 -0
- unicode_logic_kit/fol/naming.py +422 -0
- unicode_logic_kit/fol/nodes.py +241 -0
- unicode_logic_kit/fol/normalforms.py +492 -0
- unicode_logic_kit/fol/pal.py +287 -0
- unicode_logic_kit/fol/prolog_export.py +566 -0
- unicode_logic_kit/fol/prolog_input.py +505 -0
- unicode_logic_kit/fol/prover9_input.py +1325 -0
- unicode_logic_kit/fol/qml.py +1760 -0
- unicode_logic_kit/fol/qmltp_input.py +525 -0
- unicode_logic_kit/fol/sanitize.py +221 -0
- unicode_logic_kit/fol/serialize.py +79 -0
- unicode_logic_kit/fol/signature.py +1290 -0
- unicode_logic_kit/fol/simplify_check.py +544 -0
- unicode_logic_kit/fol/spans.py +594 -0
- unicode_logic_kit/fol/tptp_input.py +1503 -0
- unicode_logic_kit/fol/tptp_repair.py +941 -0
- unicode_logic_kit/fol/unification.py +157 -0
- unicode_logic_kit/fol/verbalize.py +263 -0
- unicode_logic_kit/hets/__init__.py +163 -0
- unicode_logic_kit/hets/bridge.py +142 -0
- unicode_logic_kit/hets/client.py +748 -0
- unicode_logic_kit/hets/docker.py +420 -0
- unicode_logic_kit/hets/dol.py +712 -0
- unicode_logic_kit/hets/haskell_json.py +355 -0
- unicode_logic_kit/hets/owl_backend.py +794 -0
- unicode_logic_kit/hets/owl_cli.py +598 -0
- unicode_logic_kit/hets/symbols.py +512 -0
- unicode_logic_kit/hol/__init__.py +140 -0
- unicode_logic_kit/hol/_ho_common.py +323 -0
- unicode_logic_kit/hol/_isabelle_binders.py +125 -0
- unicode_logic_kit/hol/classical.py +812 -0
- unicode_logic_kit/hol/deepshallow/__init__.py +45 -0
- unicode_logic_kit/hol/deepshallow/_common.py +177 -0
- unicode_logic_kit/hol/deepshallow/conditional.py +225 -0
- unicode_logic_kit/hol/deepshallow/intuitionistic.py +181 -0
- unicode_logic_kit/hol/deepshallow/modal.py +217 -0
- unicode_logic_kit/hol/deepshallow/qml.py +406 -0
- unicode_logic_kit/hol/deepshallow/relevant.py +206 -0
- unicode_logic_kit/hol/free.py +753 -0
- unicode_logic_kit/hol/goedel.py +336 -0
- unicode_logic_kit/hol/ho_modal.py +1743 -0
- unicode_logic_kit/hol/intuitionistic.py +403 -0
- unicode_logic_kit/hol/isabelle_conditional.py +593 -0
- unicode_logic_kit/hol/isabelle_modal.py +1908 -0
- unicode_logic_kit/hol/isabelle_relevant.py +412 -0
- unicode_logic_kit/hol/isabelle_runner.py +1147 -0
- unicode_logic_kit/hol/isabelle_substructural.py +884 -0
- unicode_logic_kit/hol/lean.py +1018 -0
- unicode_logic_kit/hol/manyvalued.py +921 -0
- unicode_logic_kit/hol/secondorder.py +687 -0
- unicode_logic_kit/hol/thf_modal.py +941 -0
- unicode_logic_kit/hol/thirdorder.py +397 -0
- unicode_logic_kit/ilp/__init__.py +89 -0
- unicode_logic_kit/ilp/readback.py +389 -0
- unicode_logic_kit/ilp/separation.py +153 -0
- unicode_logic_kit/ilp/task.py +730 -0
- unicode_logic_kit/logic.py +163 -0
- unicode_logic_kit/mcp/__init__.py +28 -0
- unicode_logic_kit/mcp/__main__.py +5 -0
- unicode_logic_kit/mcp/chem_tools.py +1031 -0
- unicode_logic_kit/mcp/server.py +2453 -0
- unicode_logic_kit/mcp/syntax_spec.py +681 -0
- unicode_logic_kit/prob/__init__.py +53 -0
- unicode_logic_kit/prob/_bdd.py +225 -0
- unicode_logic_kit/prob/_column_gen.py +668 -0
- unicode_logic_kit/prob/distribution.py +686 -0
- unicode_logic_kit/prob/nilsson.py +470 -0
- unicode_logic_kit/py.typed +0 -0
- unicode_logic_kit/semantics/__init__.py +137 -0
- unicode_logic_kit/semantics/_modal_reject.py +156 -0
- unicode_logic_kit/semantics/action_models.py +466 -0
- unicode_logic_kit/semantics/asp_models.py +1200 -0
- unicode_logic_kit/semantics/conditional.py +580 -0
- unicode_logic_kit/semantics/dynamic_epistemic.py +95 -0
- unicode_logic_kit/semantics/free_logic.py +913 -0
- unicode_logic_kit/semantics/fuzzy.py +384 -0
- unicode_logic_kit/semantics/fuzzy_kripke.py +442 -0
- unicode_logic_kit/semantics/intuitionistic.py +581 -0
- unicode_logic_kit/semantics/kripke.py +1139 -0
- unicode_logic_kit/semantics/manyvalued.py +580 -0
- unicode_logic_kit/semantics/matrix.py +342 -0
- unicode_logic_kit/semantics/model_eval.py +1135 -0
- unicode_logic_kit/semantics/modelfinder.py +1036 -0
- unicode_logic_kit/semantics/nonmonotonic.py +372 -0
- unicode_logic_kit/semantics/relevant.py +331 -0
- unicode_logic_kit/semantics/secondorder.py +657 -0
- unicode_logic_kit/semantics/structures.py +352 -0
- unicode_logic_kit/semantics/tarski.py +975 -0
- unicode_logic_kit/semantics/team.py +315 -0
- unicode_logic_kit/semantics/team_translation.py +416 -0
- unicode_logic_kit/semantics/thirdorder.py +358 -0
- unicode_logic_kit/semantics/tnorm.py +85 -0
- unicode_logic_kit/semantics/truthtable.py +201 -0
- unicode_logic_kit-0.31.0.dist-info/METADATA +333 -0
- unicode_logic_kit-0.31.0.dist-info/RECORD +237 -0
- unicode_logic_kit-0.31.0.dist-info/WHEEL +4 -0
- unicode_logic_kit-0.31.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,355 @@
|
|
|
1
|
+
r"""Repairing Haskell ``show`` escapes that leak into HETS' JSON responses.
|
|
2
|
+
|
|
3
|
+
The defect, exactly
|
|
4
|
+
-------------------
|
|
5
|
+
HETS 0.108.0's ``GET /dg/<iri>?format=json`` serialises OWL axiom strings by
|
|
6
|
+
applying Haskell's ``show`` to a ``String`` whose ``Char``\ s are the UTF-8
|
|
7
|
+
BYTES of the intended text (the classic latin1-decode of a UTF-8 input). GHC's
|
|
8
|
+
``showLitChar`` emits a DECIMAL escape for every code point above 127, so the
|
|
9
|
+
body that reaches a client contains sequences such as::
|
|
10
|
+
|
|
11
|
+
AnnotationAssertion( obo:IAO_0000112 obo:BFO_0000001 "Verdi\226\128\153s Requiem"@en )
|
|
12
|
+
|
|
13
|
+
``\226`` is not a JSON escape, so :func:`json.loads` refuses the whole 8.3 MB
|
|
14
|
+
document with ``Invalid \escape``, and every endpoint that needs the
|
|
15
|
+
development graph — including every ``hets:`` comorphism edge, which resolves
|
|
16
|
+
its node through :meth:`~unicode_logic_kit.hets.client.HetsClient.dg` — fails on
|
|
17
|
+
any library carrying one non-ASCII annotation.
|
|
18
|
+
|
|
19
|
+
Why this is lossless recovery and not an approximation
|
|
20
|
+
------------------------------------------------------
|
|
21
|
+
The emitter is known. GHC's ``showLitChar`` is::
|
|
22
|
+
|
|
23
|
+
c > '\DEL' -> '\\' : show (ord c) -- DECIMAL, with \& if a digit follows
|
|
24
|
+
c == '\DEL' -> "\\DEL"
|
|
25
|
+
c == '\\' -> "\\\\"
|
|
26
|
+
c >= ' ' -> the character itself
|
|
27
|
+
c in 7..13 -> "\\a" "\\b" "\\t" "\\n" "\\v" "\\f" "\\r"
|
|
28
|
+
otherwise -> '\\' : asciiTab !! ord c -- "\\NUL" .. "\\US", plus "\\SP"
|
|
29
|
+
|
|
30
|
+
and ``showLitString`` adds ``\"`` for a double quote. So the complete set of
|
|
31
|
+
backslash sequences a Haskell-``show``\ n string can contain is finite and
|
|
32
|
+
enumerable, and exactly five of its members are not also JSON escapes: the
|
|
33
|
+
decimal runs, the empty-string separator ``\&``, the three-letter mnemonics,
|
|
34
|
+
``\a`` and ``\v``. Everything else (``\" \\ \n \t \b \f \r``) coincides with
|
|
35
|
+
JSON and is left untouched.
|
|
36
|
+
|
|
37
|
+
The census of the real 8,345,206-character body confirms the emitter: of its
|
|
38
|
+
backslash sequences, ``\"`` occurs 17618 times, ``\n`` 8028, ``\ddd`` 897,
|
|
39
|
+
``\\`` 354, ``\t`` 29 and ``\&`` 7 — and nothing else. The 897 decimal
|
|
40
|
+
escapes form 389 maximal runs, every value lies in 128..226 (i.e. every one is
|
|
41
|
+
a byte, never a code point: ``show`` emits decimal only above 127, so a value
|
|
42
|
+
below 128 cannot occur), and all 389 runs decode as STRICT UTF-8 with zero
|
|
43
|
+
failures. All seven ``\&`` sit in the one place Haskell's lexer needs a
|
|
44
|
+
separator, ``\194\167\&7`` — without it ``\1677`` would lex as a single
|
|
45
|
+
escape.
|
|
46
|
+
|
|
47
|
+
The one judgment call: a decimal run is read as UTF-8 BYTES first
|
|
48
|
+
-----------------------------------------------------------------
|
|
49
|
+
For a maximal run of values ``v1..vk``:
|
|
50
|
+
|
|
51
|
+
1. if every ``vi < 256`` and ``bytes(v1..vk)`` decodes as strict UTF-8, the
|
|
52
|
+
run's text is that decoding (``\226\128\153`` -> ``’``);
|
|
53
|
+
2. else, if every ``vi`` is a Unicode scalar value, the run's text is
|
|
54
|
+
``"".join(chr(vi))`` — the reading a well-formed Haskell ``String`` would
|
|
55
|
+
mean (``show "\8594" == "\8594"``, and 8594 cannot be a byte);
|
|
56
|
+
3. else :class:`HaskellJsonRepairError` is raised, naming the escape, its
|
|
57
|
+
value and its offset.
|
|
58
|
+
|
|
59
|
+
Rule 1 before rule 2 is an ASSUMPTION ABOUT HETS 0.108.0, not a fact about
|
|
60
|
+
Haskell: a HETS build whose strings were NOT latin1-mangled would have a
|
|
61
|
+
genuine two-character ``é`` (``\195\169``) read here as ``é``. The evidence
|
|
62
|
+
is unanimous for the byte reading — ``Verdi’s``, ``§7 Absatz 3 ROG``, and all
|
|
63
|
+
389 runs valid UTF-8 — and the opposite order would mangle the real data, so
|
|
64
|
+
this is the right default; it is stated here as an assumption so a future
|
|
65
|
+
HETS whose output is clean can be spotted by the census this module reports.
|
|
66
|
+
|
|
67
|
+
For a lone byte in 0x80..0xFF that is NOT valid UTF-8 the two readings
|
|
68
|
+
coincide (``\233`` -> ``é`` either way), so rule 2 is never a guess in that
|
|
69
|
+
case. ``errors="replace"`` is deliberately NOT used anywhere: a U+FFFD in
|
|
70
|
+
place of a character this module could have recovered is exactly the silent
|
|
71
|
+
approximation this kit refuses, so an unrecoverable run raises instead.
|
|
72
|
+
|
|
73
|
+
What this module will NOT do
|
|
74
|
+
----------------------------
|
|
75
|
+
It will not make invalid JSON valid by guessing. The scan is STRING-AWARE: a
|
|
76
|
+
backslash outside a string literal is copied verbatim (so ``{\ "a": 1}`` still
|
|
77
|
+
fails), escaped backslashes are consumed pairwise (so the legitimate JSON
|
|
78
|
+
``"x\\226y"`` — a literal backslash followed by the text ``226`` — is left
|
|
79
|
+
alone), and ``\`` followed by anything this module does not recognise (e.g.
|
|
80
|
+
``\q``) is copied verbatim so :func:`json.loads` raises its own
|
|
81
|
+
``Invalid \escape`` as before. Re-emission goes through
|
|
82
|
+
``json.dumps(chunk)[1:-1]``, so the standard library does the escaping and a
|
|
83
|
+
decoded ``"``, ``\`` or control character cannot break the document.
|
|
84
|
+
|
|
85
|
+
:meth:`~unicode_logic_kit.hets.client.HetsClient.dg` calls this only on the
|
|
86
|
+
FAILURE path — ``json.loads`` first, repair only when it raises — so a body
|
|
87
|
+
the standard library already accepts is never touched at all, and that
|
|
88
|
+
identity is structural rather than argued.
|
|
89
|
+
|
|
90
|
+
This module touches nothing in :mod:`unicode_logic_kit.dl`, adds no OWL
|
|
91
|
+
construct and has no effect on the description-logic tableau. Its relevance to
|
|
92
|
+
the kit's two-route rule is indirect but total: until it exists, the kit cannot
|
|
93
|
+
read HETS' axiom list for any ontology with a non-ASCII annotation, so the
|
|
94
|
+
HETS FOL image cannot be cross-checked against
|
|
95
|
+
:func:`unicode_logic_kit.dl.tbox_to_fol` / :func:`unicode_logic_kit.dl.kb_to_fol`
|
|
96
|
+
at all. It changes no verdict on its own.
|
|
97
|
+
"""
|
|
98
|
+
|
|
99
|
+
from __future__ import annotations
|
|
100
|
+
|
|
101
|
+
import json
|
|
102
|
+
import re
|
|
103
|
+
from dataclasses import dataclass
|
|
104
|
+
|
|
105
|
+
__all__ = [
|
|
106
|
+
"HaskellJsonRepair",
|
|
107
|
+
"HaskellJsonRepairError",
|
|
108
|
+
"repair_haskell_json",
|
|
109
|
+
]
|
|
110
|
+
|
|
111
|
+
#: GHC's ``asciiTab``: the mnemonic ``showLitChar`` emits for code points
|
|
112
|
+
#: 0..32, plus ``DEL``. Indices 7..13 are never emitted by ``show`` (it
|
|
113
|
+
#: prefers the single-letter ``\a \b \t \n \v \f \r``) but Haskell's lexer
|
|
114
|
+
#: accepts them, so they are accepted here too.
|
|
115
|
+
_ASCII_TAB = (
|
|
116
|
+
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
|
|
117
|
+
"BS", "HT", "LF", "VT", "FF", "CR", "SO", "SI",
|
|
118
|
+
"DLE", "DC1", "DC2", "DC3", "DC4", "NAK", "SYN", "ETB",
|
|
119
|
+
"CAN", "EM", "SUB", "ESC", "FS", "GS", "RS", "US",
|
|
120
|
+
"SP",
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
#: Mnemonic -> character, longest first so ``\SOH`` is matched before ``\SO``
|
|
124
|
+
#: (the very reason GHC emits ``\SO\&H`` for a ``\SO`` followed by an ``H``).
|
|
125
|
+
#: ``\a`` and ``\v`` are included because JSON has no escape for either;
|
|
126
|
+
#: ``\b \f \n \r \t`` are deliberately ABSENT — they are valid JSON already
|
|
127
|
+
#: and are copied verbatim, which is what keeps the identity invariant exact.
|
|
128
|
+
_MNEMONICS = {
|
|
129
|
+
**{name: chr(code) for code, name in enumerate(_ASCII_TAB)},
|
|
130
|
+
"DEL": "\x7f",
|
|
131
|
+
"a": "\a",
|
|
132
|
+
"v": "\v",
|
|
133
|
+
}
|
|
134
|
+
_MNEMONIC_NAMES = tuple(sorted(_MNEMONICS, key=len, reverse=True))
|
|
135
|
+
|
|
136
|
+
#: Only ``"`` and ``\`` can change the scanner's state, so the walk jumps
|
|
137
|
+
#: between them with one compiled ``search`` instead of stepping character by
|
|
138
|
+
#: character. Measured on the real 8.3 MB body: 0.79 s per-character against
|
|
139
|
+
#: 0.02 s for the equivalent skipping walk.
|
|
140
|
+
_SPECIAL = re.compile(r'["\\]')
|
|
141
|
+
|
|
142
|
+
_DIGITS = "0123456789"
|
|
143
|
+
|
|
144
|
+
#: The highest Unicode scalar value; a decimal escape above it, or inside the
|
|
145
|
+
#: surrogate range, is not a character and is refused rather than guessed.
|
|
146
|
+
_MAX_SCALAR = 0x10FFFF
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
class HaskellJsonRepairError(RuntimeError):
|
|
150
|
+
"""A Haskell escape that cannot be read as any character.
|
|
151
|
+
|
|
152
|
+
Raised instead of substituting U+FFFD or dropping the escape: a decimal
|
|
153
|
+
value above U+10FFFF, or inside the surrogate range D800..DFFF, is not a
|
|
154
|
+
Unicode scalar value, and guessing what the server meant would be the
|
|
155
|
+
silent approximation this kit refuses.
|
|
156
|
+
"""
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
@dataclass(frozen=True)
|
|
160
|
+
class HaskellJsonRepair:
|
|
161
|
+
"""The repaired JSON text plus a census of what had to be repaired.
|
|
162
|
+
|
|
163
|
+
Fields:
|
|
164
|
+
|
|
165
|
+
* ``text`` — the repaired JSON document. Equal to the input, character for
|
|
166
|
+
character, when nothing was repaired.
|
|
167
|
+
* ``decimal_escapes`` — how many ``\\ddd`` escapes were decoded.
|
|
168
|
+
* ``decimal_runs`` — how many MAXIMAL runs those escapes formed. A run is
|
|
169
|
+
what gets decoded as UTF-8, so this is the count that matters for
|
|
170
|
+
the byte-first rule: 897 escapes in 389 runs on the real body.
|
|
171
|
+
* ``empty_separators`` — how many ``\\&`` (Haskell's empty string) were
|
|
172
|
+
removed.
|
|
173
|
+
* ``mnemonic_escapes`` — how many ``\\NUL``/``\\ESC``/``\\a``/``\\v``/... were
|
|
174
|
+
decoded.
|
|
175
|
+
"""
|
|
176
|
+
|
|
177
|
+
text: str
|
|
178
|
+
decimal_escapes: int = 0
|
|
179
|
+
decimal_runs: int = 0
|
|
180
|
+
empty_separators: int = 0
|
|
181
|
+
mnemonic_escapes: int = 0
|
|
182
|
+
|
|
183
|
+
def __bool__(self) -> bool:
|
|
184
|
+
"""True iff anything at all was repaired.
|
|
185
|
+
|
|
186
|
+
A falsy repair means :attr:`text` IS the input, so a caller can
|
|
187
|
+
re-raise the original :func:`json.loads` error against the original
|
|
188
|
+
body and keep its wording unchanged.
|
|
189
|
+
"""
|
|
190
|
+
return bool(self.decimal_escapes or self.empty_separators
|
|
191
|
+
or self.mnemonic_escapes)
|
|
192
|
+
|
|
193
|
+
def summary(self) -> str:
|
|
194
|
+
"""One-line census, for an error message or a log."""
|
|
195
|
+
return (f"{self.decimal_escapes} decimal escape(s) in "
|
|
196
|
+
f"{self.decimal_runs} run(s), "
|
|
197
|
+
f"{self.empty_separators} \\& separator(s), "
|
|
198
|
+
f"{self.mnemonic_escapes} mnemonic escape(s)")
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
#: How many significant decimal digits a Unicode scalar value can have
|
|
202
|
+
#: (``len("1114111")``): an escape with more is above U+10FFFF whatever its
|
|
203
|
+
#: digits are.
|
|
204
|
+
_MAX_SCALAR_DIGITS = len(str(_MAX_SCALAR))
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _escape_value(digits: str, offset: int, body: str) -> int:
|
|
208
|
+
"""The value of one decimal escape, from its ``digits``.
|
|
209
|
+
|
|
210
|
+
Leading zeros are not significant — Haskell's lexer reads ``\\0065`` as
|
|
211
|
+
65, and so does this. An escape with more significant digits than
|
|
212
|
+
U+10FFFF has is refused HERE, by name, before :func:`int` is asked to read
|
|
213
|
+
it: CPython's own digit limit would otherwise answer a 5 000-digit escape
|
|
214
|
+
with a bare ``ValueError`` ("Exceeds the limit (4300 digits)") that names
|
|
215
|
+
neither the escape nor its offset and is not a
|
|
216
|
+
:class:`HaskellJsonRepairError`, so a caller handling the module's own
|
|
217
|
+
refusal would be bypassed (and, with the limit lifted, the message would
|
|
218
|
+
carry all 5 000 digits).
|
|
219
|
+
"""
|
|
220
|
+
significant = digits.lstrip("0")
|
|
221
|
+
if len(significant) > _MAX_SCALAR_DIGITS:
|
|
222
|
+
line = body.count("\n", 0, offset) + 1
|
|
223
|
+
raise HaskellJsonRepairError(
|
|
224
|
+
f"hets: the Haskell escape \\{significant[:_MAX_SCALAR_DIGITS]}... "
|
|
225
|
+
f"({len(significant)} significant digits) at offset {offset} "
|
|
226
|
+
f"(line {line}) is not a Unicode scalar value (above U+10FFFF), "
|
|
227
|
+
"so it cannot be a character. This module refuses to guess: it "
|
|
228
|
+
"will not substitute U+FFFD and it will not drop the escape. "
|
|
229
|
+
"Fetch the body with HetsClient.dg_raw() and inspect it, or "
|
|
230
|
+
"report it to the HETS server's maintainers.")
|
|
231
|
+
return int(significant or "0")
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _decode_run(values, offsets, body: str) -> str:
|
|
235
|
+
"""Read one maximal decimal-escape run as text (rules 1-3 above)."""
|
|
236
|
+
if all(value < 256 for value in values):
|
|
237
|
+
try:
|
|
238
|
+
return bytes(values).decode("utf-8")
|
|
239
|
+
except UnicodeDecodeError:
|
|
240
|
+
pass
|
|
241
|
+
for value, offset in zip(values, offsets):
|
|
242
|
+
if value > _MAX_SCALAR or 0xD800 <= value <= 0xDFFF:
|
|
243
|
+
line = body.count("\n", 0, offset) + 1
|
|
244
|
+
why = "a surrogate" if value <= _MAX_SCALAR else "above U+10FFFF"
|
|
245
|
+
run = " ".join("\\" + str(v) for v in values)
|
|
246
|
+
raise HaskellJsonRepairError(
|
|
247
|
+
f"hets: the Haskell escape \\{value} at offset {offset} "
|
|
248
|
+
f"(line {line}) is not a Unicode scalar value ({why}), so it "
|
|
249
|
+
f"cannot be a character. The run it belongs to is: {run}. "
|
|
250
|
+
"This module refuses to guess: it will not substitute U+FFFD "
|
|
251
|
+
"and it will not drop the escape. Fetch the body with "
|
|
252
|
+
"HetsClient.dg_raw() and inspect it, or report it to the "
|
|
253
|
+
"HETS server's maintainers.")
|
|
254
|
+
return "".join(chr(value) for value in values)
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def repair_haskell_json(body: str) -> HaskellJsonRepair:
|
|
258
|
+
r"""Turn HETS' Haskell-``show`` escapes into JSON ones.
|
|
259
|
+
|
|
260
|
+
Handles exactly four things, and only inside a string literal: a decimal
|
|
261
|
+
escape run, ``\&``, a Haskell mnemonic, and nothing else — ``\`` followed
|
|
262
|
+
by any other character (including the JSON-legal ``\" \\ \n \t \b \f \r
|
|
263
|
+
\/ \uXXXX``) is copied verbatim. A backslash OUTSIDE a string literal is
|
|
264
|
+
copied verbatim too, so invalid JSON stays invalid: this function never
|
|
265
|
+
makes a document parse by guessing.
|
|
266
|
+
|
|
267
|
+
Args:
|
|
268
|
+
body: the response body, exactly as the server sent it.
|
|
269
|
+
|
|
270
|
+
Returns:
|
|
271
|
+
A :class:`HaskellJsonRepair`. It is falsy, with ``.text`` equal to
|
|
272
|
+
``body``, when there was nothing to repair.
|
|
273
|
+
|
|
274
|
+
Raises:
|
|
275
|
+
HaskellJsonRepairError: a decimal escape is not a Unicode scalar
|
|
276
|
+
value. Never raised for anything a Haskell ``show`` can emit.
|
|
277
|
+
"""
|
|
278
|
+
out = []
|
|
279
|
+
index = 0
|
|
280
|
+
length = len(body)
|
|
281
|
+
in_string = False
|
|
282
|
+
decimal_escapes = 0
|
|
283
|
+
decimal_runs = 0
|
|
284
|
+
empty_separators = 0
|
|
285
|
+
mnemonic_escapes = 0
|
|
286
|
+
|
|
287
|
+
while index < length:
|
|
288
|
+
match = _SPECIAL.search(body, index)
|
|
289
|
+
if match is None:
|
|
290
|
+
out.append(body[index:])
|
|
291
|
+
break
|
|
292
|
+
start = match.start()
|
|
293
|
+
if start > index:
|
|
294
|
+
out.append(body[index:start])
|
|
295
|
+
if body[start] == '"':
|
|
296
|
+
out.append('"')
|
|
297
|
+
in_string = not in_string
|
|
298
|
+
index = start + 1
|
|
299
|
+
continue
|
|
300
|
+
# A backslash. Outside a string it is not an escape at all, and
|
|
301
|
+
# touching it is how a repair turns broken JSON into a guess.
|
|
302
|
+
if not in_string:
|
|
303
|
+
out.append("\\")
|
|
304
|
+
index = start + 1
|
|
305
|
+
continue
|
|
306
|
+
nxt = body[start + 1] if start + 1 < length else ""
|
|
307
|
+
if nxt == "&":
|
|
308
|
+
empty_separators += 1
|
|
309
|
+
index = start + 2
|
|
310
|
+
continue
|
|
311
|
+
if nxt and nxt in _DIGITS:
|
|
312
|
+
values = []
|
|
313
|
+
offsets = []
|
|
314
|
+
cursor = start
|
|
315
|
+
while cursor + 1 < length and body[cursor] == "\\":
|
|
316
|
+
after = body[cursor + 1]
|
|
317
|
+
if after == "&":
|
|
318
|
+
empty_separators += 1
|
|
319
|
+
cursor += 2
|
|
320
|
+
continue
|
|
321
|
+
if after not in _DIGITS:
|
|
322
|
+
break
|
|
323
|
+
end = cursor + 1
|
|
324
|
+
while end < length and body[end] in _DIGITS:
|
|
325
|
+
end += 1
|
|
326
|
+
values.append(_escape_value(body[cursor + 1:end], cursor, body))
|
|
327
|
+
offsets.append(cursor)
|
|
328
|
+
cursor = end
|
|
329
|
+
decimal_escapes += len(values)
|
|
330
|
+
decimal_runs += 1
|
|
331
|
+
out.append(json.dumps(_decode_run(values, offsets, body))[1:-1])
|
|
332
|
+
index = cursor
|
|
333
|
+
continue
|
|
334
|
+
for name in _MNEMONIC_NAMES:
|
|
335
|
+
if body.startswith(name, start + 1):
|
|
336
|
+
mnemonic_escapes += 1
|
|
337
|
+
out.append(json.dumps(_MNEMONICS[name])[1:-1])
|
|
338
|
+
index = start + 1 + len(name)
|
|
339
|
+
break
|
|
340
|
+
else:
|
|
341
|
+
# Verbatim: the JSON-legal escapes (\" \\ \n \t \b \f \r \/
|
|
342
|
+
# \uXXXX) and anything unrecognised alike. Consuming the escaped
|
|
343
|
+
# character too is what makes `\\226` a literal backslash
|
|
344
|
+
# followed by the text 226, not a decimal escape.
|
|
345
|
+
out.append("\\" + nxt)
|
|
346
|
+
index = start + 2
|
|
347
|
+
continue
|
|
348
|
+
|
|
349
|
+
return HaskellJsonRepair(
|
|
350
|
+
text="".join(out),
|
|
351
|
+
decimal_escapes=decimal_escapes,
|
|
352
|
+
decimal_runs=decimal_runs,
|
|
353
|
+
empty_separators=empty_separators,
|
|
354
|
+
mnemonic_escapes=mnemonic_escapes,
|
|
355
|
+
)
|