vcti-escapers 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. vcti_escapers-1.0.0/LICENSE +8 -0
  2. vcti_escapers-1.0.0/PKG-INFO +162 -0
  3. vcti_escapers-1.0.0/README.md +136 -0
  4. vcti_escapers-1.0.0/pyproject.toml +72 -0
  5. vcti_escapers-1.0.0/setup.cfg +4 -0
  6. vcti_escapers-1.0.0/src/vcti/escapers/__init__.py +73 -0
  7. vcti_escapers-1.0.0/src/vcti/escapers/_csv.py +85 -0
  8. vcti_escapers-1.0.0/src/vcti/escapers/_html.py +142 -0
  9. vcti_escapers-1.0.0/src/vcti/escapers/_json.py +100 -0
  10. vcti_escapers-1.0.0/src/vcti/escapers/_markdown.py +129 -0
  11. vcti_escapers-1.0.0/src/vcti/escapers/_text.py +55 -0
  12. vcti_escapers-1.0.0/src/vcti/escapers/_url.py +69 -0
  13. vcti_escapers-1.0.0/src/vcti/escapers/_xml.py +119 -0
  14. vcti_escapers-1.0.0/src/vcti/escapers/py.typed +0 -0
  15. vcti_escapers-1.0.0/src/vcti_escapers.egg-info/PKG-INFO +162 -0
  16. vcti_escapers-1.0.0/src/vcti_escapers.egg-info/SOURCES.txt +27 -0
  17. vcti_escapers-1.0.0/src/vcti_escapers.egg-info/dependency_links.txt +1 -0
  18. vcti_escapers-1.0.0/src/vcti_escapers.egg-info/requires.txt +12 -0
  19. vcti_escapers-1.0.0/src/vcti_escapers.egg-info/top_level.txt +1 -0
  20. vcti_escapers-1.0.0/tests/test_contract.py +142 -0
  21. vcti_escapers-1.0.0/tests/test_csv.py +177 -0
  22. vcti_escapers-1.0.0/tests/test_docs.py +172 -0
  23. vcti_escapers-1.0.0/tests/test_examples.py +94 -0
  24. vcti_escapers-1.0.0/tests/test_html.py +159 -0
  25. vcti_escapers-1.0.0/tests/test_json.py +208 -0
  26. vcti_escapers-1.0.0/tests/test_markdown.py +171 -0
  27. vcti_escapers-1.0.0/tests/test_url.py +100 -0
  28. vcti_escapers-1.0.0/tests/test_version.py +15 -0
  29. vcti_escapers-1.0.0/tests/test_xml.py +121 -0
@@ -0,0 +1,8 @@
1
+ Copyright (c) 2018-2026 Visual Collaboration Technologies Inc.
2
+ All Rights Reserved.
3
+
4
+ This software is proprietary and confidential. Unauthorized copying,
5
+ distribution, or use of this software, via any medium, is strictly
6
+ prohibited. Access is granted only to authorized VCollab developers
7
+ and individuals explicitly authorized by Visual Collaboration
8
+ Technologies Inc.
@@ -0,0 +1,162 @@
1
+ Metadata-Version: 2.4
2
+ Name: vcti-escapers
3
+ Version: 1.0.0
4
+ Summary: Escapers for CSV, Markdown, XML, HTML, JSON and URL, each verified against a real parser
5
+ Author: Visual Collaboration Technologies Inc.
6
+ License-Expression: LicenseRef-Proprietary
7
+ Project-URL: Repository, https://github.com/vcollab/vcti-python-escapers
8
+ Project-URL: Changelog, https://github.com/vcollab/vcti-python-escapers/blob/main/CHANGELOG.md
9
+ Classifier: Operating System :: OS Independent
10
+ Classifier: Programming Language :: Python :: 3.12
11
+ Classifier: Programming Language :: Python :: 3.13
12
+ Classifier: Programming Language :: Python :: 3.14
13
+ Requires-Python: <3.15,>=3.12
14
+ Description-Content-Type: text/markdown
15
+ License-File: LICENSE
16
+ Provides-Extra: test
17
+ Requires-Dist: pytest; extra == "test"
18
+ Requires-Dist: pytest-cov; extra == "test"
19
+ Requires-Dist: markdown-it-py; extra == "test"
20
+ Requires-Dist: html5lib; extra == "test"
21
+ Provides-Extra: lint
22
+ Requires-Dist: ruff; extra == "lint"
23
+ Provides-Extra: typecheck
24
+ Requires-Dist: mypy; extra == "typecheck"
25
+ Dynamic: license-file
26
+
27
+ # Escapers
28
+
29
+ Escapers for CSV, Markdown, XML, HTML, JSON and URL, each verified against a real parser
30
+
31
+ ## Overview
32
+
33
+ Text that came from somewhere else — a name, a status, a free-text reason —
34
+ cannot be dropped into a document as it stands. A comma reshapes a CSV row, a
35
+ pipe splits a Markdown table cell, an unescaped quote ends an HTML attribute
36
+ early and starts whatever follows it. An escaper takes such a value and
37
+ returns one that embeds at a *specific* place in a *specific* format without
38
+ changing the document's structure.
39
+
40
+ There is no general "escaped string": a value is escaped **for** somewhere.
41
+ The same name that is safe inside a Markdown code span breaks a CSV row, and
42
+ the same value quoted for CSV is meaningless inside an XML attribute. So each
43
+ escaper here names the position it is for and is correct there and nowhere
44
+ else. Two escapers are never applied to the same value for extra safety;
45
+ where a document genuinely contains another, they nest once per layer.
46
+
47
+ Every escaper accepts any value whose `str()` succeeds — `None` included —
48
+ and returns a string that can be written as UTF-8. Each is verified by
49
+ round-tripping a hostile value through a real parser for its format, in the
50
+ position a consumer would use it, rather than by reading the specification.
51
+
52
+ ## Installation
53
+
54
+ ```bash
55
+ pip install vcti-escapers
56
+ ```
57
+
58
+ ### In `requirements.txt`
59
+
60
+ ```
61
+ vcti-escapers>=1.0.0
62
+ ```
63
+
64
+ ### In `pyproject.toml` dependencies
65
+
66
+ ```toml
67
+ dependencies = [
68
+ "vcti-escapers>=1.0.0",
69
+ ]
70
+ ```
71
+
72
+ ---
73
+
74
+ ## Quick Start
75
+
76
+ ```python
77
+ from vcti.escapers import csv_field, html_attribute, json_string, xml_text
78
+
79
+ csv_field("failed, retried") # '"failed, retried"'
80
+ csv_field(138.0) # '138.0' — numbers stay bare
81
+ xml_text("a & b") # 'a &amp; b'
82
+ html_attribute('" onclick="') # '&quot; onclick=&quot;'
83
+ json_string("</script>") # '"\\u003c/script\\u003e"'
84
+ ```
85
+
86
+ Every escaper takes any value whose `str()` succeeds, `None` included. What
87
+ absence looks like is the target's decision, so each escaper documents its
88
+ own answer.
89
+
90
+ ```python
91
+ csv_field(None) # '' — an empty field
92
+ json_string(None) # '""' — an empty JSON string, since null is not one
93
+ ```
94
+
95
+ ---
96
+
97
+ ## The escapers
98
+
99
+ They form one flat pool, grouped here by format for finding rather than
100
+ because a format owns them.
101
+
102
+ | Format | Escaper | For |
103
+ |---|---|---|
104
+ | Markdown | `markdown_code` | An inline code span in prose — content shown literally, not obeyed |
105
+ | Markdown | `markdown_table_cell` | A code span in a GFM table cell, where `\|` must also be escaped |
106
+ | CSV | `csv_field` | One RFC 4180 field; text quoted, numbers bare |
107
+ | XML | `xml_text` | Character data between tags |
108
+ | XML | `xml_attribute` | Inside an attribute (caller supplies the quotes) |
109
+ | HTML | `html_text` | Ordinary text content between tags |
110
+ | HTML | `html_attribute` | Inside a quoted ordinary attribute (caller supplies the quotes) |
111
+ | JSON | `json_string` | A string literal, quotes included, safe inside `<script>` |
112
+ | URL | `url_path_segment` | One path segment — a slash stays inside it |
113
+ | URL | `url_query_value` | One query parameter value, form-encoded |
114
+
115
+ A format appears more than once because a format needs different escaping in
116
+ different positions — that distinction is the point, and picking the wrong
117
+ one is the mistake this package exists to prevent. Read each escaper's
118
+ docstring before first use; several carry limits that matter.
119
+
120
+ Two worth knowing up front:
121
+
122
+ - **`html_attribute` is for ordinary attributes.** An event handler
123
+ (`onclick`), a URL attribute (`href`, `src`) or `style` needs more than
124
+ escaping — a perfectly escaped `javascript:` URL still runs.
125
+ - **Escapers do not chain for extra safety.** They nest, once per layer, when
126
+ a document genuinely contains another — see
127
+ [docs/patterns.md](docs/patterns.md).
128
+
129
+ ---
130
+
131
+ ## Using them with a template engine
132
+
133
+ Escapers are plain functions, so they register wherever a template engine
134
+ takes callables. Nothing here depends on a template engine or knows one
135
+ exists.
136
+
137
+ ```python
138
+ from vcti.escapers import csv_field, markdown_code
139
+
140
+ filters = {"csv_field": csv_field, "markdown_code": markdown_code}
141
+ # Jinja2: environment.filters.update(filters)
142
+ ```
143
+
144
+ Registered under their own names, they read the same in a template as in
145
+ Python — `{{ item.name | csv_field }}`.
146
+
147
+ ---
148
+
149
+ ## Dependencies
150
+
151
+ None.
152
+
153
+ ---
154
+
155
+ ## Documentation
156
+
157
+ | If you want to… | Read |
158
+ |---|---|
159
+ | See practical, real-world usage | [docs/patterns.md](docs/patterns.md) |
160
+ | Understand the architecture and design decisions | [docs/design.md](docs/design.md) |
161
+ | Navigate and understand the source | [docs/source-guide.md](docs/source-guide.md) |
162
+ | Add an escaper | [docs/extending.md](docs/extending.md) |
@@ -0,0 +1,136 @@
1
+ # Escapers
2
+
3
+ Escapers for CSV, Markdown, XML, HTML, JSON and URL, each verified against a real parser
4
+
5
+ ## Overview
6
+
7
+ Text that came from somewhere else — a name, a status, a free-text reason —
8
+ cannot be dropped into a document as it stands. A comma reshapes a CSV row, a
9
+ pipe splits a Markdown table cell, an unescaped quote ends an HTML attribute
10
+ early and starts whatever follows it. An escaper takes such a value and
11
+ returns one that embeds at a *specific* place in a *specific* format without
12
+ changing the document's structure.
13
+
14
+ There is no general "escaped string": a value is escaped **for** somewhere.
15
+ The same name that is safe inside a Markdown code span breaks a CSV row, and
16
+ the same value quoted for CSV is meaningless inside an XML attribute. So each
17
+ escaper here names the position it is for and is correct there and nowhere
18
+ else. Two escapers are never applied to the same value for extra safety;
19
+ where a document genuinely contains another, they nest once per layer.
20
+
21
+ Every escaper accepts any value whose `str()` succeeds — `None` included —
22
+ and returns a string that can be written as UTF-8. Each is verified by
23
+ round-tripping a hostile value through a real parser for its format, in the
24
+ position a consumer would use it, rather than by reading the specification.
25
+
26
+ ## Installation
27
+
28
+ ```bash
29
+ pip install vcti-escapers
30
+ ```
31
+
32
+ ### In `requirements.txt`
33
+
34
+ ```
35
+ vcti-escapers>=1.0.0
36
+ ```
37
+
38
+ ### In `pyproject.toml` dependencies
39
+
40
+ ```toml
41
+ dependencies = [
42
+ "vcti-escapers>=1.0.0",
43
+ ]
44
+ ```
45
+
46
+ ---
47
+
48
+ ## Quick Start
49
+
50
+ ```python
51
+ from vcti.escapers import csv_field, html_attribute, json_string, xml_text
52
+
53
+ csv_field("failed, retried") # '"failed, retried"'
54
+ csv_field(138.0) # '138.0' — numbers stay bare
55
+ xml_text("a & b") # 'a &amp; b'
56
+ html_attribute('" onclick="') # '&quot; onclick=&quot;'
57
+ json_string("</script>") # '"\\u003c/script\\u003e"'
58
+ ```
59
+
60
+ Every escaper takes any value whose `str()` succeeds, `None` included. What
61
+ absence looks like is the target's decision, so each escaper documents its
62
+ own answer.
63
+
64
+ ```python
65
+ csv_field(None) # '' — an empty field
66
+ json_string(None) # '""' — an empty JSON string, since null is not one
67
+ ```
68
+
69
+ ---
70
+
71
+ ## The escapers
72
+
73
+ They form one flat pool, grouped here by format for finding rather than
74
+ because a format owns them.
75
+
76
+ | Format | Escaper | For |
77
+ |---|---|---|
78
+ | Markdown | `markdown_code` | An inline code span in prose — content shown literally, not obeyed |
79
+ | Markdown | `markdown_table_cell` | A code span in a GFM table cell, where `\|` must also be escaped |
80
+ | CSV | `csv_field` | One RFC 4180 field; text quoted, numbers bare |
81
+ | XML | `xml_text` | Character data between tags |
82
+ | XML | `xml_attribute` | Inside an attribute (caller supplies the quotes) |
83
+ | HTML | `html_text` | Ordinary text content between tags |
84
+ | HTML | `html_attribute` | Inside a quoted ordinary attribute (caller supplies the quotes) |
85
+ | JSON | `json_string` | A string literal, quotes included, safe inside `<script>` |
86
+ | URL | `url_path_segment` | One path segment — a slash stays inside it |
87
+ | URL | `url_query_value` | One query parameter value, form-encoded |
88
+
89
+ A format appears more than once because a format needs different escaping in
90
+ different positions — that distinction is the point, and picking the wrong
91
+ one is the mistake this package exists to prevent. Read each escaper's
92
+ docstring before first use; several carry limits that matter.
93
+
94
+ Two worth knowing up front:
95
+
96
+ - **`html_attribute` is for ordinary attributes.** An event handler
97
+ (`onclick`), a URL attribute (`href`, `src`) or `style` needs more than
98
+ escaping — a perfectly escaped `javascript:` URL still runs.
99
+ - **Escapers do not chain for extra safety.** They nest, once per layer, when
100
+ a document genuinely contains another — see
101
+ [docs/patterns.md](docs/patterns.md).
102
+
103
+ ---
104
+
105
+ ## Using them with a template engine
106
+
107
+ Escapers are plain functions, so they register wherever a template engine
108
+ takes callables. Nothing here depends on a template engine or knows one
109
+ exists.
110
+
111
+ ```python
112
+ from vcti.escapers import csv_field, markdown_code
113
+
114
+ filters = {"csv_field": csv_field, "markdown_code": markdown_code}
115
+ # Jinja2: environment.filters.update(filters)
116
+ ```
117
+
118
+ Registered under their own names, they read the same in a template as in
119
+ Python — `{{ item.name | csv_field }}`.
120
+
121
+ ---
122
+
123
+ ## Dependencies
124
+
125
+ None.
126
+
127
+ ---
128
+
129
+ ## Documentation
130
+
131
+ | If you want to… | Read |
132
+ |---|---|
133
+ | See practical, real-world usage | [docs/patterns.md](docs/patterns.md) |
134
+ | Understand the architecture and design decisions | [docs/design.md](docs/design.md) |
135
+ | Navigate and understand the source | [docs/source-guide.md](docs/source-guide.md) |
136
+ | Add an escaper | [docs/extending.md](docs/extending.md) |
@@ -0,0 +1,72 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "vcti-escapers"
7
+ version = "1.0.0"
8
+ description = "Escapers for CSV, Markdown, XML, HTML, JSON and URL, each verified against a real parser"
9
+ readme = "README.md"
10
+ authors = [
11
+ {name = "Visual Collaboration Technologies Inc."}
12
+ ]
13
+ license = "LicenseRef-Proprietary"
14
+ license-files = ["LICENSE"]
15
+ classifiers = [
16
+ "Operating System :: OS Independent",
17
+ "Programming Language :: Python :: 3.12",
18
+ "Programming Language :: Python :: 3.13",
19
+ "Programming Language :: Python :: 3.14",
20
+ ]
21
+ requires-python = ">=3.12,<3.15"
22
+ dependencies = []
23
+
24
+ [project.urls]
25
+ Repository = "https://github.com/vcollab/vcti-python-escapers"
26
+ Changelog = "https://github.com/vcollab/vcti-python-escapers/blob/main/CHANGELOG.md"
27
+
28
+ [tool.setuptools.packages.find]
29
+ where = ["src"]
30
+ include = ["vcti.escapers", "vcti.escapers.*"]
31
+
32
+ [tool.setuptools.package-data]
33
+ "vcti.escapers" = ["py.typed"]
34
+
35
+ [project.optional-dependencies]
36
+ # markdown-it-py and html5lib are the real parsers the escapers are verified
37
+ # against. Markdown and HTML are the two targets with no parser in the
38
+ # standard library, and asserting on an escaper's output text rather than on
39
+ # what a parser makes of it is how two defects reached review.
40
+ test = ["pytest", "pytest-cov", "markdown-it-py", "html5lib"]
41
+ lint = ["ruff"]
42
+ typecheck = ["mypy"]
43
+
44
+ [tool.pytest.ini_options]
45
+ addopts = "--cov=vcti.escapers --cov-report=term-missing --cov-fail-under=95"
46
+
47
+ [tool.mypy]
48
+ python_version = "3.12"
49
+ strict = true
50
+ files = ["src"]
51
+ namespace_packages = true
52
+ explicit_package_bases = true
53
+ mypy_path = ["src"]
54
+
55
+ [tool.coverage.run]
56
+ branch = true
57
+
58
+ [tool.coverage.report]
59
+ exclude_also = [
60
+ "raise NotImplementedError",
61
+ "if TYPE_CHECKING:",
62
+ "if __name__ == .__main__.:",
63
+ "@(abc\\.)?abstractmethod",
64
+ "\\.\\.\\.",
65
+ ]
66
+
67
+ [tool.ruff]
68
+ target-version = "py312"
69
+ line-length = 99
70
+
71
+ [tool.ruff.lint]
72
+ select = ["E", "F", "W", "I", "UP"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,73 @@
1
+ # Copyright Visual Collaboration Technologies Inc. All Rights Reserved.
2
+ # See LICENSE for details.
3
+ """Escapers: functions that render a value safe to embed in one target.
4
+
5
+ Each escaper is correct for exactly one position and wrong everywhere else.
6
+ A value quoted for a CSV field is meaningless inside an HTML attribute; a
7
+ Markdown code span does not survive a JSON document; and within one format,
8
+ text and attribute positions need different escapers.
9
+
10
+ **Input domain.** An escaper accepts any value whose ``str()`` succeeds and
11
+ returns a string, ``None`` included. It does not defend against a ``str()``
12
+ that raises - nothing can - and where a target cannot represent a character
13
+ at all, the escaper says so in its own documentation rather than pretending
14
+ otherwise.
15
+
16
+ **Output guarantee.** What comes back always encodes as UTF-8. A Python
17
+ string can hold a lone surrogate and UTF-8 cannot write one, so an escaper
18
+ that passed one through would return something that raises whenever the
19
+ document is finally written. Each escaper either escapes surrogates, where
20
+ its format has a syntax for it, or drops them, where it does not.
21
+
22
+ The escapers form one flat pool. They are grouped here by the format they
23
+ target, but that grouping is a way of finding them, not a structure they
24
+ belong to: an escaping rule often serves several formats, and a format often
25
+ needs several escapers for its different positions. Nothing organises around
26
+ one owning format.
27
+
28
+ Markdown:
29
+ :func:`markdown_code`, :func:`markdown_table_cell`
30
+
31
+ CSV:
32
+ :func:`csv_field`
33
+
34
+ XML:
35
+ :func:`xml_text`, :func:`xml_attribute`
36
+
37
+ HTML:
38
+ :func:`html_text`, :func:`html_attribute`
39
+
40
+ JSON:
41
+ :func:`json_string`
42
+
43
+ URL:
44
+ :func:`url_path_segment`, :func:`url_query_value`
45
+
46
+ See docs/design.md for the contract every escaper here holds to, and for what
47
+ this package deliberately does not do.
48
+ """
49
+
50
+ from importlib.metadata import version
51
+
52
+ from ._csv import csv_field
53
+ from ._html import html_attribute, html_text
54
+ from ._json import json_string
55
+ from ._markdown import markdown_code, markdown_table_cell
56
+ from ._url import url_path_segment, url_query_value
57
+ from ._xml import xml_attribute, xml_text
58
+
59
+ __version__ = version("vcti-escapers")
60
+
61
+ __all__ = [
62
+ "__version__",
63
+ "csv_field",
64
+ "html_attribute",
65
+ "html_text",
66
+ "json_string",
67
+ "markdown_code",
68
+ "markdown_table_cell",
69
+ "url_path_segment",
70
+ "url_query_value",
71
+ "xml_attribute",
72
+ "xml_text",
73
+ ]
@@ -0,0 +1,85 @@
1
+ # Copyright Visual Collaboration Technologies Inc. All Rights Reserved.
2
+ # See LICENSE for details.
3
+ """Escapers for CSV (RFC 4180).
4
+
5
+ File layout groups escapers by the format they target. That grouping is a
6
+ finding aid, not the API — every escaper is imported from
7
+ :mod:`vcti.escapers` directly.
8
+ """
9
+
10
+ from ._text import without_surrogates
11
+
12
+ # The characters RFC 4180 requires a field to be quoted for. A value holding
13
+ # one of these cannot be written bare whatever its type says.
14
+ _NEEDS_QUOTING = frozenset(',"\r\n')
15
+
16
+
17
+ def csv_field(value: object) -> str:
18
+ """Render a value as one RFC 4180 field.
19
+
20
+ Text is quoted, with embedded quotes doubled, so a comma, a quote or a
21
+ newline in it cannot reshape the row. Numbers are written bare.
22
+
23
+ **Why quoting follows the type rather than the content.** RFC 4180 needs
24
+ a field quoted only when it holds a comma, a quote or a line break, and
25
+ most CSV writers quote exactly then. Deciding by type instead keeps a
26
+ column's quoting stable for the life of a file: a numeric column is never
27
+ quoted and a text column always is, whatever a particular row happens to
28
+ contain. That reads predictably in a text editor and diffs cleanly, which
29
+ matters for a file people review by hand and commit.
30
+
31
+ **The type proposes, the rendered text decides.** ``int``, ``float`` and
32
+ ``bool`` are candidates for going bare, but what they render to is
33
+ checked for a delimiter first and quoted if it holds one. The type alone
34
+ is not grounds: these are not sealed types, and a subclass may override
35
+ ``__str__`` to return anything at all —
36
+
37
+ .. code-block:: python
38
+
39
+ class Odd(int):
40
+ def __str__(self) -> str:
41
+ return "a,b"
42
+
43
+ — which on the strength of ``isinstance`` alone would go out bare, split
44
+ one field into two and reshape the row. Checking the text closes that
45
+ without giving up the type rule, and a genuine numeric subclass such as
46
+ ``numpy.float64`` still renders bare as intended.
47
+
48
+ For the built-in types themselves the check never fires: ``str()`` of an
49
+ ``int``, ``float`` or ``bool`` cannot produce a comma, a quote or a line
50
+ break — including ``inf``, ``nan`` and exponent forms, and regardless of
51
+ locale, since ``str()`` of a float always renders a ``.`` decimal point.
52
+ Anything not a candidate is quoted outright, so a type this does not
53
+ recognise — a ``Decimal``, a tuple whose ``str()`` holds a comma — is
54
+ quoted rather than left to split the row.
55
+
56
+ ``bool`` is handled before ``int`` deliberately, rather than by relying on
57
+ it being an ``int`` subclass, so that ``True`` rendering bare is a
58
+ decision this function states rather than an accident of Python's type
59
+ hierarchy.
60
+
61
+ Lone surrogates are dropped: CSV is plain text with no escape syntax that
62
+ could carry one, and the field has to be writable as UTF-8.
63
+
64
+ This deliberately does **not** neutralise a leading ``=``, ``+``, ``-``
65
+ or ``@``, which some spreadsheets evaluate as a formula. That would be
66
+ defensive escaping, and it conflicts with lossless escaping: prefixing a
67
+ quote to disarm a spreadsheet corrupts the value for every other consumer
68
+ of the same file, and does not stop a spreadsheet that evaluates quoted
69
+ fields anyway. See ``SECURITY.md``; a caller needing spreadsheet-safe
70
+ output needs a different escaper, not a flag on this one.
71
+
72
+ Args:
73
+ value: The value to render, of any type.
74
+
75
+ Returns:
76
+ The field, carrying its own quotes when it has them. ``None`` renders
77
+ as an empty field, which is bare rather than ``""`` — the two parse
78
+ identically, and sparse columns stay readable.
79
+ """
80
+ if value is None:
81
+ return ""
82
+ text = without_surrogates(str(value))
83
+ if isinstance(value, (bool, int, float)) and not _NEEDS_QUOTING.intersection(text):
84
+ return text
85
+ return '"' + text.replace('"', '""') + '"'
@@ -0,0 +1,142 @@
1
+ # Copyright Visual Collaboration Technologies Inc. All Rights Reserved.
2
+ # See LICENSE for details.
3
+ """Escapers for HTML.
4
+
5
+ File layout groups escapers by the format they target. That grouping is a
6
+ finding aid, not the API - every escaper is imported from
7
+ :mod:`vcti.escapers` directly.
8
+
9
+ **These are for ordinary text and ordinary attribute values only.** Three
10
+ attribute kinds need more than escaping and are not served here:
11
+
12
+ - **Event handlers** (``onclick`` and friends) hold JavaScript. Escaping for
13
+ HTML makes the value parse as one attribute; it does nothing about the
14
+ value then being executed as script. Do not interpolate into one.
15
+ - **URL-bearing attributes** (``href``, ``src``, ``action``) need the scheme
16
+ checked. ``html_attribute`` will happily escape ``javascript:alert(1)``
17
+ into a perfectly well-formed attribute that still runs on click. Validate
18
+ the URL, then escape it.
19
+ - **``style``** holds CSS, which has its own syntax and its own escapes.
20
+
21
+ See :mod:`vcti.escapers._xml` for why HTML and XML do not share an escaper.
22
+ Both normalise line endings, so both need a carriage return written as a
23
+ reference; where they part company is the rest of the control range, which
24
+ XML forbids outright and HTML mostly tolerates, and the entity set - HTML 4
25
+ never defined ``&apos;``.
26
+ """
27
+
28
+ from ._text import without_surrogates
29
+
30
+
31
+ def _writable(text: str) -> str:
32
+ """Settle the two things that are the same in both HTML positions.
33
+
34
+ Before any markup is parsed, an HTML parser normalises line endings: a
35
+ CRLF pair and a lone CR both become a single line feed. A carriage return
36
+ written literally therefore never reaches the document, in text or in an
37
+ attribute. Written as a character reference it does, because references
38
+ are resolved after preprocessing.
39
+
40
+ A lone surrogate is dropped rather than passed on. HTML cannot write one
41
+ - a reference to a surrogate is a parse error - and leaving it in returns
42
+ a string that raises when the document is encoded as UTF-8, which turns
43
+ an escaper's output into someone else's error.
44
+
45
+ Args:
46
+ text: The already-escaped text.
47
+
48
+ Returns:
49
+ The text with carriage returns written as character references and
50
+ lone surrogates removed.
51
+ """
52
+ return without_surrogates(text).replace("\r", "&#13;")
53
+
54
+
55
+ def html_text(value: object) -> str:
56
+ """Render a value as HTML text content.
57
+
58
+ For text between tags - ``<td>{{ value | html_text }}</td>``. Quotes are
59
+ left alone: they cannot end an element, and leaving them keeps the source
60
+ readable. Use :func:`html_attribute` for anything going inside an
61
+ attribute, where they can.
62
+
63
+ This does not sanitise. It makes a value safe to *show*; it does not make
64
+ untrusted markup safe to render, because there is no markup left - that
65
+ is the point.
66
+
67
+ **Not lossless for two inputs.** U+0000 has no representation at all, and
68
+ is passed through unchanged because there is nothing better to do with
69
+ it. What a parser makes of a literal one depends on where it lands: in
70
+ element text it is **discarded**, and in an attribute value it is
71
+ **replaced with U+FFFD**.
72
+
73
+ Writing it as ``&#0;`` instead would not rescue it - a numeric reference
74
+ to U+0000 is a parse error that yields U+FFFD, in both positions. So the
75
+ reference is not "the same as" a literal NUL: it is uniform where the
76
+ literal is not, and lossy either way. There is no spelling that survives.
77
+
78
+ A lone surrogate is dropped here, since HTML has no way to write one
79
+ either - a reference to a surrogate is also a parse error - and the
80
+ output has to be writable as UTF-8.
81
+
82
+ A carriage return, by contrast, *is* preserved: written as ``&#13;`` it
83
+ survives the line-ending normalisation that would otherwise turn it into
84
+ a line feed.
85
+
86
+ Args:
87
+ value: The value to render, of any type.
88
+
89
+ Returns:
90
+ The escaped text. ``None`` renders as the empty string.
91
+ """
92
+ if value is None:
93
+ return ""
94
+ text = str(value).replace("&", "&amp;").replace("<", "&lt;").replace(">", "&gt;")
95
+ return _writable(text)
96
+
97
+
98
+ def html_attribute(value: object) -> str:
99
+ """Render a value for inside a quoted HTML attribute.
100
+
101
+ Returns the attribute's *content* without delimiters, so the caller
102
+ supplies the quotes: ``<td title="{{ value | html_attribute }}">``. Both
103
+ quote characters are escaped, so either delimiter is safe.
104
+
105
+ **The caller must quote the attribute.** An unquoted attribute ends at
106
+ the first space, so no amount of quote escaping protects one - a value
107
+ with a space in it would introduce a second attribute. Escaping for
108
+ unquoted attributes would mean encoding whitespace, ``=`` and backticks
109
+ as well, which no mainstream escaper does; quoting the attribute is the
110
+ convention every HTML escaper assumes, and it is assumed here.
111
+
112
+ **For ordinary attributes only** - not an event handler, not a URL, not
113
+ ``style``. See this module's own documentation for why each needs more
114
+ than escaping.
115
+
116
+ A single quote becomes ``&#39;`` rather than ``&apos;``, which HTML 4
117
+ never defined and older parsers show literally. The numeric reference is
118
+ correct in every HTML version and in XHTML.
119
+
120
+ Carriage returns, U+0000 and lone surrogates carry the same caveats as in
121
+ :func:`html_text`, with one difference worth knowing: U+0000 is replaced
122
+ with U+FFFD in an attribute value, where in element text it is discarded
123
+ outright. Neither survives, but they fail differently.
124
+
125
+ Args:
126
+ value: The value to render, of any type.
127
+
128
+ Returns:
129
+ The escaped attribute content, without surrounding quotes. ``None``
130
+ renders as the empty string.
131
+ """
132
+ if value is None:
133
+ return ""
134
+ text = (
135
+ str(value)
136
+ .replace("&", "&amp;")
137
+ .replace("<", "&lt;")
138
+ .replace(">", "&gt;")
139
+ .replace('"', "&quot;")
140
+ .replace("'", "&#39;")
141
+ )
142
+ return _writable(text)