vcti-escapers 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vcti_escapers-1.0.0/LICENSE +8 -0
- vcti_escapers-1.0.0/PKG-INFO +162 -0
- vcti_escapers-1.0.0/README.md +136 -0
- vcti_escapers-1.0.0/pyproject.toml +72 -0
- vcti_escapers-1.0.0/setup.cfg +4 -0
- vcti_escapers-1.0.0/src/vcti/escapers/__init__.py +73 -0
- vcti_escapers-1.0.0/src/vcti/escapers/_csv.py +85 -0
- vcti_escapers-1.0.0/src/vcti/escapers/_html.py +142 -0
- vcti_escapers-1.0.0/src/vcti/escapers/_json.py +100 -0
- vcti_escapers-1.0.0/src/vcti/escapers/_markdown.py +129 -0
- vcti_escapers-1.0.0/src/vcti/escapers/_text.py +55 -0
- vcti_escapers-1.0.0/src/vcti/escapers/_url.py +69 -0
- vcti_escapers-1.0.0/src/vcti/escapers/_xml.py +119 -0
- vcti_escapers-1.0.0/src/vcti/escapers/py.typed +0 -0
- vcti_escapers-1.0.0/src/vcti_escapers.egg-info/PKG-INFO +162 -0
- vcti_escapers-1.0.0/src/vcti_escapers.egg-info/SOURCES.txt +27 -0
- vcti_escapers-1.0.0/src/vcti_escapers.egg-info/dependency_links.txt +1 -0
- vcti_escapers-1.0.0/src/vcti_escapers.egg-info/requires.txt +12 -0
- vcti_escapers-1.0.0/src/vcti_escapers.egg-info/top_level.txt +1 -0
- vcti_escapers-1.0.0/tests/test_contract.py +142 -0
- vcti_escapers-1.0.0/tests/test_csv.py +177 -0
- vcti_escapers-1.0.0/tests/test_docs.py +172 -0
- vcti_escapers-1.0.0/tests/test_examples.py +94 -0
- vcti_escapers-1.0.0/tests/test_html.py +159 -0
- vcti_escapers-1.0.0/tests/test_json.py +208 -0
- vcti_escapers-1.0.0/tests/test_markdown.py +171 -0
- vcti_escapers-1.0.0/tests/test_url.py +100 -0
- vcti_escapers-1.0.0/tests/test_version.py +15 -0
- vcti_escapers-1.0.0/tests/test_xml.py +121 -0
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
Copyright (c) 2018-2026 Visual Collaboration Technologies Inc.
|
|
2
|
+
All Rights Reserved.
|
|
3
|
+
|
|
4
|
+
This software is proprietary and confidential. Unauthorized copying,
|
|
5
|
+
distribution, or use of this software, via any medium, is strictly
|
|
6
|
+
prohibited. Access is granted only to authorized VCollab developers
|
|
7
|
+
and individuals explicitly authorized by Visual Collaboration
|
|
8
|
+
Technologies Inc.
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: vcti-escapers
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Escapers for CSV, Markdown, XML, HTML, JSON and URL, each verified against a real parser
|
|
5
|
+
Author: Visual Collaboration Technologies Inc.
|
|
6
|
+
License-Expression: LicenseRef-Proprietary
|
|
7
|
+
Project-URL: Repository, https://github.com/vcollab/vcti-python-escapers
|
|
8
|
+
Project-URL: Changelog, https://github.com/vcollab/vcti-python-escapers/blob/main/CHANGELOG.md
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
13
|
+
Requires-Python: <3.15,>=3.12
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
License-File: LICENSE
|
|
16
|
+
Provides-Extra: test
|
|
17
|
+
Requires-Dist: pytest; extra == "test"
|
|
18
|
+
Requires-Dist: pytest-cov; extra == "test"
|
|
19
|
+
Requires-Dist: markdown-it-py; extra == "test"
|
|
20
|
+
Requires-Dist: html5lib; extra == "test"
|
|
21
|
+
Provides-Extra: lint
|
|
22
|
+
Requires-Dist: ruff; extra == "lint"
|
|
23
|
+
Provides-Extra: typecheck
|
|
24
|
+
Requires-Dist: mypy; extra == "typecheck"
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+
# Escapers
|
|
28
|
+
|
|
29
|
+
Escapers for CSV, Markdown, XML, HTML, JSON and URL, each verified against a real parser
|
|
30
|
+
|
|
31
|
+
## Overview
|
|
32
|
+
|
|
33
|
+
Text that came from somewhere else — a name, a status, a free-text reason —
|
|
34
|
+
cannot be dropped into a document as it stands. A comma reshapes a CSV row, a
|
|
35
|
+
pipe splits a Markdown table cell, an unescaped quote ends an HTML attribute
|
|
36
|
+
early and starts whatever follows it. An escaper takes such a value and
|
|
37
|
+
returns one that embeds at a *specific* place in a *specific* format without
|
|
38
|
+
changing the document's structure.
|
|
39
|
+
|
|
40
|
+
There is no general "escaped string": a value is escaped **for** somewhere.
|
|
41
|
+
The same name that is safe inside a Markdown code span breaks a CSV row, and
|
|
42
|
+
the same value quoted for CSV is meaningless inside an XML attribute. So each
|
|
43
|
+
escaper here names the position it is for and is correct there and nowhere
|
|
44
|
+
else. Two escapers are never applied to the same value for extra safety;
|
|
45
|
+
where a document genuinely contains another, they nest once per layer.
|
|
46
|
+
|
|
47
|
+
Every escaper accepts any value whose `str()` succeeds — `None` included —
|
|
48
|
+
and returns a string that can be written as UTF-8. Each is verified by
|
|
49
|
+
round-tripping a hostile value through a real parser for its format, in the
|
|
50
|
+
position a consumer would use it, rather than by reading the specification.
|
|
51
|
+
|
|
52
|
+
## Installation
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
pip install vcti-escapers
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
### In `requirements.txt`
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
vcti-escapers>=1.0.0
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### In `pyproject.toml` dependencies
|
|
65
|
+
|
|
66
|
+
```toml
|
|
67
|
+
dependencies = [
|
|
68
|
+
"vcti-escapers>=1.0.0",
|
|
69
|
+
]
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
---
|
|
73
|
+
|
|
74
|
+
## Quick Start
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
from vcti.escapers import csv_field, html_attribute, json_string, xml_text
|
|
78
|
+
|
|
79
|
+
csv_field("failed, retried") # '"failed, retried"'
|
|
80
|
+
csv_field(138.0) # '138.0' — numbers stay bare
|
|
81
|
+
xml_text("a & b") # 'a & b'
|
|
82
|
+
html_attribute('" onclick="') # '" onclick="'
|
|
83
|
+
json_string("</script>") # '"\\u003c/script\\u003e"'
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Every escaper takes any value whose `str()` succeeds, `None` included. What
|
|
87
|
+
absence looks like is the target's decision, so each escaper documents its
|
|
88
|
+
own answer.
|
|
89
|
+
|
|
90
|
+
```python
|
|
91
|
+
csv_field(None) # '' — an empty field
|
|
92
|
+
json_string(None) # '""' — an empty JSON string, since null is not one
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
---
|
|
96
|
+
|
|
97
|
+
## The escapers
|
|
98
|
+
|
|
99
|
+
They form one flat pool, grouped here by format for finding rather than
|
|
100
|
+
because a format owns them.
|
|
101
|
+
|
|
102
|
+
| Format | Escaper | For |
|
|
103
|
+
|---|---|---|
|
|
104
|
+
| Markdown | `markdown_code` | An inline code span in prose — content shown literally, not obeyed |
|
|
105
|
+
| Markdown | `markdown_table_cell` | A code span in a GFM table cell, where `\|` must also be escaped |
|
|
106
|
+
| CSV | `csv_field` | One RFC 4180 field; text quoted, numbers bare |
|
|
107
|
+
| XML | `xml_text` | Character data between tags |
|
|
108
|
+
| XML | `xml_attribute` | Inside an attribute (caller supplies the quotes) |
|
|
109
|
+
| HTML | `html_text` | Ordinary text content between tags |
|
|
110
|
+
| HTML | `html_attribute` | Inside a quoted ordinary attribute (caller supplies the quotes) |
|
|
111
|
+
| JSON | `json_string` | A string literal, quotes included, safe inside `<script>` |
|
|
112
|
+
| URL | `url_path_segment` | One path segment — a slash stays inside it |
|
|
113
|
+
| URL | `url_query_value` | One query parameter value, form-encoded |
|
|
114
|
+
|
|
115
|
+
A format appears more than once because a format needs different escaping in
|
|
116
|
+
different positions — that distinction is the point, and picking the wrong
|
|
117
|
+
one is the mistake this package exists to prevent. Read each escaper's
|
|
118
|
+
docstring before first use; several carry limits that matter.
|
|
119
|
+
|
|
120
|
+
Two worth knowing up front:
|
|
121
|
+
|
|
122
|
+
- **`html_attribute` is for ordinary attributes.** An event handler
|
|
123
|
+
(`onclick`), a URL attribute (`href`, `src`) or `style` needs more than
|
|
124
|
+
escaping — a perfectly escaped `javascript:` URL still runs.
|
|
125
|
+
- **Escapers do not chain for extra safety.** They nest, once per layer, when
|
|
126
|
+
a document genuinely contains another — see
|
|
127
|
+
[docs/patterns.md](docs/patterns.md).
|
|
128
|
+
|
|
129
|
+
---
|
|
130
|
+
|
|
131
|
+
## Using them with a template engine
|
|
132
|
+
|
|
133
|
+
Escapers are plain functions, so they register wherever a template engine
|
|
134
|
+
takes callables. Nothing here depends on a template engine or knows one
|
|
135
|
+
exists.
|
|
136
|
+
|
|
137
|
+
```python
|
|
138
|
+
from vcti.escapers import csv_field, markdown_code
|
|
139
|
+
|
|
140
|
+
filters = {"csv_field": csv_field, "markdown_code": markdown_code}
|
|
141
|
+
# Jinja2: environment.filters.update(filters)
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
Registered under their own names, they read the same in a template as in
|
|
145
|
+
Python — `{{ item.name | csv_field }}`.
|
|
146
|
+
|
|
147
|
+
---
|
|
148
|
+
|
|
149
|
+
## Dependencies
|
|
150
|
+
|
|
151
|
+
None.
|
|
152
|
+
|
|
153
|
+
---
|
|
154
|
+
|
|
155
|
+
## Documentation
|
|
156
|
+
|
|
157
|
+
| If you want to… | Read |
|
|
158
|
+
|---|---|
|
|
159
|
+
| See practical, real-world usage | [docs/patterns.md](docs/patterns.md) |
|
|
160
|
+
| Understand the architecture and design decisions | [docs/design.md](docs/design.md) |
|
|
161
|
+
| Navigate and understand the source | [docs/source-guide.md](docs/source-guide.md) |
|
|
162
|
+
| Add an escaper | [docs/extending.md](docs/extending.md) |
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# Escapers
|
|
2
|
+
|
|
3
|
+
Escapers for CSV, Markdown, XML, HTML, JSON and URL, each verified against a real parser
|
|
4
|
+
|
|
5
|
+
## Overview
|
|
6
|
+
|
|
7
|
+
Text that came from somewhere else — a name, a status, a free-text reason —
|
|
8
|
+
cannot be dropped into a document as it stands. A comma reshapes a CSV row, a
|
|
9
|
+
pipe splits a Markdown table cell, an unescaped quote ends an HTML attribute
|
|
10
|
+
early and starts whatever follows it. An escaper takes such a value and
|
|
11
|
+
returns one that embeds at a *specific* place in a *specific* format without
|
|
12
|
+
changing the document's structure.
|
|
13
|
+
|
|
14
|
+
There is no general "escaped string": a value is escaped **for** somewhere.
|
|
15
|
+
The same name that is safe inside a Markdown code span breaks a CSV row, and
|
|
16
|
+
the same value quoted for CSV is meaningless inside an XML attribute. So each
|
|
17
|
+
escaper here names the position it is for and is correct there and nowhere
|
|
18
|
+
else. Two escapers are never applied to the same value for extra safety;
|
|
19
|
+
where a document genuinely contains another, they nest once per layer.
|
|
20
|
+
|
|
21
|
+
Every escaper accepts any value whose `str()` succeeds — `None` included —
|
|
22
|
+
and returns a string that can be written as UTF-8. Each is verified by
|
|
23
|
+
round-tripping a hostile value through a real parser for its format, in the
|
|
24
|
+
position a consumer would use it, rather than by reading the specification.
|
|
25
|
+
|
|
26
|
+
## Installation
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install vcti-escapers
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
### In `requirements.txt`
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
vcti-escapers>=1.0.0
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
### In `pyproject.toml` dependencies
|
|
39
|
+
|
|
40
|
+
```toml
|
|
41
|
+
dependencies = [
|
|
42
|
+
"vcti-escapers>=1.0.0",
|
|
43
|
+
]
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
## Quick Start
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from vcti.escapers import csv_field, html_attribute, json_string, xml_text
|
|
52
|
+
|
|
53
|
+
csv_field("failed, retried") # '"failed, retried"'
|
|
54
|
+
csv_field(138.0) # '138.0' — numbers stay bare
|
|
55
|
+
xml_text("a & b") # 'a & b'
|
|
56
|
+
html_attribute('" onclick="') # '" onclick="'
|
|
57
|
+
json_string("</script>") # '"\\u003c/script\\u003e"'
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Every escaper takes any value whose `str()` succeeds, `None` included. What
|
|
61
|
+
absence looks like is the target's decision, so each escaper documents its
|
|
62
|
+
own answer.
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
csv_field(None) # '' — an empty field
|
|
66
|
+
json_string(None) # '""' — an empty JSON string, since null is not one
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
---
|
|
70
|
+
|
|
71
|
+
## The escapers
|
|
72
|
+
|
|
73
|
+
They form one flat pool, grouped here by format for finding rather than
|
|
74
|
+
because a format owns them.
|
|
75
|
+
|
|
76
|
+
| Format | Escaper | For |
|
|
77
|
+
|---|---|---|
|
|
78
|
+
| Markdown | `markdown_code` | An inline code span in prose — content shown literally, not obeyed |
|
|
79
|
+
| Markdown | `markdown_table_cell` | A code span in a GFM table cell, where `\|` must also be escaped |
|
|
80
|
+
| CSV | `csv_field` | One RFC 4180 field; text quoted, numbers bare |
|
|
81
|
+
| XML | `xml_text` | Character data between tags |
|
|
82
|
+
| XML | `xml_attribute` | Inside an attribute (caller supplies the quotes) |
|
|
83
|
+
| HTML | `html_text` | Ordinary text content between tags |
|
|
84
|
+
| HTML | `html_attribute` | Inside a quoted ordinary attribute (caller supplies the quotes) |
|
|
85
|
+
| JSON | `json_string` | A string literal, quotes included, safe inside `<script>` |
|
|
86
|
+
| URL | `url_path_segment` | One path segment — a slash stays inside it |
|
|
87
|
+
| URL | `url_query_value` | One query parameter value, form-encoded |
|
|
88
|
+
|
|
89
|
+
A format appears more than once because a format needs different escaping in
|
|
90
|
+
different positions — that distinction is the point, and picking the wrong
|
|
91
|
+
one is the mistake this package exists to prevent. Read each escaper's
|
|
92
|
+
docstring before first use; several carry limits that matter.
|
|
93
|
+
|
|
94
|
+
Two worth knowing up front:
|
|
95
|
+
|
|
96
|
+
- **`html_attribute` is for ordinary attributes.** An event handler
|
|
97
|
+
(`onclick`), a URL attribute (`href`, `src`) or `style` needs more than
|
|
98
|
+
escaping — a perfectly escaped `javascript:` URL still runs.
|
|
99
|
+
- **Escapers do not chain for extra safety.** They nest, once per layer, when
|
|
100
|
+
a document genuinely contains another — see
|
|
101
|
+
[docs/patterns.md](docs/patterns.md).
|
|
102
|
+
|
|
103
|
+
---
|
|
104
|
+
|
|
105
|
+
## Using them with a template engine
|
|
106
|
+
|
|
107
|
+
Escapers are plain functions, so they register wherever a template engine
|
|
108
|
+
takes callables. Nothing here depends on a template engine or knows one
|
|
109
|
+
exists.
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
from vcti.escapers import csv_field, markdown_code
|
|
113
|
+
|
|
114
|
+
filters = {"csv_field": csv_field, "markdown_code": markdown_code}
|
|
115
|
+
# Jinja2: environment.filters.update(filters)
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Registered under their own names, they read the same in a template as in
|
|
119
|
+
Python — `{{ item.name | csv_field }}`.
|
|
120
|
+
|
|
121
|
+
---
|
|
122
|
+
|
|
123
|
+
## Dependencies
|
|
124
|
+
|
|
125
|
+
None.
|
|
126
|
+
|
|
127
|
+
---
|
|
128
|
+
|
|
129
|
+
## Documentation
|
|
130
|
+
|
|
131
|
+
| If you want to… | Read |
|
|
132
|
+
|---|---|
|
|
133
|
+
| See practical, real-world usage | [docs/patterns.md](docs/patterns.md) |
|
|
134
|
+
| Understand the architecture and design decisions | [docs/design.md](docs/design.md) |
|
|
135
|
+
| Navigate and understand the source | [docs/source-guide.md](docs/source-guide.md) |
|
|
136
|
+
| Add an escaper | [docs/extending.md](docs/extending.md) |
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "vcti-escapers"
|
|
7
|
+
version = "1.0.0"
|
|
8
|
+
description = "Escapers for CSV, Markdown, XML, HTML, JSON and URL, each verified against a real parser"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
authors = [
|
|
11
|
+
{name = "Visual Collaboration Technologies Inc."}
|
|
12
|
+
]
|
|
13
|
+
license = "LicenseRef-Proprietary"
|
|
14
|
+
license-files = ["LICENSE"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Operating System :: OS Independent",
|
|
17
|
+
"Programming Language :: Python :: 3.12",
|
|
18
|
+
"Programming Language :: Python :: 3.13",
|
|
19
|
+
"Programming Language :: Python :: 3.14",
|
|
20
|
+
]
|
|
21
|
+
requires-python = ">=3.12,<3.15"
|
|
22
|
+
dependencies = []
|
|
23
|
+
|
|
24
|
+
[project.urls]
|
|
25
|
+
Repository = "https://github.com/vcollab/vcti-python-escapers"
|
|
26
|
+
Changelog = "https://github.com/vcollab/vcti-python-escapers/blob/main/CHANGELOG.md"
|
|
27
|
+
|
|
28
|
+
[tool.setuptools.packages.find]
|
|
29
|
+
where = ["src"]
|
|
30
|
+
include = ["vcti.escapers", "vcti.escapers.*"]
|
|
31
|
+
|
|
32
|
+
[tool.setuptools.package-data]
|
|
33
|
+
"vcti.escapers" = ["py.typed"]
|
|
34
|
+
|
|
35
|
+
[project.optional-dependencies]
|
|
36
|
+
# markdown-it-py and html5lib are the real parsers the escapers are verified
|
|
37
|
+
# against. Markdown and HTML are the two targets with no parser in the
|
|
38
|
+
# standard library, and asserting on an escaper's output text rather than on
|
|
39
|
+
# what a parser makes of it is how two defects reached review.
|
|
40
|
+
test = ["pytest", "pytest-cov", "markdown-it-py", "html5lib"]
|
|
41
|
+
lint = ["ruff"]
|
|
42
|
+
typecheck = ["mypy"]
|
|
43
|
+
|
|
44
|
+
[tool.pytest.ini_options]
|
|
45
|
+
addopts = "--cov=vcti.escapers --cov-report=term-missing --cov-fail-under=95"
|
|
46
|
+
|
|
47
|
+
[tool.mypy]
|
|
48
|
+
python_version = "3.12"
|
|
49
|
+
strict = true
|
|
50
|
+
files = ["src"]
|
|
51
|
+
namespace_packages = true
|
|
52
|
+
explicit_package_bases = true
|
|
53
|
+
mypy_path = ["src"]
|
|
54
|
+
|
|
55
|
+
[tool.coverage.run]
|
|
56
|
+
branch = true
|
|
57
|
+
|
|
58
|
+
[tool.coverage.report]
|
|
59
|
+
exclude_also = [
|
|
60
|
+
"raise NotImplementedError",
|
|
61
|
+
"if TYPE_CHECKING:",
|
|
62
|
+
"if __name__ == .__main__.:",
|
|
63
|
+
"@(abc\\.)?abstractmethod",
|
|
64
|
+
"\\.\\.\\.",
|
|
65
|
+
]
|
|
66
|
+
|
|
67
|
+
[tool.ruff]
|
|
68
|
+
target-version = "py312"
|
|
69
|
+
line-length = 99
|
|
70
|
+
|
|
71
|
+
[tool.ruff.lint]
|
|
72
|
+
select = ["E", "F", "W", "I", "UP"]
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
# Copyright Visual Collaboration Technologies Inc. All Rights Reserved.
|
|
2
|
+
# See LICENSE for details.
|
|
3
|
+
"""Escapers: functions that render a value safe to embed in one target.
|
|
4
|
+
|
|
5
|
+
Each escaper is correct for exactly one position and wrong everywhere else.
|
|
6
|
+
A value quoted for a CSV field is meaningless inside an HTML attribute; a
|
|
7
|
+
Markdown code span does not survive a JSON document; and within one format,
|
|
8
|
+
text and attribute positions need different escapers.
|
|
9
|
+
|
|
10
|
+
**Input domain.** An escaper accepts any value whose ``str()`` succeeds and
|
|
11
|
+
returns a string, ``None`` included. It does not defend against a ``str()``
|
|
12
|
+
that raises - nothing can - and where a target cannot represent a character
|
|
13
|
+
at all, the escaper says so in its own documentation rather than pretending
|
|
14
|
+
otherwise.
|
|
15
|
+
|
|
16
|
+
**Output guarantee.** What comes back always encodes as UTF-8. A Python
|
|
17
|
+
string can hold a lone surrogate and UTF-8 cannot write one, so an escaper
|
|
18
|
+
that passed one through would return something that raises whenever the
|
|
19
|
+
document is finally written. Each escaper either escapes surrogates, where
|
|
20
|
+
its format has a syntax for it, or drops them, where it does not.
|
|
21
|
+
|
|
22
|
+
The escapers form one flat pool. They are grouped here by the format they
|
|
23
|
+
target, but that grouping is a way of finding them, not a structure they
|
|
24
|
+
belong to: an escaping rule often serves several formats, and a format often
|
|
25
|
+
needs several escapers for its different positions. Nothing organises around
|
|
26
|
+
one owning format.
|
|
27
|
+
|
|
28
|
+
Markdown:
|
|
29
|
+
:func:`markdown_code`, :func:`markdown_table_cell`
|
|
30
|
+
|
|
31
|
+
CSV:
|
|
32
|
+
:func:`csv_field`
|
|
33
|
+
|
|
34
|
+
XML:
|
|
35
|
+
:func:`xml_text`, :func:`xml_attribute`
|
|
36
|
+
|
|
37
|
+
HTML:
|
|
38
|
+
:func:`html_text`, :func:`html_attribute`
|
|
39
|
+
|
|
40
|
+
JSON:
|
|
41
|
+
:func:`json_string`
|
|
42
|
+
|
|
43
|
+
URL:
|
|
44
|
+
:func:`url_path_segment`, :func:`url_query_value`
|
|
45
|
+
|
|
46
|
+
See docs/design.md for the contract every escaper here holds to, and for what
|
|
47
|
+
this package deliberately does not do.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
from importlib.metadata import version
|
|
51
|
+
|
|
52
|
+
from ._csv import csv_field
|
|
53
|
+
from ._html import html_attribute, html_text
|
|
54
|
+
from ._json import json_string
|
|
55
|
+
from ._markdown import markdown_code, markdown_table_cell
|
|
56
|
+
from ._url import url_path_segment, url_query_value
|
|
57
|
+
from ._xml import xml_attribute, xml_text
|
|
58
|
+
|
|
59
|
+
__version__ = version("vcti-escapers")
|
|
60
|
+
|
|
61
|
+
__all__ = [
|
|
62
|
+
"__version__",
|
|
63
|
+
"csv_field",
|
|
64
|
+
"html_attribute",
|
|
65
|
+
"html_text",
|
|
66
|
+
"json_string",
|
|
67
|
+
"markdown_code",
|
|
68
|
+
"markdown_table_cell",
|
|
69
|
+
"url_path_segment",
|
|
70
|
+
"url_query_value",
|
|
71
|
+
"xml_attribute",
|
|
72
|
+
"xml_text",
|
|
73
|
+
]
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
# Copyright Visual Collaboration Technologies Inc. All Rights Reserved.
|
|
2
|
+
# See LICENSE for details.
|
|
3
|
+
"""Escapers for CSV (RFC 4180).
|
|
4
|
+
|
|
5
|
+
File layout groups escapers by the format they target. That grouping is a
|
|
6
|
+
finding aid, not the API — every escaper is imported from
|
|
7
|
+
:mod:`vcti.escapers` directly.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from ._text import without_surrogates
|
|
11
|
+
|
|
12
|
+
# The characters RFC 4180 requires a field to be quoted for. A value holding
|
|
13
|
+
# one of these cannot be written bare whatever its type says.
|
|
14
|
+
_NEEDS_QUOTING = frozenset(',"\r\n')
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def csv_field(value: object) -> str:
|
|
18
|
+
"""Render a value as one RFC 4180 field.
|
|
19
|
+
|
|
20
|
+
Text is quoted, with embedded quotes doubled, so a comma, a quote or a
|
|
21
|
+
newline in it cannot reshape the row. Numbers are written bare.
|
|
22
|
+
|
|
23
|
+
**Why quoting follows the type rather than the content.** RFC 4180 needs
|
|
24
|
+
a field quoted only when it holds a comma, a quote or a line break, and
|
|
25
|
+
most CSV writers quote exactly then. Deciding by type instead keeps a
|
|
26
|
+
column's quoting stable for the life of a file: a numeric column is never
|
|
27
|
+
quoted and a text column always is, whatever a particular row happens to
|
|
28
|
+
contain. That reads predictably in a text editor and diffs cleanly, which
|
|
29
|
+
matters for a file people review by hand and commit.
|
|
30
|
+
|
|
31
|
+
**The type proposes, the rendered text decides.** ``int``, ``float`` and
|
|
32
|
+
``bool`` are candidates for going bare, but what they render to is
|
|
33
|
+
checked for a delimiter first and quoted if it holds one. The type alone
|
|
34
|
+
is not grounds: these are not sealed types, and a subclass may override
|
|
35
|
+
``__str__`` to return anything at all —
|
|
36
|
+
|
|
37
|
+
.. code-block:: python
|
|
38
|
+
|
|
39
|
+
class Odd(int):
|
|
40
|
+
def __str__(self) -> str:
|
|
41
|
+
return "a,b"
|
|
42
|
+
|
|
43
|
+
— which on the strength of ``isinstance`` alone would go out bare, split
|
|
44
|
+
one field into two and reshape the row. Checking the text closes that
|
|
45
|
+
without giving up the type rule, and a genuine numeric subclass such as
|
|
46
|
+
``numpy.float64`` still renders bare as intended.
|
|
47
|
+
|
|
48
|
+
For the built-in types themselves the check never fires: ``str()`` of an
|
|
49
|
+
``int``, ``float`` or ``bool`` cannot produce a comma, a quote or a line
|
|
50
|
+
break — including ``inf``, ``nan`` and exponent forms, and regardless of
|
|
51
|
+
locale, since ``str()`` of a float always renders a ``.`` decimal point.
|
|
52
|
+
Anything not a candidate is quoted outright, so a type this does not
|
|
53
|
+
recognise — a ``Decimal``, a tuple whose ``str()`` holds a comma — is
|
|
54
|
+
quoted rather than left to split the row.
|
|
55
|
+
|
|
56
|
+
``bool`` is handled before ``int`` deliberately, rather than by relying on
|
|
57
|
+
it being an ``int`` subclass, so that ``True`` rendering bare is a
|
|
58
|
+
decision this function states rather than an accident of Python's type
|
|
59
|
+
hierarchy.
|
|
60
|
+
|
|
61
|
+
Lone surrogates are dropped: CSV is plain text with no escape syntax that
|
|
62
|
+
could carry one, and the field has to be writable as UTF-8.
|
|
63
|
+
|
|
64
|
+
This deliberately does **not** neutralise a leading ``=``, ``+``, ``-``
|
|
65
|
+
or ``@``, which some spreadsheets evaluate as a formula. That would be
|
|
66
|
+
defensive escaping, and it conflicts with lossless escaping: prefixing a
|
|
67
|
+
quote to disarm a spreadsheet corrupts the value for every other consumer
|
|
68
|
+
of the same file, and does not stop a spreadsheet that evaluates quoted
|
|
69
|
+
fields anyway. See ``SECURITY.md``; a caller needing spreadsheet-safe
|
|
70
|
+
output needs a different escaper, not a flag on this one.
|
|
71
|
+
|
|
72
|
+
Args:
|
|
73
|
+
value: The value to render, of any type.
|
|
74
|
+
|
|
75
|
+
Returns:
|
|
76
|
+
The field, carrying its own quotes when it has them. ``None`` renders
|
|
77
|
+
as an empty field, which is bare rather than ``""`` — the two parse
|
|
78
|
+
identically, and sparse columns stay readable.
|
|
79
|
+
"""
|
|
80
|
+
if value is None:
|
|
81
|
+
return ""
|
|
82
|
+
text = without_surrogates(str(value))
|
|
83
|
+
if isinstance(value, (bool, int, float)) and not _NEEDS_QUOTING.intersection(text):
|
|
84
|
+
return text
|
|
85
|
+
return '"' + text.replace('"', '""') + '"'
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
# Copyright Visual Collaboration Technologies Inc. All Rights Reserved.
|
|
2
|
+
# See LICENSE for details.
|
|
3
|
+
"""Escapers for HTML.
|
|
4
|
+
|
|
5
|
+
File layout groups escapers by the format they target. That grouping is a
|
|
6
|
+
finding aid, not the API - every escaper is imported from
|
|
7
|
+
:mod:`vcti.escapers` directly.
|
|
8
|
+
|
|
9
|
+
**These are for ordinary text and ordinary attribute values only.** Three
|
|
10
|
+
attribute kinds need more than escaping and are not served here:
|
|
11
|
+
|
|
12
|
+
- **Event handlers** (``onclick`` and friends) hold JavaScript. Escaping for
|
|
13
|
+
HTML makes the value parse as one attribute; it does nothing about the
|
|
14
|
+
value then being executed as script. Do not interpolate into one.
|
|
15
|
+
- **URL-bearing attributes** (``href``, ``src``, ``action``) need the scheme
|
|
16
|
+
checked. ``html_attribute`` will happily escape ``javascript:alert(1)``
|
|
17
|
+
into a perfectly well-formed attribute that still runs on click. Validate
|
|
18
|
+
the URL, then escape it.
|
|
19
|
+
- **``style``** holds CSS, which has its own syntax and its own escapes.
|
|
20
|
+
|
|
21
|
+
See :mod:`vcti.escapers._xml` for why HTML and XML do not share an escaper.
|
|
22
|
+
Both normalise line endings, so both need a carriage return written as a
|
|
23
|
+
reference; where they part company is the rest of the control range, which
|
|
24
|
+
XML forbids outright and HTML mostly tolerates, and the entity set - HTML 4
|
|
25
|
+
never defined ``'``.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from ._text import without_surrogates
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _writable(text: str) -> str:
|
|
32
|
+
"""Settle the two things that are the same in both HTML positions.
|
|
33
|
+
|
|
34
|
+
Before any markup is parsed, an HTML parser normalises line endings: a
|
|
35
|
+
CRLF pair and a lone CR both become a single line feed. A carriage return
|
|
36
|
+
written literally therefore never reaches the document, in text or in an
|
|
37
|
+
attribute. Written as a character reference it does, because references
|
|
38
|
+
are resolved after preprocessing.
|
|
39
|
+
|
|
40
|
+
A lone surrogate is dropped rather than passed on. HTML cannot write one
|
|
41
|
+
- a reference to a surrogate is a parse error - and leaving it in returns
|
|
42
|
+
a string that raises when the document is encoded as UTF-8, which turns
|
|
43
|
+
an escaper's output into someone else's error.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
text: The already-escaped text.
|
|
47
|
+
|
|
48
|
+
Returns:
|
|
49
|
+
The text with carriage returns written as character references and
|
|
50
|
+
lone surrogates removed.
|
|
51
|
+
"""
|
|
52
|
+
return without_surrogates(text).replace("\r", " ")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def html_text(value: object) -> str:
|
|
56
|
+
"""Render a value as HTML text content.
|
|
57
|
+
|
|
58
|
+
For text between tags - ``<td>{{ value | html_text }}</td>``. Quotes are
|
|
59
|
+
left alone: they cannot end an element, and leaving them keeps the source
|
|
60
|
+
readable. Use :func:`html_attribute` for anything going inside an
|
|
61
|
+
attribute, where they can.
|
|
62
|
+
|
|
63
|
+
This does not sanitise. It makes a value safe to *show*; it does not make
|
|
64
|
+
untrusted markup safe to render, because there is no markup left - that
|
|
65
|
+
is the point.
|
|
66
|
+
|
|
67
|
+
**Not lossless for two inputs.** U+0000 has no representation at all, and
|
|
68
|
+
is passed through unchanged because there is nothing better to do with
|
|
69
|
+
it. What a parser makes of a literal one depends on where it lands: in
|
|
70
|
+
element text it is **discarded**, and in an attribute value it is
|
|
71
|
+
**replaced with U+FFFD**.
|
|
72
|
+
|
|
73
|
+
Writing it as ``�`` instead would not rescue it - a numeric reference
|
|
74
|
+
to U+0000 is a parse error that yields U+FFFD, in both positions. So the
|
|
75
|
+
reference is not "the same as" a literal NUL: it is uniform where the
|
|
76
|
+
literal is not, and lossy either way. There is no spelling that survives.
|
|
77
|
+
|
|
78
|
+
A lone surrogate is dropped here, since HTML has no way to write one
|
|
79
|
+
either - a reference to a surrogate is also a parse error - and the
|
|
80
|
+
output has to be writable as UTF-8.
|
|
81
|
+
|
|
82
|
+
A carriage return, by contrast, *is* preserved: written as `` `` it
|
|
83
|
+
survives the line-ending normalisation that would otherwise turn it into
|
|
84
|
+
a line feed.
|
|
85
|
+
|
|
86
|
+
Args:
|
|
87
|
+
value: The value to render, of any type.
|
|
88
|
+
|
|
89
|
+
Returns:
|
|
90
|
+
The escaped text. ``None`` renders as the empty string.
|
|
91
|
+
"""
|
|
92
|
+
if value is None:
|
|
93
|
+
return ""
|
|
94
|
+
text = str(value).replace("&", "&").replace("<", "<").replace(">", ">")
|
|
95
|
+
return _writable(text)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def html_attribute(value: object) -> str:
|
|
99
|
+
"""Render a value for inside a quoted HTML attribute.
|
|
100
|
+
|
|
101
|
+
Returns the attribute's *content* without delimiters, so the caller
|
|
102
|
+
supplies the quotes: ``<td title="{{ value | html_attribute }}">``. Both
|
|
103
|
+
quote characters are escaped, so either delimiter is safe.
|
|
104
|
+
|
|
105
|
+
**The caller must quote the attribute.** An unquoted attribute ends at
|
|
106
|
+
the first space, so no amount of quote escaping protects one - a value
|
|
107
|
+
with a space in it would introduce a second attribute. Escaping for
|
|
108
|
+
unquoted attributes would mean encoding whitespace, ``=`` and backticks
|
|
109
|
+
as well, which no mainstream escaper does; quoting the attribute is the
|
|
110
|
+
convention every HTML escaper assumes, and it is assumed here.
|
|
111
|
+
|
|
112
|
+
**For ordinary attributes only** - not an event handler, not a URL, not
|
|
113
|
+
``style``. See this module's own documentation for why each needs more
|
|
114
|
+
than escaping.
|
|
115
|
+
|
|
116
|
+
A single quote becomes ``'`` rather than ``'``, which HTML 4
|
|
117
|
+
never defined and older parsers show literally. The numeric reference is
|
|
118
|
+
correct in every HTML version and in XHTML.
|
|
119
|
+
|
|
120
|
+
Carriage returns, U+0000 and lone surrogates carry the same caveats as in
|
|
121
|
+
:func:`html_text`, with one difference worth knowing: U+0000 is replaced
|
|
122
|
+
with U+FFFD in an attribute value, where in element text it is discarded
|
|
123
|
+
outright. Neither survives, but they fail differently.
|
|
124
|
+
|
|
125
|
+
Args:
|
|
126
|
+
value: The value to render, of any type.
|
|
127
|
+
|
|
128
|
+
Returns:
|
|
129
|
+
The escaped attribute content, without surrounding quotes. ``None``
|
|
130
|
+
renders as the empty string.
|
|
131
|
+
"""
|
|
132
|
+
if value is None:
|
|
133
|
+
return ""
|
|
134
|
+
text = (
|
|
135
|
+
str(value)
|
|
136
|
+
.replace("&", "&")
|
|
137
|
+
.replace("<", "<")
|
|
138
|
+
.replace(">", ">")
|
|
139
|
+
.replace('"', """)
|
|
140
|
+
.replace("'", "'")
|
|
141
|
+
)
|
|
142
|
+
return _writable(text)
|