matrx-content-guard 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- matrx_content_guard-0.1.0/.gitignore +271 -0
- matrx_content_guard-0.1.0/AUDIT_RESULTS.md +24 -0
- matrx_content_guard-0.1.0/CONTRACT.md +188 -0
- matrx_content_guard-0.1.0/FEATURE.md +104 -0
- matrx_content_guard-0.1.0/PKG-INFO +46 -0
- matrx_content_guard-0.1.0/README.md +23 -0
- matrx_content_guard-0.1.0/matrx_content_guard/__init__.py +63 -0
- matrx_content_guard-0.1.0/matrx_content_guard/_text_scan.py +81 -0
- matrx_content_guard-0.1.0/matrx_content_guard/css_engine.py +155 -0
- matrx_content_guard-0.1.0/matrx_content_guard/exceptions.py +74 -0
- matrx_content_guard-0.1.0/matrx_content_guard/field_map.py +57 -0
- matrx_content_guard-0.1.0/matrx_content_guard/html_engine.py +210 -0
- matrx_content_guard-0.1.0/matrx_content_guard/js_engine.py +60 -0
- matrx_content_guard-0.1.0/matrx_content_guard/models.py +197 -0
- matrx_content_guard-0.1.0/matrx_content_guard/patch.py +244 -0
- matrx_content_guard-0.1.0/matrx_content_guard/policy.py +69 -0
- matrx_content_guard-0.1.0/matrx_content_guard/profiles.py +123 -0
- matrx_content_guard-0.1.0/matrx_content_guard/rules.py +176 -0
- matrx_content_guard-0.1.0/matrx_content_guard/validator.py +62 -0
- matrx_content_guard-0.1.0/pyproject.toml +54 -0
- matrx_content_guard-0.1.0/scripts/audit_live_content.py +176 -0
- matrx_content_guard-0.1.0/scripts/demo.py +86 -0
- matrx_content_guard-0.1.0/tests/conftest.py +16 -0
- matrx_content_guard-0.1.0/tests/fixtures/cms/component_footer.html +39 -0
- matrx_content_guard-0.1.0/tests/fixtures/cms/component_header.html +28 -0
- matrx_content_guard-0.1.0/tests/fixtures/cms/page_concierge_fragment.html +11 -0
- matrx_content_guard-0.1.0/tests/fixtures/cms/page_dairy_gut_health.css +77 -0
- matrx_content_guard-0.1.0/tests/fixtures/cms/page_dairy_gut_health.html +93 -0
- matrx_content_guard-0.1.0/tests/fixtures/cms/page_education_fragment.html +1 -0
- matrx_content_guard-0.1.0/tests/fixtures/cms/site_iopbm_global.css +220 -0
- matrx_content_guard-0.1.0/tests/fixtures/html_pages/iframe_embed.html +10 -0
- matrx_content_guard-0.1.0/tests/fixtures/html_pages/no_doctype.html +4 -0
- matrx_content_guard-0.1.0/tests/fixtures/html_pages/script_and_events.html +121 -0
- matrx_content_guard-0.1.0/tests/test_css_engine.py +106 -0
- matrx_content_guard-0.1.0/tests/test_exceptions.py +125 -0
- matrx_content_guard-0.1.0/tests/test_field_map.py +40 -0
- matrx_content_guard-0.1.0/tests/test_html_engine.py +190 -0
- matrx_content_guard-0.1.0/tests/test_js_engine.py +52 -0
- matrx_content_guard-0.1.0/tests/test_live_fixtures.py +92 -0
- matrx_content_guard-0.1.0/tests/test_patch.py +144 -0
- matrx_content_guard-0.1.0/tests/test_pathological.py +81 -0
- matrx_content_guard-0.1.0/tests/test_profiles.py +36 -0
- matrx_content_guard-0.1.0/tests/test_rules_registry.py +50 -0
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
*.pyc
|
|
2
|
+
secrets/
|
|
3
|
+
ignore/
|
|
4
|
+
temp/
|
|
5
|
+
logs/
|
|
6
|
+
# The broad `logs/` rule above is for RUNTIME log output, but it also matched
|
|
7
|
+
# the dashboard's SOURCE directory and silently swallowed an entire feature's
|
|
8
|
+
# files (only the pre-existing index.tsx stayed tracked), breaking the prod
|
|
9
|
+
# Docker build with "Could not resolve ./structured-tab". Re-include the source.
|
|
10
|
+
!apps/dashboard/src/features/logs/
|
|
11
|
+
!apps/dashboard/src/features/logs/**
|
|
12
|
+
todo
|
|
13
|
+
text_notes/
|
|
14
|
+
aidream/secrets/2.env
|
|
15
|
+
automation_matrix/matrix_processing/temp/*
|
|
16
|
+
cd
|
|
17
|
+
# Byte-compiled / optimized / DLL files
|
|
18
|
+
__pycache__/
|
|
19
|
+
*.py[cod]
|
|
20
|
+
*$py.class
|
|
21
|
+
|
|
22
|
+
# C extensions
|
|
23
|
+
*.so
|
|
24
|
+
.venv/
|
|
25
|
+
|
|
26
|
+
# Distribution / packaging
|
|
27
|
+
.Python
|
|
28
|
+
build/
|
|
29
|
+
develop-eggs/
|
|
30
|
+
dist/
|
|
31
|
+
downloads/
|
|
32
|
+
eggs/
|
|
33
|
+
.eggs/
|
|
34
|
+
lib/
|
|
35
|
+
lib64/
|
|
36
|
+
# The blanket lib/ rule above is from the standard Python .gitignore template
|
|
37
|
+
# and was silently swallowing TS source under the SPA `src/lib/` folders.
|
|
38
|
+
# Re-allow them explicitly so frontend builds don't ship without their lib layer.
|
|
39
|
+
!apps/dashboard/src/lib/
|
|
40
|
+
!apps/dashboard/src/lib/**
|
|
41
|
+
!apps/workflow-studio/src/lib/
|
|
42
|
+
!apps/workflow-studio/src/lib/**
|
|
43
|
+
parts/
|
|
44
|
+
sdist/
|
|
45
|
+
var/
|
|
46
|
+
wheels/
|
|
47
|
+
share/python-wheels/
|
|
48
|
+
*.egg-info/
|
|
49
|
+
.installed.cfg
|
|
50
|
+
*.egg
|
|
51
|
+
MANIFEST
|
|
52
|
+
|
|
53
|
+
# PyInstaller
|
|
54
|
+
# Usually these files are written by a python script from a template
|
|
55
|
+
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
|
56
|
+
*.manifest
|
|
57
|
+
*.spec
|
|
58
|
+
|
|
59
|
+
# Installer logs
|
|
60
|
+
pip-log.txt
|
|
61
|
+
pip-delete-this-directory.txt
|
|
62
|
+
|
|
63
|
+
# Unit test / coverage reports
|
|
64
|
+
ai/tests/clean_response.json
|
|
65
|
+
ai/tests/cx_storage_response.json
|
|
66
|
+
ai/tests/execution_test.py
|
|
67
|
+
ai/tests/final_response.json
|
|
68
|
+
htmlcov/
|
|
69
|
+
.tox/
|
|
70
|
+
.nox/
|
|
71
|
+
.coverage
|
|
72
|
+
.coverage.*
|
|
73
|
+
.cache
|
|
74
|
+
nosetests.xml
|
|
75
|
+
coverage.xml
|
|
76
|
+
*.cover
|
|
77
|
+
*.py,cover
|
|
78
|
+
.hypothesis/
|
|
79
|
+
.pytest_cache/
|
|
80
|
+
cover/
|
|
81
|
+
|
|
82
|
+
# Translations
|
|
83
|
+
*.mo
|
|
84
|
+
*.pot
|
|
85
|
+
|
|
86
|
+
# Django stuff:
|
|
87
|
+
*.log
|
|
88
|
+
local_settings.py
|
|
89
|
+
db.sqlite3
|
|
90
|
+
db.sqlite3-journal
|
|
91
|
+
|
|
92
|
+
# Flask stuff:
|
|
93
|
+
instance/
|
|
94
|
+
.webassets-cache
|
|
95
|
+
|
|
96
|
+
# Scrapy stuff:
|
|
97
|
+
.scrapy
|
|
98
|
+
|
|
99
|
+
# Sphinx documentation
|
|
100
|
+
docs/_build/
|
|
101
|
+
|
|
102
|
+
# PyBuilder
|
|
103
|
+
.pybuilder/
|
|
104
|
+
target/
|
|
105
|
+
|
|
106
|
+
# Jupyter Notebook
|
|
107
|
+
.ipynb_checkpoints
|
|
108
|
+
|
|
109
|
+
# IPython
|
|
110
|
+
profile_default/
|
|
111
|
+
ipython_config.py
|
|
112
|
+
|
|
113
|
+
# pyenv
|
|
114
|
+
# For a library or package, you might want to ignore these files since the code is
|
|
115
|
+
# intended to run in multiple environments; otherwise, check them in:
|
|
116
|
+
# .python-version
|
|
117
|
+
|
|
118
|
+
# pipenv
|
|
119
|
+
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
|
120
|
+
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
|
121
|
+
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
|
122
|
+
# install all needed dependencies.
|
|
123
|
+
#Pipfile.lock
|
|
124
|
+
|
|
125
|
+
# poetry
|
|
126
|
+
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
|
127
|
+
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
|
128
|
+
# commonly ignored for libraries.
|
|
129
|
+
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
|
130
|
+
|
|
131
|
+
# pdm
|
|
132
|
+
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
|
133
|
+
#pdm.lock
|
|
134
|
+
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
|
|
135
|
+
# in version control.
|
|
136
|
+
# https://pdm.fming.dev/#use-with-ide
|
|
137
|
+
.pdm.toml
|
|
138
|
+
|
|
139
|
+
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
|
140
|
+
__pypackages__/
|
|
141
|
+
|
|
142
|
+
# Celery stuff
|
|
143
|
+
celerybeat-schedule
|
|
144
|
+
celerybeat.pid
|
|
145
|
+
|
|
146
|
+
# SageMath parsed files
|
|
147
|
+
*.sage.py
|
|
148
|
+
|
|
149
|
+
# Environments
|
|
150
|
+
.env
|
|
151
|
+
.env_remote
|
|
152
|
+
.venv
|
|
153
|
+
env/
|
|
154
|
+
venv/
|
|
155
|
+
ENV/
|
|
156
|
+
env.bak/
|
|
157
|
+
venv.bak/
|
|
158
|
+
.env.armanonly
|
|
159
|
+
|
|
160
|
+
# Spyder project settings
|
|
161
|
+
.spyderproject
|
|
162
|
+
.spyproject
|
|
163
|
+
|
|
164
|
+
# Rope project settings
|
|
165
|
+
.ropeproject
|
|
166
|
+
|
|
167
|
+
# mkdocs documentation
|
|
168
|
+
/site
|
|
169
|
+
|
|
170
|
+
# mypy
|
|
171
|
+
.mypy_cache/
|
|
172
|
+
.dmypy.json
|
|
173
|
+
dmypy.json
|
|
174
|
+
|
|
175
|
+
# Pyre type checker
|
|
176
|
+
.pyre/
|
|
177
|
+
|
|
178
|
+
# random armani files
|
|
179
|
+
/armani_dev/secrets/
|
|
180
|
+
/armani/
|
|
181
|
+
/_armani/
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
# pytype static type analyzer
|
|
186
|
+
.pytype/
|
|
187
|
+
|
|
188
|
+
# Cython debug symbols
|
|
189
|
+
cython_debug/
|
|
190
|
+
|
|
191
|
+
.idea/
|
|
192
|
+
.vscode/
|
|
193
|
+
/node_modules/
|
|
194
|
+
|
|
195
|
+
# Frontend pnpm workspace (apps/) — node_modules at the workspace root and any
|
|
196
|
+
# member, plus Vite caches and build output. The unified lockfile (apps/pnpm-lock.yaml)
|
|
197
|
+
# IS committed; everything below is regenerated.
|
|
198
|
+
node_modules/
|
|
199
|
+
apps/**/.vite/
|
|
200
|
+
apps/**/dist/
|
|
201
|
+
.vite/
|
|
202
|
+
|
|
203
|
+
dump.rdb
|
|
204
|
+
|
|
205
|
+
frontend/
|
|
206
|
+
|
|
207
|
+
# AME Temp Files and directory structure
|
|
208
|
+
# Ignore all files in the temp directory and its subdirectories
|
|
209
|
+
/temp/**/*
|
|
210
|
+
/tmp/**/*
|
|
211
|
+
|
|
212
|
+
# Allow .gitkeep files to retain directory structure
|
|
213
|
+
!/temp/**/.gitkeep
|
|
214
|
+
!/tmp/**/.gitkeep
|
|
215
|
+
|
|
216
|
+
# Armani
|
|
217
|
+
.history*
|
|
218
|
+
.history/
|
|
219
|
+
local_data/
|
|
220
|
+
local_reports_data/
|
|
221
|
+
webscraper/quick_scrapes/temp/
|
|
222
|
+
automation_matrix/ai_apis/fireworks/_dev/*
|
|
223
|
+
automation_matrix/ai_apis/fireworks/_dev/fireworks_sample.py
|
|
224
|
+
*.pdf
|
|
225
|
+
*.flac
|
|
226
|
+
*.mp3
|
|
227
|
+
*.wav
|
|
228
|
+
miniconda.sh
|
|
229
|
+
/database/python_sql/temp_data/
|
|
230
|
+
.history*
|
|
231
|
+
.history/
|
|
232
|
+
.history/
|
|
233
|
+
|
|
234
|
+
_dev/
|
|
235
|
+
/_dev/
|
|
236
|
+
requirements_filtered.txt
|
|
237
|
+
|
|
238
|
+
# matrx-dev-tools backups
|
|
239
|
+
.env-backups/
|
|
240
|
+
# Matrx Ship config (contains API key)
|
|
241
|
+
.matrx-ship.json
|
|
242
|
+
|
|
243
|
+
# Matrx config (contains API keys)
|
|
244
|
+
.matrx.json
|
|
245
|
+
.matrx-tools.conf
|
|
246
|
+
|
|
247
|
+
# Claude Code local worktrees and per-user settings
|
|
248
|
+
.claude/worktrees/
|
|
249
|
+
.claude/settings.local.json
|
|
250
|
+
|
|
251
|
+
# Append-only snapshots from matrx_utils.update_history (unbounded; do not commit)
|
|
252
|
+
common/utils/data_in_code/data_history.json
|
|
253
|
+
packages/matrx-utils/matrx_utils/data_in_code/data_history.json
|
|
254
|
+
|
|
255
|
+
# Tool-dispatch debug logs — one file per server start, never committed
|
|
256
|
+
.matrx-debug/
|
|
257
|
+
|
|
258
|
+
# macOS Finder metadata
|
|
259
|
+
.DS_Store
|
|
260
|
+
**/.DS_Store
|
|
261
|
+
|
|
262
|
+
# Environment files
|
|
263
|
+
.env
|
|
264
|
+
.env.*
|
|
265
|
+
*.env
|
|
266
|
+
*.env.*
|
|
267
|
+
|
|
268
|
+
# Keep safe templates trackable
|
|
269
|
+
!.env.example
|
|
270
|
+
!.env.sample
|
|
271
|
+
!.env.template
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# matrx-content-guard — live audit results
|
|
2
|
+
|
|
3
|
+
Run: 2026-07-11T14:05:12.180277+00:00
|
|
4
|
+
Project: `viyklljfdhtidwecakwx` (CMS). Read-only. No content persisted — ids/rule-ids/counts only.
|
|
5
|
+
|
|
6
|
+
## Row counts checked
|
|
7
|
+
|
|
8
|
+
- `html_pages`: 242 rows
|
|
9
|
+
- `client_pages`: 23 rows
|
|
10
|
+
- `client_components`: 2 rows
|
|
11
|
+
- `client_sites`: 2 rows
|
|
12
|
+
|
|
13
|
+
## Result: 285 content fields validated, 0 blocked
|
|
14
|
+
|
|
15
|
+
**Zero blocked fields.** Every live content field on the CMS project validates `blocked=False` under its `standard` profile.
|
|
16
|
+
|
|
17
|
+
## Warning frequency (advisory only — never blocks under `standard`)
|
|
18
|
+
|
|
19
|
+
| rule_id | occurrences |
|
|
20
|
+
|---|---|
|
|
21
|
+
| `html.event_handler_attribute` | 140 |
|
|
22
|
+
| `html.external_resource_origin` | 107 |
|
|
23
|
+
| `js.inner_html_assignment` | 22 |
|
|
24
|
+
| `html.parser_recovered` | 12 |
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
# CONTRACT — matrx-content-guard (C3)
|
|
2
|
+
|
|
3
|
+
> Published day 1, 2026-07-09. Owner: P3 (Safety & patch engine). Consumers: P2 (dry-run UX,
|
|
4
|
+
> `cms_inspect.rules`), P1 (fail-closed write-boundary hook). Master plan: `../../docs/cms_agent_authoring/README.md`.
|
|
5
|
+
|
|
6
|
+
## Authority rule
|
|
7
|
+
|
|
8
|
+
**P1's service-layer hook is authoritative. Tool-layer validation (P2) is advisory/dry-run UX.**
|
|
9
|
+
A P2 tool calling `validate_patch`/`validate_content` before writing is a courtesy to the agent
|
|
10
|
+
(fast feedback, no wasted round trip) — it is NOT the enforcement point. P1's services MUST call
|
|
11
|
+
`validate_content` (or reject via the injected hook) at the actual write boundary, fail-closed, even
|
|
12
|
+
if P2 already checked. Two callers computing the same profile via `profile_for` guarantees they never
|
|
13
|
+
disagree about *which* policy applies; each still independently decides whether to persist.
|
|
14
|
+
|
|
15
|
+
## Public API
|
|
16
|
+
|
|
17
|
+
All five surface as top-level imports: `from matrx_content_guard import ...`.
|
|
18
|
+
|
|
19
|
+
```python
|
|
20
|
+
def validate_content(
|
|
21
|
+
content: str,
|
|
22
|
+
profile: str, # "name:variant" or bare "name" (-> :standard)
|
|
23
|
+
exceptions: list[ContentException] | None = None,
|
|
24
|
+
) -> ValidationReport: ...
|
|
25
|
+
|
|
26
|
+
def apply_patch(
|
|
27
|
+
original: str,
|
|
28
|
+
patch: SearchReplacePatch | UnifiedDiffPatch, # the `Patch` union
|
|
29
|
+
*,
|
|
30
|
+
dry_run: bool = False,
|
|
31
|
+
) -> PatchResult: ...
|
|
32
|
+
|
|
33
|
+
def validate_patch(
|
|
34
|
+
original: str,
|
|
35
|
+
patch: SearchReplacePatch | UnifiedDiffPatch,
|
|
36
|
+
profile: str,
|
|
37
|
+
*,
|
|
38
|
+
exceptions: list[ContentException] | None = None,
|
|
39
|
+
) -> ValidatePatchResult: ...
|
|
40
|
+
|
|
41
|
+
def describe_policy(profile: str) -> dict: ...
|
|
42
|
+
|
|
43
|
+
def profile_for(system: str, entity: str, field: str) -> str: ... # raises KeyError, never silently defaults
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
All five functions are pure: no DB access, no network, no filesystem I/O, no imports from `aidream`
|
|
47
|
+
or any tool framework. Same inputs -> byte-identical outputs, always.
|
|
48
|
+
|
|
49
|
+
## Models
|
|
50
|
+
|
|
51
|
+
### `Severity` (str enum)
|
|
52
|
+
`"warning"` | `"block"`. Only `"block"` (after exception resolution) sets `ValidationReport.blocked`.
|
|
53
|
+
|
|
54
|
+
### `Violation`
|
|
55
|
+
| field | type | meaning |
|
|
56
|
+
|---|---|---|
|
|
57
|
+
| `rule_id` | `str` | key into the rule registry (`matrx_content_guard.rules.RULES`) |
|
|
58
|
+
| `node_path` | `str` | HTML: lxml xpath (`/html/body/div[2]/script[1]`, `.../@href` for an attribute). CSS/JS: `"<prefix>:<line>:<col>"`. |
|
|
59
|
+
| `excerpt` | `str` | ≤200 chars, the offending snippet |
|
|
60
|
+
| `severity` | `Severity` | resolved for the profile that produced it |
|
|
61
|
+
| `fix_hint` | `str` | plain-English, actionable |
|
|
62
|
+
|
|
63
|
+
### `ValidationReport`
|
|
64
|
+
| field | type | meaning |
|
|
65
|
+
|---|---|---|
|
|
66
|
+
| `blocked` | `bool` | `True` iff `violations` is non-empty after exceptions are applied |
|
|
67
|
+
| `violations` | `list[Violation]` | BLOCK-severity, NOT excepted — these are what actually blocks |
|
|
68
|
+
| `warnings` | `list[Violation]` | WARNING-severity — never blocks, always surfaced |
|
|
69
|
+
| `excepted` | `list[Violation]` | BLOCK-severity findings suppressed by a matching `ContentException` — **never silently dropped**, always visible here for audit (extension beyond the brief's literal two-list spec, kept because "loud recovery, no silent defaults" is a repo-wide mandate) |
|
|
70
|
+
| `profile` | `str` | the resolved `"name:variant"` key that was actually applied |
|
|
71
|
+
|
|
72
|
+
### Patch models
|
|
73
|
+
- `SearchReplaceHunk { search: str, replace: str }`
|
|
74
|
+
- `SearchReplacePatch { format: "search_replace", hunks: list[SearchReplaceHunk] }`
|
|
75
|
+
- `UnifiedDiffPatch { format: "unified_diff", diff: str }` — a standard unified-diff string; converted
|
|
76
|
+
internally to search/replace hunks (context+removed -> context+added) and applied through the
|
|
77
|
+
identical matcher, so both formats give identical failure-context quality.
|
|
78
|
+
- `Patch = SearchReplacePatch | UnifiedDiffPatch`
|
|
79
|
+
- `PatchFailureReason` (str enum): `"no_match"` | `"ambiguous_match"` | `"malformed_diff"`
|
|
80
|
+
- `PatchFailure { hunk_index, reason, match_count, positions: list[int], closest_match: str|None, closest_match_position: int|None, closest_match_similarity: float|None, message }`
|
|
81
|
+
- `PatchStats { hunks_total, hunks_applied, chars_before, chars_after, lines_changed }`
|
|
82
|
+
- `PatchResult { applied: bool, result: str|None, failures: list[PatchFailure], stats: PatchStats, dry_run: bool }`
|
|
83
|
+
— `result` is `None` unless **every** hunk applied (all-or-nothing). `dry_run` is echoed from the
|
|
84
|
+
call for the caller's audit trail; `apply_patch` has no I/O either way, so it doesn't change what's
|
|
85
|
+
computed — only the caller's own write-boundary decides whether to persist `result`.
|
|
86
|
+
- `ValidatePatchResult { patch: PatchResult, validation: ValidationReport|None }` — `validation` is
|
|
87
|
+
`None` iff the patch itself failed to apply (nothing to validate).
|
|
88
|
+
|
|
89
|
+
### `ContentException` (the F3 exception/approval data shape)
|
|
90
|
+
| field | type | engine-checked? | meaning |
|
|
91
|
+
|---|---|---|---|
|
|
92
|
+
| `rule_id` | `str` | yes | exact match required |
|
|
93
|
+
| `scope_site_id` | `str\|None` | **no** | store-schema/informational only — see below |
|
|
94
|
+
| `scope_page_id` | `str\|None` | **no** | store-schema/informational only — see below |
|
|
95
|
+
| `match_node_path_prefix` | `str\|None` | yes | violation's `node_path` must start with this |
|
|
96
|
+
| `match_excerpt_contains` | `str\|None` | yes | violation's `excerpt` must contain this substring |
|
|
97
|
+
| `status` | `str` | yes | only `"approved"` is ever honored |
|
|
98
|
+
| `note` | `str\|None` | no | free text |
|
|
99
|
+
|
|
100
|
+
**Scoping is two-tier.** Site/page granularity is resolved by the CALLER, not this engine: P1's
|
|
101
|
+
service layer knows which site/page it's writing to, queries only the exception rows in scope for
|
|
102
|
+
that write, and passes just those into `validate_content`/`validate_patch`. The engine never sees a
|
|
103
|
+
site_id/page_id and never filters on them. Rule + node granularity IS engine-checked: `rule_id` is
|
|
104
|
+
always exact; `match_node_path_prefix`/`match_excerpt_contains` optionally narrow an already-in-scope
|
|
105
|
+
exception to one subtree/snippet instead of blanket-suppressing the rule everywhere in the content.
|
|
106
|
+
|
|
107
|
+
**Store:** a small table on the CMS project (`viyklljfdhtidwecakwx`), owned by P1, schema coordinated
|
|
108
|
+
against this shape — migration + shared ledger rules apply (master plan §6.10). **Review/approve UI:**
|
|
109
|
+
P5. This package owns only the shape and the matching semantics
|
|
110
|
+
(`matrx_content_guard.exceptions.exception_matches` / `apply_exceptions`).
|
|
111
|
+
|
|
112
|
+
## Profiles
|
|
113
|
+
|
|
114
|
+
Five names × two variants = ten resolvable keys. A bare name resolves to `:standard`.
|
|
115
|
+
|
|
116
|
+
| name | content_kind | is_fragment | maps to (via `profile_for`) |
|
|
117
|
+
|---|---|---|---|
|
|
118
|
+
| `html_page_document` | full HTML document | no | `html_pages.html_content` |
|
|
119
|
+
| `cms_page_fragment` | HTML fragment | yes | `client_pages.html_content[_draft]` |
|
|
120
|
+
| `cms_component_fragment` | HTML fragment | yes | `client_components.html_content[_draft]` |
|
|
121
|
+
| `css` | CSS text | n/a | `client_pages.css_content[_draft]`, `client_components.css_content[_draft]`, `client_sites.global_css` |
|
|
122
|
+
| `js` | JS text | n/a | `client_pages.js_content[_draft]` |
|
|
123
|
+
|
|
124
|
+
`standard` is permissive-first (F3): every rule's default severity was calibrated against the live
|
|
125
|
+
corpus (239 `html_pages` + 19 `client_pages` rows, audited 2026-07-09) so **inline `<script>` (73
|
|
126
|
+
rows), event-handler attributes (40 rows), and external `<iframe>`/`<script src>` embeds (7 rows) all
|
|
127
|
+
validate `blocked=false`.** `strict` today makes exactly one change: `html.event_handler_attribute`
|
|
128
|
+
escalates from WARNING to BLOCK (for a future tighter surface, e.g. a public template gallery — no
|
|
129
|
+
current caller uses `strict`). Profiles are pure data (`matrx_content_guard/profiles.py`) — a future
|
|
130
|
+
policy change is a data edit, never an engine change.
|
|
131
|
+
|
|
132
|
+
## Rule registry
|
|
133
|
+
|
|
134
|
+
Every rule a violation can cite lives in `matrx_content_guard.rules.RULES`, one entry per `rule_id`,
|
|
135
|
+
each with a `category`, a ≤140-char plain-English `summary`, a `fix_hint`, and a `default_severity`.
|
|
136
|
+
`describe_policy(profile)` resolves each rule's *actual* severity for that profile/variant. See the
|
|
137
|
+
module for the full, current list (17 rules as of 2026-07-09) — adding a rule is one dict entry plus
|
|
138
|
+
the check that fires it; there is no second place severities live.
|
|
139
|
+
|
|
140
|
+
**Hard-block minimum (BLOCK in every variant, zero live false positives as of 2026-07-09):**
|
|
141
|
+
`html.denied_tag` (`<base>` only today), `html.dangerous_url_scheme` (`javascript:`/`vbscript:` in
|
|
142
|
+
any URL-bearing attribute, browser-style scheme-sniffing so `java\tscript:` doesn't bypass it),
|
|
143
|
+
`html.iframe_html_data_uri`, `html.css_expression_injection`, `html.css_javascript_url`,
|
|
144
|
+
`css.expression_injection`, `css.javascript_url`.
|
|
145
|
+
|
|
146
|
+
**Permissive/warn-only (never blocks under `standard`):** inline event handlers, external
|
|
147
|
+
script/iframe/form/link origins, `@import`/`url()` external origins, parser-recovered-from-malformed-
|
|
148
|
+
markup, and the entire `js.*` family (pattern-tier lint — `eval`, `document.write`, `new Function`,
|
|
149
|
+
`innerHTML=` — **JS static analysis is an explicit non-goal**; these never escalate to BLOCK in any
|
|
150
|
+
variant).
|
|
151
|
+
|
|
152
|
+
## Parser / engine boundaries (read before extending)
|
|
153
|
+
|
|
154
|
+
- **HTML:** `lxml.html` with `recover=True`. Validation walks the PARSED tree for reporting ONLY.
|
|
155
|
+
`apply_patch` never parses anything — pure string search/replace. A byte in storage never
|
|
156
|
+
round-trips through the HTML parser on write; this is the hard byte-preservation guarantee.
|
|
157
|
+
- **Fragments vs documents:** each profile declares `is_fragment`; fragments parse via
|
|
158
|
+
`lxml.html.fragment_fromstring(..., create_parent="div")` (always succeeds, one synthetic root for
|
|
159
|
+
stable xpaths), documents via `lxml.html.document_fromstring` (fills in missing `<html>`/`<body>`
|
|
160
|
+
when absent — covers the one live `html_pages` row with no doctype).
|
|
161
|
+
- **CSS:** bounded-regex pattern scans (`expression()`, `url()`, `@import`), not a full CSS parser.
|
|
162
|
+
Every pattern uses only bounded/non-nested quantifiers — no catastrophic-backtracking path — and
|
|
163
|
+
`PATTERN_SCAN_MAX_CHARS` (2,000,000) caps worst-case work regardless of document size, firing a
|
|
164
|
+
`content.pattern_scan_truncated` WARNING rather than silently skipping.
|
|
165
|
+
- **JS:** pattern-tier lint only, explicitly bounded — no parser, no AST, never BLOCK. Documented
|
|
166
|
+
non-goal per the brief.
|
|
167
|
+
|
|
168
|
+
## Wiring status (2026-07-09, live end-to-end as of this commit)
|
|
169
|
+
|
|
170
|
+
- **P2 is wired and live.** `aidream/tools/_cms_shared.py::content_guard()` lazily imports
|
|
171
|
+
`matrx_content_guard` and finds every C3 entry point (verified: `cms_inspect_tool.py` calls
|
|
172
|
+
`guard.describe_policy(...)` / `guard.PROFILE_NAMES`; `cms_page_tool.py`'s dry-run patch preview
|
|
173
|
+
calls `guard.profile_for("cms", "page", field)` and `guard.validate_patch(...)` — both using this
|
|
174
|
+
package's canonical vocabulary directly, no hand-translation).
|
|
175
|
+
- **P1's write-boundary hook is wired and live** — `aidream/package_integration.py::_configure_cms()`
|
|
176
|
+
imports `matrx_content_guard` and calls `configure_cms(content_validator=_ContentGuardAdapter(),
|
|
177
|
+
profile_resolver=matrx_content_guard.profile_for)`, replacing P1's interim placeholders wholesale as
|
|
178
|
+
their own docstrings said this package would. `_ContentGuardAdapter.validate_content` is a one-line
|
|
179
|
+
delegation to `matrx_content_guard.validate_content` — every CMS content-bearing write now fails
|
|
180
|
+
closed on a blocked report via `enforce_content`, not just P2's dry-run preview. Verified: this
|
|
181
|
+
package's `validate_content`/`validate_patch` accept the exceptions shape P1's `ContentValidator`
|
|
182
|
+
Protocol types as `list[dict]` (coerced via `exceptions.coerce_exceptions` — added after confirming
|
|
183
|
+
a raw dict would otherwise crash with `AttributeError`, since no exception store exists yet to have
|
|
184
|
+
caught this in practice).
|
|
185
|
+
- This package is complete, installed (`packages/matrx-content-guard`, in the root `uv` workspace +
|
|
186
|
+
`pyproject.toml` deps), 128 tests passing, live-audited (0/280 real content fields blocked,
|
|
187
|
+
`AUDIT_RESULTS.md`), and adversarially reviewed (4 confirmed bypasses found and fixed — see
|
|
188
|
+
`FEATURE.md`'s "Security hardening" note) as of this commit.
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# FEATURE.md — matrx-content-guard
|
|
2
|
+
|
|
3
|
+
> Durable design notes. The typed contract lives in `CONTRACT.md` (C3) — read that first for the
|
|
4
|
+
> API. This file is the WHY behind the choices, for whoever extends this package next.
|
|
5
|
+
|
|
6
|
+
## What this is
|
|
7
|
+
|
|
8
|
+
The deterministic validation + patch/diff engine every CMS agent mutation flows through
|
|
9
|
+
(`html_pages.html_content`, `client_pages.*_content[_draft]`, `client_components.*_content[_draft]`,
|
|
10
|
+
`client_sites.global_css`). Pure Python, zero DB access, zero tool-framework imports — a library, not
|
|
11
|
+
a service. Two jobs: (1) walk content and report policy violations, (2) apply agent-authored
|
|
12
|
+
search/replace or unified-diff patches to stored content.
|
|
13
|
+
|
|
14
|
+
## Why permissive-first (F3)
|
|
15
|
+
|
|
16
|
+
Arman's explicit failure mode: rules so restrictive that nobody wants to build websites on the
|
|
17
|
+
platform. The live corpus (239 `html_pages` + 21 `client_pages` as of 2026-07-09, up from 19 at
|
|
18
|
+
brief-time — `client_pages` grows) already contains inline `<script>` (73 rows), event-handler
|
|
19
|
+
attributes (138 field-hits), and external `<iframe>`/`<script src>` embeds (107 field-hits) — all of
|
|
20
|
+
that is normal, legitimate content, not an attack. The hard-block list is therefore deliberately tiny
|
|
21
|
+
and picked for near-zero false-positive risk: a `javascript:`/`vbscript:` URL, a CSS
|
|
22
|
+
`expression()`/`url(javascript:...)`, an HTML-data-URI iframe, and the `<base>` tag. Everything else
|
|
23
|
+
(event handlers, external origins, JS pattern hits) is WARNING — visible, never blocking. See
|
|
24
|
+
`rules.py`'s module docstring for the full reasoning per rule.
|
|
25
|
+
|
|
26
|
+
## Why lxml over html5lib
|
|
27
|
+
|
|
28
|
+
C-speed matters (some live rows are 200KB+ and get walked on every mutation), `recover=True` already
|
|
29
|
+
tolerates everything actually seen in production (including the one row with no `<!doctype>`/`<html>`
|
|
30
|
+
at all), and `getroottree().getpath(el)` gives a free, stable node path. html5lib's closer-to-a-real-
|
|
31
|
+
browser error correction wasn't worth the speed cost for content this well-behaved. If a future
|
|
32
|
+
pathological doc needs browser-exact recovery, that's the first thing to revisit.
|
|
33
|
+
|
|
34
|
+
## Security hardening (2026-07-09, adversarial review)
|
|
35
|
+
|
|
36
|
+
An adversarial subagent review targeting bypass techniques against the hard-block rules found and
|
|
37
|
+
confirmed four real gaps, all fixed the same day (regression tests in the matching `test_*.py`):
|
|
38
|
+
|
|
39
|
+
1. **`xlink:href` (any `prefix:href`/`prefix:src`) bypassed the hard-block entirely.** lxml's HTML
|
|
40
|
+
parser doesn't resolve XML namespaces, so `xlink:href` is literally the attribute name
|
|
41
|
+
`"xlink:href"` — comparing against `"href"` misses it completely. A classic real-world XSS bypass
|
|
42
|
+
technique targets exactly this gap. Fixed in `html_engine.py` by matching on the attribute's local
|
|
43
|
+
name (after the last `:`) instead of the raw name.
|
|
44
|
+
2. **`<iframe srcdoc="...">` was entirely unscanned** despite being the same risk class as the
|
|
45
|
+
already-blocked `data:text/html` iframe `src` (both create an executing nested browsing context
|
|
46
|
+
from inline markup). Fixed by recursively scanning `srcdoc` content under the same profile
|
|
47
|
+
(`MAX_SRCDOC_DEPTH = 6` backstop against pathological nesting), rather than blanket-denying the
|
|
48
|
+
attribute — a legitimate `srcdoc` embed is still only flagged for what it actually contains.
|
|
49
|
+
3. **CSS comments (`/**/`) and unicode escapes (`\65`) could split the literal `"expression("` /
|
|
50
|
+
`"javascript:"` strings past the hard-block regexes.** Real-world exploitability against THIS
|
|
51
|
+
platform's evergreen-browser-served pages is near-nil (modern browsers don't execute either
|
|
52
|
+
construct via CSS regardless), but the package's own contract calls these a hard BLOCK, so
|
|
53
|
+
`css_engine.py::_deobfuscate` now strips comments and decodes CSS unicode escapes before matching
|
|
54
|
+
— belt-and-suspenders for any non-evergreen consumer (legacy webviews, PDF renderers, email
|
|
55
|
+
clients) that might still honor them.
|
|
56
|
+
4. **An exception's `match_node_path_prefix`/`match_excerpt_contains` set to `""` silently behaved as
|
|
57
|
+
an always-match wildcard**, identical to a much broader "no constraint" — a likely footgun for any
|
|
58
|
+
UI/serializer that sends `""` instead of omitting an unset field, silently widening an approver's
|
|
59
|
+
intended narrow scope. `exceptions.py::exception_matches` now treats blank strings identically to
|
|
60
|
+
`None` in both fields.
|
|
61
|
+
|
|
62
|
+
Also hardened: `validate_content`/`validate_patch` now accept exceptions as plain `dict`s as well as
|
|
63
|
+
`ContentException` instances (`exceptions.coerce_exceptions`) — P1's own `ContentValidator` Protocol
|
|
64
|
+
types its `exceptions` param as `list[dict]`, and any real DB-backed exception store will round-trip
|
|
65
|
+
as dicts too; the strict-model-only version would have crashed with `AttributeError` the moment either
|
|
66
|
+
path carried a non-empty exceptions list.
|
|
67
|
+
|
|
68
|
+
## The byte-preservation split (the one invariant that can't regress)
|
|
69
|
+
|
|
70
|
+
Validation and patching are two completely separate code paths that never touch each other's data:
|
|
71
|
+
`html_engine.py`/`css_engine.py`/`js_engine.py` PARSE content (via lxml / regex) to build a report —
|
|
72
|
+
the parsed tree is discarded, never serialized back out. `patch.py` operates on the raw string ONLY —
|
|
73
|
+
no parser anywhere in that file. A byte in storage that a patch didn't explicitly target is therefore
|
|
74
|
+
*guaranteed* unchanged, because nothing in the write path ever round-trips content through a parser.
|
|
75
|
+
If you're ever tempted to "normalize" or "pretty-print" content on write inside this package — don't;
|
|
76
|
+
that's exactly the bug this split exists to prevent (brief gotcha (a)).
|
|
77
|
+
|
|
78
|
+
## Where the F3 exception store actually lives
|
|
79
|
+
|
|
80
|
+
This package only defines the shape (`ContentException`) and the matching semantics
|
|
81
|
+
(`exceptions.py`). The actual table is P1's (`aidream/services/cms/`, coordinated schema, migration +
|
|
82
|
+
ledger rules per master plan §6.10) and the approve/review UI is P5's. `validate_content` takes
|
|
83
|
+
already-scoped-by-the-caller exceptions — it never queries anything.
|
|
84
|
+
|
|
85
|
+
## Coordination gap as of this commit (see CONTRACT.md "Wiring status")
|
|
86
|
+
|
|
87
|
+
P1 already built the consumer seam (`aidream/services/cms/validation.py` + `profiles.py`) with
|
|
88
|
+
interim, hand-mapped placeholders that predate this package. Wiring the real thing in is P1's job
|
|
89
|
+
(directory ownership), but there are two shape mismatches to close — documented in detail in
|
|
90
|
+
`CONTRACT.md`. P2 hasn't wired `cms_inspect.rules` (`describe_policy`) or dry-run (`validate_patch`)
|
|
91
|
+
yet either.
|
|
92
|
+
|
|
93
|
+
## Extending this package
|
|
94
|
+
|
|
95
|
+
- **New rule:** one entry in `rules.py::RULES` (id, category, ≤140-char summary, fix_hint,
|
|
96
|
+
default_severity) + the check that fires it (in `html_engine.py`/`css_engine.py`/`js_engine.py`) +
|
|
97
|
+
a test in the matching `test_*_engine.py`. Never invent a second place for severity/copy.
|
|
98
|
+
- **New profile:** add to `profiles.py::_CONTENT_KIND_BY_NAME` (+ any `severity_overrides`) — profiles
|
|
99
|
+
are pure data, a policy change should never touch engine code.
|
|
100
|
+
- **New content field:** one entry in `field_map.py::_FIELD_PROFILE_MAP`. `profile_for` raises loudly
|
|
101
|
+
on anything unmapped — that's intentional, never add a silent default branch.
|
|
102
|
+
- **Re-running the live audit:** `scripts/audit_live_content.py` (needs `SUPABASE_CMS_*` env vars,
|
|
103
|
+
read-only) — re-proves the "0 blocked fields" guarantee as content grows; output committed to
|
|
104
|
+
`AUDIT_RESULTS.md`.
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: matrx-content-guard
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Deterministic HTML/CSS/JS content validation + patch/diff editing engine for the Matrx CMS agent-authoring layer
|
|
5
|
+
Project-URL: Homepage, https://github.com/AI-Matrix-Engine/aidream-current
|
|
6
|
+
Project-URL: Repository, https://github.com/AI-Matrix-Engine/aidream-current
|
|
7
|
+
Author-email: Matrx <admin@aimatrx.com>
|
|
8
|
+
Maintainer-email: Matrx <admin@aimatrx.com>
|
|
9
|
+
License: MIT
|
|
10
|
+
Keywords: cms,content-validation,html-sanitization,matrx,patch-diff
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
17
|
+
Classifier: Topic :: Text Processing :: Markup :: HTML
|
|
18
|
+
Requires-Python: >=3.13
|
|
19
|
+
Requires-Dist: lxml>=5.0
|
|
20
|
+
Requires-Dist: pydantic>=2.0.0
|
|
21
|
+
Requires-Dist: tinycss2>=1.3
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# matrx-content-guard
|
|
25
|
+
|
|
26
|
+
Deterministic HTML/CSS/JS content validation + patch/diff editing for the
|
|
27
|
+
Matrx CMS agent-authoring layer. Pure Python, no DB access, no tool-framework
|
|
28
|
+
imports — every agent mutation to `html_pages` / `client_*` content flows
|
|
29
|
+
through this library before it's written.
|
|
30
|
+
|
|
31
|
+
See `CONTRACT.md` for the full typed API and `FEATURE.md` for the durable
|
|
32
|
+
design notes (parser choice, permissive-first policy rationale, byte-
|
|
33
|
+
preservation guarantee).
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
from matrx_content_guard import validate_content, apply_patch, SearchReplacePatch, SearchReplaceHunk
|
|
37
|
+
|
|
38
|
+
report = validate_content("<script>alert(1)</script>", "html_page_document")
|
|
39
|
+
report.blocked # False — inline script is a supported feature; only genuinely dangerous constructs block.
|
|
40
|
+
|
|
41
|
+
result = apply_patch(
|
|
42
|
+
"<h1>Hello</h1>",
|
|
43
|
+
SearchReplacePatch(hunks=[SearchReplaceHunk(search="Hello", replace="Hi")]),
|
|
44
|
+
)
|
|
45
|
+
result.result # "<h1>Hi</h1>"
|
|
46
|
+
```
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# matrx-content-guard
|
|
2
|
+
|
|
3
|
+
Deterministic HTML/CSS/JS content validation + patch/diff editing for the
|
|
4
|
+
Matrx CMS agent-authoring layer. Pure Python, no DB access, no tool-framework
|
|
5
|
+
imports — every agent mutation to `html_pages` / `client_*` content flows
|
|
6
|
+
through this library before it's written.
|
|
7
|
+
|
|
8
|
+
See `CONTRACT.md` for the full typed API and `FEATURE.md` for the durable
|
|
9
|
+
design notes (parser choice, permissive-first policy rationale, byte-
|
|
10
|
+
preservation guarantee).
|
|
11
|
+
|
|
12
|
+
```python
|
|
13
|
+
from matrx_content_guard import validate_content, apply_patch, SearchReplacePatch, SearchReplaceHunk
|
|
14
|
+
|
|
15
|
+
report = validate_content("<script>alert(1)</script>", "html_page_document")
|
|
16
|
+
report.blocked # False — inline script is a supported feature; only genuinely dangerous constructs block.
|
|
17
|
+
|
|
18
|
+
result = apply_patch(
|
|
19
|
+
"<h1>Hello</h1>",
|
|
20
|
+
SearchReplacePatch(hunks=[SearchReplaceHunk(search="Hello", replace="Hi")]),
|
|
21
|
+
)
|
|
22
|
+
result.result # "<h1>Hi</h1>"
|
|
23
|
+
```
|