matrx-content-guard 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. matrx_content_guard-0.1.0/.gitignore +271 -0
  2. matrx_content_guard-0.1.0/AUDIT_RESULTS.md +24 -0
  3. matrx_content_guard-0.1.0/CONTRACT.md +188 -0
  4. matrx_content_guard-0.1.0/FEATURE.md +104 -0
  5. matrx_content_guard-0.1.0/PKG-INFO +46 -0
  6. matrx_content_guard-0.1.0/README.md +23 -0
  7. matrx_content_guard-0.1.0/matrx_content_guard/__init__.py +63 -0
  8. matrx_content_guard-0.1.0/matrx_content_guard/_text_scan.py +81 -0
  9. matrx_content_guard-0.1.0/matrx_content_guard/css_engine.py +155 -0
  10. matrx_content_guard-0.1.0/matrx_content_guard/exceptions.py +74 -0
  11. matrx_content_guard-0.1.0/matrx_content_guard/field_map.py +57 -0
  12. matrx_content_guard-0.1.0/matrx_content_guard/html_engine.py +210 -0
  13. matrx_content_guard-0.1.0/matrx_content_guard/js_engine.py +60 -0
  14. matrx_content_guard-0.1.0/matrx_content_guard/models.py +197 -0
  15. matrx_content_guard-0.1.0/matrx_content_guard/patch.py +244 -0
  16. matrx_content_guard-0.1.0/matrx_content_guard/policy.py +69 -0
  17. matrx_content_guard-0.1.0/matrx_content_guard/profiles.py +123 -0
  18. matrx_content_guard-0.1.0/matrx_content_guard/rules.py +176 -0
  19. matrx_content_guard-0.1.0/matrx_content_guard/validator.py +62 -0
  20. matrx_content_guard-0.1.0/pyproject.toml +54 -0
  21. matrx_content_guard-0.1.0/scripts/audit_live_content.py +176 -0
  22. matrx_content_guard-0.1.0/scripts/demo.py +86 -0
  23. matrx_content_guard-0.1.0/tests/conftest.py +16 -0
  24. matrx_content_guard-0.1.0/tests/fixtures/cms/component_footer.html +39 -0
  25. matrx_content_guard-0.1.0/tests/fixtures/cms/component_header.html +28 -0
  26. matrx_content_guard-0.1.0/tests/fixtures/cms/page_concierge_fragment.html +11 -0
  27. matrx_content_guard-0.1.0/tests/fixtures/cms/page_dairy_gut_health.css +77 -0
  28. matrx_content_guard-0.1.0/tests/fixtures/cms/page_dairy_gut_health.html +93 -0
  29. matrx_content_guard-0.1.0/tests/fixtures/cms/page_education_fragment.html +1 -0
  30. matrx_content_guard-0.1.0/tests/fixtures/cms/site_iopbm_global.css +220 -0
  31. matrx_content_guard-0.1.0/tests/fixtures/html_pages/iframe_embed.html +10 -0
  32. matrx_content_guard-0.1.0/tests/fixtures/html_pages/no_doctype.html +4 -0
  33. matrx_content_guard-0.1.0/tests/fixtures/html_pages/script_and_events.html +121 -0
  34. matrx_content_guard-0.1.0/tests/test_css_engine.py +106 -0
  35. matrx_content_guard-0.1.0/tests/test_exceptions.py +125 -0
  36. matrx_content_guard-0.1.0/tests/test_field_map.py +40 -0
  37. matrx_content_guard-0.1.0/tests/test_html_engine.py +190 -0
  38. matrx_content_guard-0.1.0/tests/test_js_engine.py +52 -0
  39. matrx_content_guard-0.1.0/tests/test_live_fixtures.py +92 -0
  40. matrx_content_guard-0.1.0/tests/test_patch.py +144 -0
  41. matrx_content_guard-0.1.0/tests/test_pathological.py +81 -0
  42. matrx_content_guard-0.1.0/tests/test_profiles.py +36 -0
  43. matrx_content_guard-0.1.0/tests/test_rules_registry.py +50 -0
@@ -0,0 +1,271 @@
1
+ *.pyc
2
+ secrets/
3
+ ignore/
4
+ temp/
5
+ logs/
6
+ # The broad `logs/` rule above is for RUNTIME log output, but it also matched
7
+ # the dashboard's SOURCE directory and silently swallowed an entire feature's
8
+ # files (only the pre-existing index.tsx stayed tracked), breaking the prod
9
+ # Docker build with "Could not resolve ./structured-tab". Re-include the source.
10
+ !apps/dashboard/src/features/logs/
11
+ !apps/dashboard/src/features/logs/**
12
+ todo
13
+ text_notes/
14
+ aidream/secrets/2.env
15
+ automation_matrix/matrix_processing/temp/*
16
+ cd
17
+ # Byte-compiled / optimized / DLL files
18
+ __pycache__/
19
+ *.py[cod]
20
+ *$py.class
21
+
22
+ # C extensions
23
+ *.so
24
+ .venv/
25
+
26
+ # Distribution / packaging
27
+ .Python
28
+ build/
29
+ develop-eggs/
30
+ dist/
31
+ downloads/
32
+ eggs/
33
+ .eggs/
34
+ lib/
35
+ lib64/
36
+ # The blanket lib/ rule above is from the standard Python .gitignore template
37
+ # and was silently swallowing TS source under the SPA `src/lib/` folders.
38
+ # Re-allow them explicitly so frontend builds don't ship without their lib layer.
39
+ !apps/dashboard/src/lib/
40
+ !apps/dashboard/src/lib/**
41
+ !apps/workflow-studio/src/lib/
42
+ !apps/workflow-studio/src/lib/**
43
+ parts/
44
+ sdist/
45
+ var/
46
+ wheels/
47
+ share/python-wheels/
48
+ *.egg-info/
49
+ .installed.cfg
50
+ *.egg
51
+ MANIFEST
52
+
53
+ # PyInstaller
54
+ # Usually these files are written by a python script from a template
55
+ # before PyInstaller builds the exe, so as to inject date/other infos into it.
56
+ *.manifest
57
+ *.spec
58
+
59
+ # Installer logs
60
+ pip-log.txt
61
+ pip-delete-this-directory.txt
62
+
63
+ # Unit test / coverage reports
64
+ ai/tests/clean_response.json
65
+ ai/tests/cx_storage_response.json
66
+ ai/tests/execution_test.py
67
+ ai/tests/final_response.json
68
+ htmlcov/
69
+ .tox/
70
+ .nox/
71
+ .coverage
72
+ .coverage.*
73
+ .cache
74
+ nosetests.xml
75
+ coverage.xml
76
+ *.cover
77
+ *.py,cover
78
+ .hypothesis/
79
+ .pytest_cache/
80
+ cover/
81
+
82
+ # Translations
83
+ *.mo
84
+ *.pot
85
+
86
+ # Django stuff:
87
+ *.log
88
+ local_settings.py
89
+ db.sqlite3
90
+ db.sqlite3-journal
91
+
92
+ # Flask stuff:
93
+ instance/
94
+ .webassets-cache
95
+
96
+ # Scrapy stuff:
97
+ .scrapy
98
+
99
+ # Sphinx documentation
100
+ docs/_build/
101
+
102
+ # PyBuilder
103
+ .pybuilder/
104
+ target/
105
+
106
+ # Jupyter Notebook
107
+ .ipynb_checkpoints
108
+
109
+ # IPython
110
+ profile_default/
111
+ ipython_config.py
112
+
113
+ # pyenv
114
+ # For a library or package, you might want to ignore these files since the code is
115
+ # intended to run in multiple environments; otherwise, check them in:
116
+ # .python-version
117
+
118
+ # pipenv
119
+ # According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
120
+ # However, in case of collaboration, if having platform-specific dependencies or dependencies
121
+ # having no cross-platform support, pipenv may install dependencies that don't work, or not
122
+ # install all needed dependencies.
123
+ #Pipfile.lock
124
+
125
+ # poetry
126
+ # Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
127
+ # This is especially recommended for binary packages to ensure reproducibility, and is more
128
+ # commonly ignored for libraries.
129
+ # https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
130
+
131
+ # pdm
132
+ # Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
133
+ #pdm.lock
134
+ # pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
135
+ # in version control.
136
+ # https://pdm.fming.dev/#use-with-ide
137
+ .pdm.toml
138
+
139
+ # PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
140
+ __pypackages__/
141
+
142
+ # Celery stuff
143
+ celerybeat-schedule
144
+ celerybeat.pid
145
+
146
+ # SageMath parsed files
147
+ *.sage.py
148
+
149
+ # Environments
150
+ .env
151
+ .env_remote
152
+ .venv
153
+ env/
154
+ venv/
155
+ ENV/
156
+ env.bak/
157
+ venv.bak/
158
+ .env.armanonly
159
+
160
+ # Spyder project settings
161
+ .spyderproject
162
+ .spyproject
163
+
164
+ # Rope project settings
165
+ .ropeproject
166
+
167
+ # mkdocs documentation
168
+ /site
169
+
170
+ # mypy
171
+ .mypy_cache/
172
+ .dmypy.json
173
+ dmypy.json
174
+
175
+ # Pyre type checker
176
+ .pyre/
177
+
178
+ # random armani files
179
+ /armani_dev/secrets/
180
+ /armani/
181
+ /_armani/
182
+
183
+
184
+
185
+ # pytype static type analyzer
186
+ .pytype/
187
+
188
+ # Cython debug symbols
189
+ cython_debug/
190
+
191
+ .idea/
192
+ .vscode/
193
+ /node_modules/
194
+
195
+ # Frontend pnpm workspace (apps/) — node_modules at the workspace root and any
196
+ # member, plus Vite caches and build output. The unified lockfile (apps/pnpm-lock.yaml)
197
+ # IS committed; everything below is regenerated.
198
+ node_modules/
199
+ apps/**/.vite/
200
+ apps/**/dist/
201
+ .vite/
202
+
203
+ dump.rdb
204
+
205
+ frontend/
206
+
207
+ # AME Temp Files and directory structure
208
+ # Ignore all files in the temp directory and its subdirectories
209
+ /temp/**/*
210
+ /tmp/**/*
211
+
212
+ # Allow .gitkeep files to retain directory structure
213
+ !/temp/**/.gitkeep
214
+ !/tmp/**/.gitkeep
215
+
216
+ # Armani
217
+ .history*
218
+ .history/
219
+ local_data/
220
+ local_reports_data/
221
+ webscraper/quick_scrapes/temp/
222
+ automation_matrix/ai_apis/fireworks/_dev/*
223
+ automation_matrix/ai_apis/fireworks/_dev/fireworks_sample.py
224
+ *.pdf
225
+ *.flac
226
+ *.mp3
227
+ *.wav
228
+ miniconda.sh
229
+ /database/python_sql/temp_data/
230
+ .history*
231
+ .history/
232
+ .history/
233
+
234
+ _dev/
235
+ /_dev/
236
+ requirements_filtered.txt
237
+
238
+ # matrx-dev-tools backups
239
+ .env-backups/
240
+ # Matrx Ship config (contains API key)
241
+ .matrx-ship.json
242
+
243
+ # Matrx config (contains API keys)
244
+ .matrx.json
245
+ .matrx-tools.conf
246
+
247
+ # Claude Code local worktrees and per-user settings
248
+ .claude/worktrees/
249
+ .claude/settings.local.json
250
+
251
+ # Append-only snapshots from matrx_utils.update_history (unbounded; do not commit)
252
+ common/utils/data_in_code/data_history.json
253
+ packages/matrx-utils/matrx_utils/data_in_code/data_history.json
254
+
255
+ # Tool-dispatch debug logs — one file per server start, never committed
256
+ .matrx-debug/
257
+
258
+ # macOS Finder metadata
259
+ .DS_Store
260
+ **/.DS_Store
261
+
262
+ # Environment files
263
+ .env
264
+ .env.*
265
+ *.env
266
+ *.env.*
267
+
268
+ # Keep safe templates trackable
269
+ !.env.example
270
+ !.env.sample
271
+ !.env.template
@@ -0,0 +1,24 @@
1
+ # matrx-content-guard — live audit results
2
+
3
+ Run: 2026-07-11T14:05:12.180277+00:00
4
+ Project: `viyklljfdhtidwecakwx` (CMS). Read-only. No content persisted — ids/rule-ids/counts only.
5
+
6
+ ## Row counts checked
7
+
8
+ - `html_pages`: 242 rows
9
+ - `client_pages`: 23 rows
10
+ - `client_components`: 2 rows
11
+ - `client_sites`: 2 rows
12
+
13
+ ## Result: 285 content fields validated, 0 blocked
14
+
15
+ **Zero blocked fields.** Every live content field on the CMS project validates `blocked=False` under its `standard` profile.
16
+
17
+ ## Warning frequency (advisory only — never blocks under `standard`)
18
+
19
+ | rule_id | occurrences |
20
+ |---|---|
21
+ | `html.event_handler_attribute` | 140 |
22
+ | `html.external_resource_origin` | 107 |
23
+ | `js.inner_html_assignment` | 22 |
24
+ | `html.parser_recovered` | 12 |
@@ -0,0 +1,188 @@
1
+ # CONTRACT — matrx-content-guard (C3)
2
+
3
+ > Published day 1, 2026-07-09. Owner: P3 (Safety & patch engine). Consumers: P2 (dry-run UX,
4
+ > `cms_inspect.rules`), P1 (fail-closed write-boundary hook). Master plan: `../../docs/cms_agent_authoring/README.md`.
5
+
6
+ ## Authority rule
7
+
8
+ **P1's service-layer hook is authoritative. Tool-layer validation (P2) is advisory/dry-run UX.**
9
+ A P2 tool calling `validate_patch`/`validate_content` before writing is a courtesy to the agent
10
+ (fast feedback, no wasted round trip) — it is NOT the enforcement point. P1's services MUST call
11
+ `validate_content` (or reject via the injected hook) at the actual write boundary, fail-closed, even
12
+ if P2 already checked. Two callers computing the same profile via `profile_for` guarantees they never
13
+ disagree about *which* policy applies; each still independently decides whether to persist.
14
+
15
+ ## Public API
16
+
17
+ All five surface as top-level imports: `from matrx_content_guard import ...`.
18
+
19
+ ```python
20
+ def validate_content(
21
+ content: str,
22
+ profile: str, # "name:variant" or bare "name" (-> :standard)
23
+ exceptions: list[ContentException] | None = None,
24
+ ) -> ValidationReport: ...
25
+
26
+ def apply_patch(
27
+ original: str,
28
+ patch: SearchReplacePatch | UnifiedDiffPatch, # the `Patch` union
29
+ *,
30
+ dry_run: bool = False,
31
+ ) -> PatchResult: ...
32
+
33
+ def validate_patch(
34
+ original: str,
35
+ patch: SearchReplacePatch | UnifiedDiffPatch,
36
+ profile: str,
37
+ *,
38
+ exceptions: list[ContentException] | None = None,
39
+ ) -> ValidatePatchResult: ...
40
+
41
+ def describe_policy(profile: str) -> dict: ...
42
+
43
+ def profile_for(system: str, entity: str, field: str) -> str: ... # raises KeyError, never silently defaults
44
+ ```
45
+
46
+ All five functions are pure: no DB access, no network, no filesystem I/O, no imports from `aidream`
47
+ or any tool framework. Same inputs -> byte-identical outputs, always.
48
+
49
+ ## Models
50
+
51
+ ### `Severity` (str enum)
52
+ `"warning"` | `"block"`. Only `"block"` (after exception resolution) sets `ValidationReport.blocked`.
53
+
54
+ ### `Violation`
55
+ | field | type | meaning |
56
+ |---|---|---|
57
+ | `rule_id` | `str` | key into the rule registry (`matrx_content_guard.rules.RULES`) |
58
+ | `node_path` | `str` | HTML: lxml xpath (`/html/body/div[2]/script[1]`, `.../@href` for an attribute). CSS/JS: `"<prefix>:<line>:<col>"`. |
59
+ | `excerpt` | `str` | ≤200 chars, the offending snippet |
60
+ | `severity` | `Severity` | resolved for the profile that produced it |
61
+ | `fix_hint` | `str` | plain-English, actionable |
62
+
63
+ ### `ValidationReport`
64
+ | field | type | meaning |
65
+ |---|---|---|
66
+ | `blocked` | `bool` | `True` iff `violations` is non-empty after exceptions are applied |
67
+ | `violations` | `list[Violation]` | BLOCK-severity, NOT excepted — these are what actually blocks |
68
+ | `warnings` | `list[Violation]` | WARNING-severity — never blocks, always surfaced |
69
+ | `excepted` | `list[Violation]` | BLOCK-severity findings suppressed by a matching `ContentException` — **never silently dropped**, always visible here for audit (extension beyond the brief's literal two-list spec, kept because "loud recovery, no silent defaults" is a repo-wide mandate) |
70
+ | `profile` | `str` | the resolved `"name:variant"` key that was actually applied |
71
+
72
+ ### Patch models
73
+ - `SearchReplaceHunk { search: str, replace: str }`
74
+ - `SearchReplacePatch { format: "search_replace", hunks: list[SearchReplaceHunk] }`
75
+ - `UnifiedDiffPatch { format: "unified_diff", diff: str }` — a standard unified-diff string; converted
76
+ internally to search/replace hunks (context+removed -> context+added) and applied through the
77
+ identical matcher, so both formats give identical failure-context quality.
78
+ - `Patch = SearchReplacePatch | UnifiedDiffPatch`
79
+ - `PatchFailureReason` (str enum): `"no_match"` | `"ambiguous_match"` | `"malformed_diff"`
80
+ - `PatchFailure { hunk_index, reason, match_count, positions: list[int], closest_match: str|None, closest_match_position: int|None, closest_match_similarity: float|None, message }`
81
+ - `PatchStats { hunks_total, hunks_applied, chars_before, chars_after, lines_changed }`
82
+ - `PatchResult { applied: bool, result: str|None, failures: list[PatchFailure], stats: PatchStats, dry_run: bool }`
83
+ — `result` is `None` unless **every** hunk applied (all-or-nothing). `dry_run` is echoed from the
84
+ call for the caller's audit trail; `apply_patch` has no I/O either way, so it doesn't change what's
85
+ computed — only the caller's own write-boundary decides whether to persist `result`.
86
+ - `ValidatePatchResult { patch: PatchResult, validation: ValidationReport|None }` — `validation` is
87
+ `None` iff the patch itself failed to apply (nothing to validate).
88
+
89
+ ### `ContentException` (the F3 exception/approval data shape)
90
+ | field | type | engine-checked? | meaning |
91
+ |---|---|---|---|
92
+ | `rule_id` | `str` | yes | exact match required |
93
+ | `scope_site_id` | `str\|None` | **no** | store-schema/informational only — see below |
94
+ | `scope_page_id` | `str\|None` | **no** | store-schema/informational only — see below |
95
+ | `match_node_path_prefix` | `str\|None` | yes | violation's `node_path` must start with this |
96
+ | `match_excerpt_contains` | `str\|None` | yes | violation's `excerpt` must contain this substring |
97
+ | `status` | `str` | yes | only `"approved"` is ever honored |
98
+ | `note` | `str\|None` | no | free text |
99
+
100
+ **Scoping is two-tier.** Site/page granularity is resolved by the CALLER, not this engine: P1's
101
+ service layer knows which site/page it's writing to, queries only the exception rows in scope for
102
+ that write, and passes just those into `validate_content`/`validate_patch`. The engine never sees a
103
+ site_id/page_id and never filters on them. Rule + node granularity IS engine-checked: `rule_id` is
104
+ always exact; `match_node_path_prefix`/`match_excerpt_contains` optionally narrow an already-in-scope
105
+ exception to one subtree/snippet instead of blanket-suppressing the rule everywhere in the content.
106
+
107
+ **Store:** a small table on the CMS project (`viyklljfdhtidwecakwx`), owned by P1, schema coordinated
108
+ against this shape — migration + shared ledger rules apply (master plan §6.10). **Review/approve UI:**
109
+ P5. This package owns only the shape and the matching semantics
110
+ (`matrx_content_guard.exceptions.exception_matches` / `apply_exceptions`).
111
+
112
+ ## Profiles
113
+
114
+ Five names × two variants = ten resolvable keys. A bare name resolves to `:standard`.
115
+
116
+ | name | content_kind | is_fragment | maps to (via `profile_for`) |
117
+ |---|---|---|---|
118
+ | `html_page_document` | full HTML document | no | `html_pages.html_content` |
119
+ | `cms_page_fragment` | HTML fragment | yes | `client_pages.html_content[_draft]` |
120
+ | `cms_component_fragment` | HTML fragment | yes | `client_components.html_content[_draft]` |
121
+ | `css` | CSS text | n/a | `client_pages.css_content[_draft]`, `client_components.css_content[_draft]`, `client_sites.global_css` |
122
+ | `js` | JS text | n/a | `client_pages.js_content[_draft]` |
123
+
124
+ `standard` is permissive-first (F3): every rule's default severity was calibrated against the live
125
+ corpus (239 `html_pages` + 19 `client_pages` rows, audited 2026-07-09) so **inline `<script>` (73
126
+ rows), event-handler attributes (40 rows), and external `<iframe>`/`<script src>` embeds (7 rows) all
127
+ validate `blocked=false`.** `strict` today makes exactly one change: `html.event_handler_attribute`
128
+ escalates from WARNING to BLOCK (for a future tighter surface, e.g. a public template gallery — no
129
+ current caller uses `strict`). Profiles are pure data (`matrx_content_guard/profiles.py`) — a future
130
+ policy change is a data edit, never an engine change.
131
+
132
+ ## Rule registry
133
+
134
+ Every rule a violation can cite lives in `matrx_content_guard.rules.RULES`, one entry per `rule_id`,
135
+ each with a `category`, a ≤140-char plain-English `summary`, a `fix_hint`, and a `default_severity`.
136
+ `describe_policy(profile)` resolves each rule's *actual* severity for that profile/variant. See the
137
+ module for the full, current list (17 rules as of 2026-07-09) — adding a rule is one dict entry plus
138
+ the check that fires it; there is no second place severities live.
139
+
140
+ **Hard-block minimum (BLOCK in every variant, zero live false positives as of 2026-07-09):**
141
+ `html.denied_tag` (`<base>` only today), `html.dangerous_url_scheme` (`javascript:`/`vbscript:` in
142
+ any URL-bearing attribute, browser-style scheme-sniffing so `java\tscript:` doesn't bypass it),
143
+ `html.iframe_html_data_uri`, `html.css_expression_injection`, `html.css_javascript_url`,
144
+ `css.expression_injection`, `css.javascript_url`.
145
+
146
+ **Permissive/warn-only (never blocks under `standard`):** inline event handlers, external
147
+ script/iframe/form/link origins, `@import`/`url()` external origins, parser-recovered-from-malformed-
148
+ markup, and the entire `js.*` family (pattern-tier lint — `eval`, `document.write`, `new Function`,
149
+ `innerHTML=` — **JS static analysis is an explicit non-goal**; these never escalate to BLOCK in any
150
+ variant).
151
+
152
+ ## Parser / engine boundaries (read before extending)
153
+
154
+ - **HTML:** `lxml.html` with `recover=True`. Validation walks the PARSED tree for reporting ONLY.
155
+ `apply_patch` never parses anything — pure string search/replace. A byte in storage never
156
+ round-trips through the HTML parser on write; this is the hard byte-preservation guarantee.
157
+ - **Fragments vs documents:** each profile declares `is_fragment`; fragments parse via
158
+ `lxml.html.fragment_fromstring(..., create_parent="div")` (always succeeds, one synthetic root for
159
+ stable xpaths), documents via `lxml.html.document_fromstring` (fills in missing `<html>`/`<body>`
160
+ when absent — covers the one live `html_pages` row with no doctype).
161
+ - **CSS:** bounded-regex pattern scans (`expression()`, `url()`, `@import`), not a full CSS parser.
162
+ Every pattern uses only bounded/non-nested quantifiers — no catastrophic-backtracking path — and
163
+ `PATTERN_SCAN_MAX_CHARS` (2,000,000) caps worst-case work regardless of document size, firing a
164
+ `content.pattern_scan_truncated` WARNING rather than silently skipping.
165
+ - **JS:** pattern-tier lint only, explicitly bounded — no parser, no AST, never BLOCK. Documented
166
+ non-goal per the brief.
167
+
168
+ ## Wiring status (2026-07-09, live end-to-end as of this commit)
169
+
170
+ - **P2 is wired and live.** `aidream/tools/_cms_shared.py::content_guard()` lazily imports
171
+ `matrx_content_guard` and finds every C3 entry point (verified: `cms_inspect_tool.py` calls
172
+ `guard.describe_policy(...)` / `guard.PROFILE_NAMES`; `cms_page_tool.py`'s dry-run patch preview
173
+ calls `guard.profile_for("cms", "page", field)` and `guard.validate_patch(...)` — both using this
174
+ package's canonical vocabulary directly, no hand-translation).
175
+ - **P1's write-boundary hook is wired and live** — `aidream/package_integration.py::_configure_cms()`
176
+ imports `matrx_content_guard` and calls `configure_cms(content_validator=_ContentGuardAdapter(),
177
+ profile_resolver=matrx_content_guard.profile_for)`, replacing P1's interim placeholders wholesale as
178
+ their own docstrings said this package would. `_ContentGuardAdapter.validate_content` is a one-line
179
+ delegation to `matrx_content_guard.validate_content` — every CMS content-bearing write now fails
180
+ closed on a blocked report via `enforce_content`, not just P2's dry-run preview. Verified: this
181
+ package's `validate_content`/`validate_patch` accept the exceptions shape P1's `ContentValidator`
182
+ Protocol types as `list[dict]` (coerced via `exceptions.coerce_exceptions` — added after confirming
183
+ a raw dict would otherwise crash with `AttributeError`, since no exception store exists yet to have
184
+ caught this in practice).
185
+ - This package is complete, installed (`packages/matrx-content-guard`, in the root `uv` workspace +
186
+ `pyproject.toml` deps), 128 tests passing, live-audited (0/280 real content fields blocked,
187
+ `AUDIT_RESULTS.md`), and adversarially reviewed (4 confirmed bypasses found and fixed — see
188
+ `FEATURE.md`'s "Security hardening" note) as of this commit.
@@ -0,0 +1,104 @@
1
+ # FEATURE.md — matrx-content-guard
2
+
3
+ > Durable design notes. The typed contract lives in `CONTRACT.md` (C3) — read that first for the
4
+ > API. This file is the WHY behind the choices, for whoever extends this package next.
5
+
6
+ ## What this is
7
+
8
+ The deterministic validation + patch/diff engine every CMS agent mutation flows through
9
+ (`html_pages.html_content`, `client_pages.*_content[_draft]`, `client_components.*_content[_draft]`,
10
+ `client_sites.global_css`). Pure Python, zero DB access, zero tool-framework imports — a library, not
11
+ a service. Two jobs: (1) walk content and report policy violations, (2) apply agent-authored
12
+ search/replace or unified-diff patches to stored content.
13
+
14
+ ## Why permissive-first (F3)
15
+
16
+ Arman's explicit failure mode: rules so restrictive that nobody wants to build websites on the
17
+ platform. The live corpus (239 `html_pages` + 21 `client_pages` as of 2026-07-09, up from 19 at
18
+ brief-time — `client_pages` grows) already contains inline `<script>` (73 rows), event-handler
19
+ attributes (138 field-hits), and external `<iframe>`/`<script src>` embeds (107 field-hits) — all of
20
+ that is normal, legitimate content, not an attack. The hard-block list is therefore deliberately tiny
21
+ and picked for near-zero false-positive risk: a `javascript:`/`vbscript:` URL, a CSS
22
+ `expression()`/`url(javascript:...)`, an HTML-data-URI iframe, and the `<base>` tag. Everything else
23
+ (event handlers, external origins, JS pattern hits) is WARNING — visible, never blocking. See
24
+ `rules.py`'s module docstring for the full reasoning per rule.
25
+
26
+ ## Why lxml over html5lib
27
+
28
+ C-speed matters (some live rows are 200KB+ and get walked on every mutation), `recover=True` already
29
+ tolerates everything actually seen in production (including the one row with no `<!doctype>`/`<html>`
30
+ at all), and `getroottree().getpath(el)` gives a free, stable node path. html5lib's closer-to-a-real-
31
+ browser error correction wasn't worth the speed cost for content this well-behaved. If a future
32
+ pathological doc needs browser-exact recovery, that's the first thing to revisit.
33
+
34
+ ## Security hardening (2026-07-09, adversarial review)
35
+
36
+ An adversarial subagent review targeting bypass techniques against the hard-block rules found and
37
+ confirmed four real gaps, all fixed the same day (regression tests in the matching `test_*.py`):
38
+
39
+ 1. **`xlink:href` (any `prefix:href`/`prefix:src`) bypassed the hard-block entirely.** lxml's HTML
40
+ parser doesn't resolve XML namespaces, so `xlink:href` is literally the attribute name
41
+ `"xlink:href"` — comparing against `"href"` misses it completely. A classic real-world XSS bypass
42
+ technique targets exactly this gap. Fixed in `html_engine.py` by matching on the attribute's local
43
+ name (after the last `:`) instead of the raw name.
44
+ 2. **`<iframe srcdoc="...">` was entirely unscanned** despite being the same risk class as the
45
+ already-blocked `data:text/html` iframe `src` (both create an executing nested browsing context
46
+ from inline markup). Fixed by recursively scanning `srcdoc` content under the same profile
47
+ (`MAX_SRCDOC_DEPTH = 6` backstop against pathological nesting), rather than blanket-denying the
48
+ attribute — a legitimate `srcdoc` embed is still only flagged for what it actually contains.
49
+ 3. **CSS comments (`/**/`) and unicode escapes (`\65`) could split the literal `"expression("` /
50
+ `"javascript:"` strings past the hard-block regexes.** Real-world exploitability against THIS
51
+ platform's evergreen-browser-served pages is near-nil (modern browsers don't execute either
52
+ construct via CSS regardless), but the package's own contract calls these a hard BLOCK, so
53
+ `css_engine.py::_deobfuscate` now strips comments and decodes CSS unicode escapes before matching
54
+ — belt-and-suspenders for any non-evergreen consumer (legacy webviews, PDF renderers, email
55
+ clients) that might still honor them.
56
+ 4. **An exception's `match_node_path_prefix`/`match_excerpt_contains` set to `""` silently behaved as
57
+ an always-match wildcard**, identical to a much broader "no constraint" — a likely footgun for any
58
+ UI/serializer that sends `""` instead of omitting an unset field, silently widening an approver's
59
+ intended narrow scope. `exceptions.py::exception_matches` now treats blank strings identically to
60
+ `None` in both fields.
61
+
62
+ Also hardened: `validate_content`/`validate_patch` now accept exceptions as plain `dict`s as well as
63
+ `ContentException` instances (`exceptions.coerce_exceptions`) — P1's own `ContentValidator` Protocol
64
+ types its `exceptions` param as `list[dict]`, and any real DB-backed exception store will round-trip
65
+ as dicts too; the strict-model-only version would have crashed with `AttributeError` the moment either
66
+ path carried a non-empty exceptions list.
67
+
68
+ ## The byte-preservation split (the one invariant that can't regress)
69
+
70
+ Validation and patching are two completely separate code paths that never touch each other's data:
71
+ `html_engine.py`/`css_engine.py`/`js_engine.py` PARSE content (via lxml / regex) to build a report —
72
+ the parsed tree is discarded, never serialized back out. `patch.py` operates on the raw string ONLY —
73
+ no parser anywhere in that file. A byte in storage that a patch didn't explicitly target is therefore
74
+ *guaranteed* unchanged, because nothing in the write path ever round-trips content through a parser.
75
+ If you're ever tempted to "normalize" or "pretty-print" content on write inside this package — don't;
76
+ that's exactly the bug this split exists to prevent (brief gotcha (a)).
77
+
78
+ ## Where the F3 exception store actually lives
79
+
80
+ This package only defines the shape (`ContentException`) and the matching semantics
81
+ (`exceptions.py`). The actual table is P1's (`aidream/services/cms/`, coordinated schema, migration +
82
+ ledger rules per master plan §6.10) and the approve/review UI is P5's. `validate_content` takes
83
+ already-scoped-by-the-caller exceptions — it never queries anything.
84
+
85
+ ## Coordination gap as of this commit (see CONTRACT.md "Wiring status")
86
+
87
+ P1 already built the consumer seam (`aidream/services/cms/validation.py` + `profiles.py`) with
88
+ interim, hand-mapped placeholders that predate this package. Wiring the real thing in is P1's job
89
+ (directory ownership), but there are two shape mismatches to close — documented in detail in
90
+ `CONTRACT.md`. P2 hasn't wired `cms_inspect.rules` (`describe_policy`) or dry-run (`validate_patch`)
91
+ yet either.
92
+
93
+ ## Extending this package
94
+
95
+ - **New rule:** one entry in `rules.py::RULES` (id, category, ≤140-char summary, fix_hint,
96
+ default_severity) + the check that fires it (in `html_engine.py`/`css_engine.py`/`js_engine.py`) +
97
+ a test in the matching `test_*_engine.py`. Never invent a second place for severity/copy.
98
+ - **New profile:** add to `profiles.py::_CONTENT_KIND_BY_NAME` (+ any `severity_overrides`) — profiles
99
+ are pure data, a policy change should never touch engine code.
100
+ - **New content field:** one entry in `field_map.py::_FIELD_PROFILE_MAP`. `profile_for` raises loudly
101
+ on anything unmapped — that's intentional, never add a silent default branch.
102
+ - **Re-running the live audit:** `scripts/audit_live_content.py` (needs `SUPABASE_CMS_*` env vars,
103
+ read-only) — re-proves the "0 blocked fields" guarantee as content grows; output committed to
104
+ `AUDIT_RESULTS.md`.
@@ -0,0 +1,46 @@
1
+ Metadata-Version: 2.4
2
+ Name: matrx-content-guard
3
+ Version: 0.1.0
4
+ Summary: Deterministic HTML/CSS/JS content validation + patch/diff editing engine for the Matrx CMS agent-authoring layer
5
+ Project-URL: Homepage, https://github.com/AI-Matrix-Engine/aidream-current
6
+ Project-URL: Repository, https://github.com/AI-Matrix-Engine/aidream-current
7
+ Author-email: Matrx <admin@aimatrx.com>
8
+ Maintainer-email: Matrx <admin@aimatrx.com>
9
+ License: MIT
10
+ Keywords: cms,content-validation,html-sanitization,matrx,patch-diff
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
17
+ Classifier: Topic :: Text Processing :: Markup :: HTML
18
+ Requires-Python: >=3.13
19
+ Requires-Dist: lxml>=5.0
20
+ Requires-Dist: pydantic>=2.0.0
21
+ Requires-Dist: tinycss2>=1.3
22
+ Description-Content-Type: text/markdown
23
+
24
+ # matrx-content-guard
25
+
26
+ Deterministic HTML/CSS/JS content validation + patch/diff editing for the
27
+ Matrx CMS agent-authoring layer. Pure Python, no DB access, no tool-framework
28
+ imports — every agent mutation to `html_pages` / `client_*` content flows
29
+ through this library before it's written.
30
+
31
+ See `CONTRACT.md` for the full typed API and `FEATURE.md` for the durable
32
+ design notes (parser choice, permissive-first policy rationale, byte-
33
+ preservation guarantee).
34
+
35
+ ```python
36
+ from matrx_content_guard import validate_content, apply_patch, SearchReplacePatch, SearchReplaceHunk
37
+
38
+ report = validate_content("<script>alert(1)</script>", "html_page_document")
39
+ report.blocked # False — inline script is a supported feature; only genuinely dangerous constructs block.
40
+
41
+ result = apply_patch(
42
+ "<h1>Hello</h1>",
43
+ SearchReplacePatch(hunks=[SearchReplaceHunk(search="Hello", replace="Hi")]),
44
+ )
45
+ result.result # "<h1>Hi</h1>"
46
+ ```
@@ -0,0 +1,23 @@
1
+ # matrx-content-guard
2
+
3
+ Deterministic HTML/CSS/JS content validation + patch/diff editing for the
4
+ Matrx CMS agent-authoring layer. Pure Python, no DB access, no tool-framework
5
+ imports — every agent mutation to `html_pages` / `client_*` content flows
6
+ through this library before it's written.
7
+
8
+ See `CONTRACT.md` for the full typed API and `FEATURE.md` for the durable
9
+ design notes (parser choice, permissive-first policy rationale, byte-
10
+ preservation guarantee).
11
+
12
+ ```python
13
+ from matrx_content_guard import validate_content, apply_patch, SearchReplacePatch, SearchReplaceHunk
14
+
15
+ report = validate_content("<script>alert(1)</script>", "html_page_document")
16
+ report.blocked # False — inline script is a supported feature; only genuinely dangerous constructs block.
17
+
18
+ result = apply_patch(
19
+ "<h1>Hello</h1>",
20
+ SearchReplacePatch(hunks=[SearchReplaceHunk(search="Hello", replace="Hi")]),
21
+ )
22
+ result.result # "<h1>Hi</h1>"
23
+ ```