graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,336 @@
|
|
|
1
|
+
# Semantic fragment sanitizer — converts sentence-like rationale nodes into
|
|
2
|
+
# attributes on related nodes and removes invalid file_type values.
|
|
3
|
+
#
|
|
4
|
+
# Called from the skill merge path (see skill-devin.md) and from the in-process
|
|
5
|
+
# `graphify merge-chunks` command — both ingest untrusted agent-written chunk
|
|
6
|
+
# JSON, and validate_semantic_fragment() rejects malformed/oversized payloads and
|
|
7
|
+
# crafted node/edge IDs before they touch the graph. The primary build/load paths
|
|
8
|
+
# (build_from_json, load_graph_json) deliberately do NOT run this: they must keep
|
|
9
|
+
# loading valid pre-existing graphs whose AST node IDs predate the stricter
|
|
10
|
+
# semantic-ID charset.
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import re
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from .build import _normalize_hyperedge_members
|
|
18
|
+
|
|
19
|
+
# Labels longer than this many characters, or containing >= this many words,
|
|
20
|
+
# are candidates for being sentence-like rationale text rather than entity names.
|
|
21
|
+
_RATIONALE_MIN_CHARS = 80
|
|
22
|
+
_RATIONALE_MIN_WORDS = 8
|
|
23
|
+
|
|
24
|
+
# Validation limits for untrusted semantic-fragment payloads. See
|
|
25
|
+
# validate_semantic_fragment(). Issue #825: returned-JSON normalization for
|
|
26
|
+
# OpenCode and Codex agents requires a Python enforcement boundary so a
|
|
27
|
+
# malicious or runaway agent response cannot exhaust memory or escape the
|
|
28
|
+
# graphify-out chunk directory via crafted node/edge IDs.
|
|
29
|
+
MAX_SEMANTIC_FRAGMENT_BYTES = 25 * 1024 * 1024
|
|
30
|
+
MAX_SEMANTIC_FRAGMENT_NODES = 10_000
|
|
31
|
+
MAX_SEMANTIC_FRAGMENT_EDGES = 100_000
|
|
32
|
+
MAX_SEMANTIC_FRAGMENT_HYPEREDGES = 10_000
|
|
33
|
+
MAX_SEMANTIC_HYPEREDGE_NODES = 256
|
|
34
|
+
MAX_SEMANTIC_ID_LENGTH = 256
|
|
35
|
+
VALID_SEMANTIC_FILE_TYPES = frozenset({"code", "document", "paper", "image", "rationale", "concept"})
|
|
36
|
+
# Unicode word characters are allowed: build's normalize_id preserves CJK /
|
|
37
|
+
# Cyrillic / accented-Latin identifiers, so an ASCII-only gate would reject valid
|
|
38
|
+
# ids that the loader accepts. The explicit path-separator / ".." check in
|
|
39
|
+
# _validate_semantic_id still blocks directory escape (#825); "/", "\\", spaces,
|
|
40
|
+
# "@", "#" etc. are not \w and remain rejected.
|
|
41
|
+
_SEMANTIC_ID_RE = re.compile(r"^[\w.:-]+$")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def validate_semantic_fragment(fragment: object) -> list[str]:
|
|
45
|
+
"""Return validation errors for an untrusted semantic extraction fragment.
|
|
46
|
+
|
|
47
|
+
Empty list means valid. Called by skill merge code before
|
|
48
|
+
sanitize_semantic_fragment() so malformed or malicious agent JSON is
|
|
49
|
+
rejected before it touches the graph. Parameter is `object` (not `dict`)
|
|
50
|
+
because we may be handed arbitrary deserialized JSON — the first check
|
|
51
|
+
rejects anything that isn't a dict.
|
|
52
|
+
"""
|
|
53
|
+
if not isinstance(fragment, dict):
|
|
54
|
+
return ["fragment must be a JSON object"]
|
|
55
|
+
|
|
56
|
+
errors: list[str] = []
|
|
57
|
+
try:
|
|
58
|
+
payload = json.dumps(fragment, ensure_ascii=False).encode("utf-8")
|
|
59
|
+
except (TypeError, ValueError) as exc:
|
|
60
|
+
return [f"fragment is not JSON-serializable: {exc}"]
|
|
61
|
+
|
|
62
|
+
if len(payload) > MAX_SEMANTIC_FRAGMENT_BYTES:
|
|
63
|
+
errors.append(f"payload is {len(payload)} bytes; max is {MAX_SEMANTIC_FRAGMENT_BYTES}")
|
|
64
|
+
|
|
65
|
+
nodes = fragment.get("nodes", [])
|
|
66
|
+
edges = fragment.get("edges", [])
|
|
67
|
+
if not isinstance(nodes, list):
|
|
68
|
+
errors.append("nodes must be a list")
|
|
69
|
+
nodes = []
|
|
70
|
+
elif len(nodes) > MAX_SEMANTIC_FRAGMENT_NODES:
|
|
71
|
+
errors.append(f"nodes has {len(nodes)} entries; max is {MAX_SEMANTIC_FRAGMENT_NODES}")
|
|
72
|
+
|
|
73
|
+
if not isinstance(edges, list):
|
|
74
|
+
errors.append("edges must be a list")
|
|
75
|
+
edges = []
|
|
76
|
+
elif len(edges) > MAX_SEMANTIC_FRAGMENT_EDGES:
|
|
77
|
+
errors.append(f"edges has {len(edges)} entries; max is {MAX_SEMANTIC_FRAGMENT_EDGES}")
|
|
78
|
+
|
|
79
|
+
for i, node in enumerate(nodes):
|
|
80
|
+
if not isinstance(node, dict):
|
|
81
|
+
errors.append(f"nodes[{i}] must be an object")
|
|
82
|
+
continue
|
|
83
|
+
_validate_semantic_id(errors, f"nodes[{i}].id", node.get("id"))
|
|
84
|
+
# file_type is intentionally NOT rejected here. It carries no security
|
|
85
|
+
# risk (it can't exhaust memory or escape a directory), and
|
|
86
|
+
# build_from_json already coerces every value via _FILE_TYPE_SYNONYMS
|
|
87
|
+
# (unknown -> "concept", #840). Rejecting a whole chunk over a synonym
|
|
88
|
+
# like "markdown"/"tool"/"framework" that the loader would happily map is
|
|
89
|
+
# pure data loss, so leave file_type normalization to build.
|
|
90
|
+
|
|
91
|
+
for i, edge in enumerate(edges):
|
|
92
|
+
if not isinstance(edge, dict):
|
|
93
|
+
errors.append(f"edges[{i}] must be an object")
|
|
94
|
+
continue
|
|
95
|
+
_validate_semantic_id(errors, f"edges[{i}].source", edge.get("source"))
|
|
96
|
+
_validate_semantic_id(errors, f"edges[{i}].target", edge.get("target"))
|
|
97
|
+
|
|
98
|
+
hyperedges = fragment.get("hyperedges", [])
|
|
99
|
+
if hyperedges is None:
|
|
100
|
+
hyperedges = []
|
|
101
|
+
if not isinstance(hyperedges, list):
|
|
102
|
+
errors.append("hyperedges must be a list")
|
|
103
|
+
else:
|
|
104
|
+
if len(hyperedges) > MAX_SEMANTIC_FRAGMENT_HYPEREDGES:
|
|
105
|
+
errors.append(
|
|
106
|
+
f"hyperedges has {len(hyperedges)} entries; "
|
|
107
|
+
f"max is {MAX_SEMANTIC_FRAGMENT_HYPEREDGES}"
|
|
108
|
+
)
|
|
109
|
+
for i, he in enumerate(hyperedges):
|
|
110
|
+
if not isinstance(he, dict):
|
|
111
|
+
errors.append(f"hyperedges[{i}] must be an object")
|
|
112
|
+
continue
|
|
113
|
+
# Fold alias member keys (members/node_ids) onto `nodes` (#1561) so
|
|
114
|
+
# an alias-keyed hyperedge isn't rejected here for "nodes must be a
|
|
115
|
+
# list" before it ever reaches build's normalization.
|
|
116
|
+
_normalize_hyperedge_members(he)
|
|
117
|
+
_validate_semantic_id(errors, f"hyperedges[{i}].id", he.get("id"))
|
|
118
|
+
he_nodes = he.get("nodes")
|
|
119
|
+
if not isinstance(he_nodes, list):
|
|
120
|
+
errors.append(f"hyperedges[{i}].nodes must be a list")
|
|
121
|
+
continue
|
|
122
|
+
if len(he_nodes) > MAX_SEMANTIC_HYPEREDGE_NODES:
|
|
123
|
+
errors.append(
|
|
124
|
+
f"hyperedges[{i}].nodes has {len(he_nodes)} entries; "
|
|
125
|
+
f"max is {MAX_SEMANTIC_HYPEREDGE_NODES}"
|
|
126
|
+
)
|
|
127
|
+
for j, ref in enumerate(he_nodes):
|
|
128
|
+
_validate_semantic_id(errors, f"hyperedges[{i}].nodes[{j}]", ref)
|
|
129
|
+
|
|
130
|
+
return errors
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def load_validated_semantic_fragment(path: Path) -> tuple[dict | None, list[str]]:
|
|
134
|
+
"""Load and validate a semantic chunk, rejecting oversize files before parsing.
|
|
135
|
+
|
|
136
|
+
The size guard runs against `path.stat().st_size` so an attacker-supplied
|
|
137
|
+
multi-gigabyte chunk file cannot blow up memory at `read_text()` time.
|
|
138
|
+
JSON decode errors are returned as validation errors rather than raised,
|
|
139
|
+
so callers can `continue` past bad chunks without a try/except.
|
|
140
|
+
"""
|
|
141
|
+
try:
|
|
142
|
+
size = path.stat().st_size
|
|
143
|
+
except OSError as exc:
|
|
144
|
+
return None, [f"could not stat {path}: {exc}"]
|
|
145
|
+
if size > MAX_SEMANTIC_FRAGMENT_BYTES:
|
|
146
|
+
return None, [f"payload is {size} bytes; max is {MAX_SEMANTIC_FRAGMENT_BYTES}"]
|
|
147
|
+
try:
|
|
148
|
+
fragment = json.loads(path.read_text(encoding="utf-8"))
|
|
149
|
+
except json.JSONDecodeError as exc:
|
|
150
|
+
return None, [f"invalid JSON: {exc}"]
|
|
151
|
+
except OSError as exc:
|
|
152
|
+
return None, [f"could not read {path}: {exc}"]
|
|
153
|
+
errors = validate_semantic_fragment(fragment)
|
|
154
|
+
return (None, errors) if errors else (fragment, [])
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _validate_semantic_id(errors: list[str], field: str, value: object) -> None:
|
|
158
|
+
if not isinstance(value, str):
|
|
159
|
+
errors.append(f"{field} must be a string")
|
|
160
|
+
return
|
|
161
|
+
if not value:
|
|
162
|
+
errors.append(f"{field} must not be empty")
|
|
163
|
+
return
|
|
164
|
+
if len(value) > MAX_SEMANTIC_ID_LENGTH:
|
|
165
|
+
errors.append(f"{field} is {len(value)} chars; max is {MAX_SEMANTIC_ID_LENGTH}")
|
|
166
|
+
if "/" in value or "\\" in value or ".." in value:
|
|
167
|
+
errors.append(f"{field} must not contain path separators or '..'")
|
|
168
|
+
if not _SEMANTIC_ID_RE.fullmatch(value):
|
|
169
|
+
errors.append(f"{field} contains unsupported characters")
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def sanitize_semantic_fragment(fragment: dict) -> dict:
|
|
173
|
+
"""Clean up a semantic extraction fragment in-place.
|
|
174
|
+
|
|
175
|
+
Operations:
|
|
176
|
+
1. Removes nodes with ``file_type: "rationale"`` or ``file_type: "concept"``
|
|
177
|
+
that were emitted by an LLM (these are not valid semantic entity types).
|
|
178
|
+
2. Detects nodes whose label reads like a sentence / rationale paragraph
|
|
179
|
+
AND that participate in a ``rationale_for`` edge, then converts the
|
|
180
|
+
label into a ``rationale`` attribute on the target node and removes
|
|
181
|
+
the source-node + its edges. The ``rationale_for`` edge signal applies
|
|
182
|
+
regardless of the source node's ``file_type`` — sentence-like nodes
|
|
183
|
+
with allowed types (``document``, ``code``) are still cleaned up when
|
|
184
|
+
they're explicitly marked as rationale.
|
|
185
|
+
3. Strips nodes whose only distinguishing field is the label itself
|
|
186
|
+
(empty id — likely LLM hallucination).
|
|
187
|
+
4. Filters hyperedges so they cannot reference removed or unknown node
|
|
188
|
+
IDs after the cleanup passes above. A hyperedge with fewer than two
|
|
189
|
+
surviving members is dropped.
|
|
190
|
+
|
|
191
|
+
Returns the same dict for convenience.
|
|
192
|
+
"""
|
|
193
|
+
_invalid_ft = frozenset({"rationale", "concept"})
|
|
194
|
+
|
|
195
|
+
nodes: list[dict] = fragment.get("nodes", [])
|
|
196
|
+
edges: list[dict] = fragment.get("edges", [])
|
|
197
|
+
hyperedges: list[dict] = fragment.get("hyperedges", []) or []
|
|
198
|
+
|
|
199
|
+
# ---- build lookup maps --------------------------------------------------
|
|
200
|
+
node_by_id: dict[str, dict] = {}
|
|
201
|
+
for n in nodes:
|
|
202
|
+
nid = n.get("id", "")
|
|
203
|
+
if nid:
|
|
204
|
+
node_by_id[nid] = n
|
|
205
|
+
|
|
206
|
+
# Pre-collect node IDs that source a `rationale_for` edge — these are
|
|
207
|
+
# candidates for sentence-like cleanup even when file_type is allowed.
|
|
208
|
+
rationale_for_sources: set[str] = set()
|
|
209
|
+
for e in edges:
|
|
210
|
+
if e.get("relation") == "rationale_for":
|
|
211
|
+
src = e.get("source", "")
|
|
212
|
+
if src:
|
|
213
|
+
rationale_for_sources.add(src)
|
|
214
|
+
|
|
215
|
+
# ---- pass 1: identify nodes to remove + rationale candidates -----------
|
|
216
|
+
rationale_candidates: list[dict] = []
|
|
217
|
+
remove_ids: set[str] = set()
|
|
218
|
+
keep_nodes: list[dict] = []
|
|
219
|
+
for n in nodes:
|
|
220
|
+
nid = n.get("id", "")
|
|
221
|
+
if not nid:
|
|
222
|
+
# Node without an id cannot be referenced — discard.
|
|
223
|
+
continue
|
|
224
|
+
ft = n.get("file_type", "")
|
|
225
|
+
label = n.get("label", "")
|
|
226
|
+
if ft in _invalid_ft:
|
|
227
|
+
# Explicitly-invalid file_type ("rationale" or "concept"): if
|
|
228
|
+
# the label looks like a sentence we may convert to attribute.
|
|
229
|
+
if _is_sentence_like_rationale_label(label):
|
|
230
|
+
rationale_candidates.append(n)
|
|
231
|
+
remove_ids.add(nid)
|
|
232
|
+
continue
|
|
233
|
+
if nid in rationale_for_sources and _is_sentence_like_rationale_label(label):
|
|
234
|
+
# Allowed file_type, but the node sources a `rationale_for` edge
|
|
235
|
+
# AND its label is sentence-like prose. Treat it as rationale
|
|
236
|
+
# cleanup material rather than a real graph entity.
|
|
237
|
+
rationale_candidates.append(n)
|
|
238
|
+
remove_ids.add(nid)
|
|
239
|
+
continue
|
|
240
|
+
keep_nodes.append(n)
|
|
241
|
+
|
|
242
|
+
# ---- pass 2: convert sentence-nodes → rationale attributes --------------
|
|
243
|
+
# Only `rationale_for` edges propagate the rationale text. Other outgoing
|
|
244
|
+
# edges (e.g. references, conceptually_related_to) are NOT used as
|
|
245
|
+
# attribute-propagation paths — that would corrupt unrelated nodes by
|
|
246
|
+
# attaching rationale meant for a different target.
|
|
247
|
+
rationale_attrs: dict[str, list[str]] = {}
|
|
248
|
+
for rn in rationale_candidates:
|
|
249
|
+
rn_id = rn.get("id", "")
|
|
250
|
+
text = rn.get("label", "").strip()
|
|
251
|
+
for e in edges:
|
|
252
|
+
if e.get("relation") != "rationale_for":
|
|
253
|
+
continue
|
|
254
|
+
if e.get("source") != rn_id:
|
|
255
|
+
continue
|
|
256
|
+
target_id = e.get("target")
|
|
257
|
+
if target_id not in node_by_id or target_id in remove_ids:
|
|
258
|
+
continue
|
|
259
|
+
rationale_attrs.setdefault(target_id, []).append(text)
|
|
260
|
+
|
|
261
|
+
for target_id, texts in rationale_attrs.items():
|
|
262
|
+
if target_id in node_by_id and target_id not in remove_ids:
|
|
263
|
+
_append_rationale_attr(node_by_id[target_id], texts)
|
|
264
|
+
|
|
265
|
+
# ---- pass 3: strip edges referencing removed nodes ----------------------
|
|
266
|
+
keep_edges: list[dict] = []
|
|
267
|
+
for e in edges:
|
|
268
|
+
src = e.get("source", "")
|
|
269
|
+
tgt = e.get("target", "")
|
|
270
|
+
if src in remove_ids or tgt in remove_ids:
|
|
271
|
+
continue
|
|
272
|
+
keep_edges.append(e)
|
|
273
|
+
|
|
274
|
+
# ---- pass 4: filter hyperedges to surviving node IDs --------------------
|
|
275
|
+
surviving_ids: set[str] = {n.get("id", "") for n in keep_nodes}
|
|
276
|
+
surviving_ids.discard("")
|
|
277
|
+
keep_hyperedges: list[dict] = []
|
|
278
|
+
for he in hyperedges:
|
|
279
|
+
if not isinstance(he, dict):
|
|
280
|
+
continue
|
|
281
|
+
# Fold alias member keys (members/node_ids) onto `nodes` (#1561) so an
|
|
282
|
+
# alias-keyed hyperedge isn't silently dropped below for a missing
|
|
283
|
+
# `nodes` list before build can canonicalize it.
|
|
284
|
+
_normalize_hyperedge_members(he)
|
|
285
|
+
he_nodes = he.get("nodes")
|
|
286
|
+
if not isinstance(he_nodes, list):
|
|
287
|
+
continue
|
|
288
|
+
filtered = [ref for ref in he_nodes if isinstance(ref, str) and ref in surviving_ids]
|
|
289
|
+
if len(filtered) < 2:
|
|
290
|
+
# A hyperedge needs at least two surviving members to be meaningful.
|
|
291
|
+
continue
|
|
292
|
+
if len(filtered) != len(he_nodes):
|
|
293
|
+
he = dict(he)
|
|
294
|
+
he["nodes"] = filtered
|
|
295
|
+
keep_hyperedges.append(he)
|
|
296
|
+
|
|
297
|
+
fragment["nodes"] = keep_nodes
|
|
298
|
+
fragment["edges"] = keep_edges
|
|
299
|
+
fragment["hyperedges"] = keep_hyperedges
|
|
300
|
+
return fragment
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _is_sentence_like_rationale_label(label: str) -> bool:
|
|
304
|
+
"""Return True if *label* looks like prose / rationale text rather than an
|
|
305
|
+
entity or concept name.
|
|
306
|
+
|
|
307
|
+
Heuristics (no false positives on short-concept-edge-cases):
|
|
308
|
+
- Longer than *_RATIONALE_MIN_CHARS* chars, OR
|
|
309
|
+
- At least *_RATIONALE_MIN_WORDS* whitespace-delimited tokens, AND
|
|
310
|
+
- Contains at least one sentence-ending punctuation mark (``. ! ?``) or a
|
|
311
|
+
colon (common in "Decision: ..." rationales).
|
|
312
|
+
"""
|
|
313
|
+
if not label:
|
|
314
|
+
return False
|
|
315
|
+
label = label.strip()
|
|
316
|
+
if len(label) < _RATIONALE_MIN_CHARS:
|
|
317
|
+
word_count = len(label.split())
|
|
318
|
+
if word_count < _RATIONALE_MIN_WORDS:
|
|
319
|
+
return False
|
|
320
|
+
# Must look like actual prose: has sentence-ending punctuation or a colon.
|
|
321
|
+
return bool(re.search(r"[.!?:]", label))
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _append_rationale_attr(node: dict, texts: list[str]) -> None:
|
|
325
|
+
"""Append one or more rationale strings to *node*'s ``rationale`` attribute.
|
|
326
|
+
|
|
327
|
+
If the attribute already exists the new texts are appended with a
|
|
328
|
+
double-newline separator so downstream consumers can distinguish distinct
|
|
329
|
+
rationale fragments.
|
|
330
|
+
"""
|
|
331
|
+
existing = node.get("rationale", "")
|
|
332
|
+
new_text = "\n\n".join(texts).strip()
|
|
333
|
+
if existing:
|
|
334
|
+
node["rationale"] = existing + "\n\n" + new_text
|
|
335
|
+
else:
|
|
336
|
+
node["rationale"] = new_text
|