graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/ids.py
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""Single source of truth for node-ID normalization.
|
|
2
|
+
|
|
3
|
+
Three independent producers must agree on node IDs or the graph splits a single
|
|
4
|
+
entity into disconnected ghost nodes:
|
|
5
|
+
|
|
6
|
+
1. The AST extractor (``extract._make_id``) — deterministic, per-language.
|
|
7
|
+
2. The semantic subagents (LLM) — follow the node-ID spec in the skill prompt.
|
|
8
|
+
3. The graph builder (``build._normalize_id``) — reconciles edge endpoints when
|
|
9
|
+
the LLM emits IDs with slightly different punctuation or casing than the AST.
|
|
10
|
+
|
|
11
|
+
Historically the normalization recipe was copy-pasted into ``extract._make_id``
|
|
12
|
+
and ``build._normalize_id`` and kept in sync only by mirrored docstrings, which
|
|
13
|
+
is exactly how the recurring ID-drift bug class crept in (#811 Unicode collapse,
|
|
14
|
+
#550 same-filename collisions, #1033 AST-vs-LLM file-node mismatch, #1104). This
|
|
15
|
+
module exists so the recipe lives in one place and the two callers can no longer
|
|
16
|
+
diverge.
|
|
17
|
+
|
|
18
|
+
The recipe: iterate ``casefold`` then NFKC-normalize to a fixpoint (casefold can
|
|
19
|
+
*expand* a character into a base letter plus a combining mark — ``İ`` -> ``i`` +
|
|
20
|
+
U+0307 — and NFKC then recomposes what can be recomposed; because the two do not
|
|
21
|
+
commute and neither is a fixpoint of the other, a single pass is not
|
|
22
|
+
caseless-stable), then replace runs of non-word characters with a single
|
|
23
|
+
underscore (``re.UNICODE`` so CJK/Cyrillic/Arabic/accented-Latin letters survive
|
|
24
|
+
instead of collapsing to a per-file node), collapse repeated underscores, then
|
|
25
|
+
strip leading/trailing underscores.
|
|
26
|
+
|
|
27
|
+
Casefolding runs BEFORE the non-word filter, not after. With it last, the
|
|
28
|
+
combining marks casefold introduces were never filtered: ``İslemYap`` produced
|
|
29
|
+
``i̇slemyap`` — an id containing U+0307, which is not a ``\\w`` character — and a
|
|
30
|
+
second pass collapsed it to ``i_slemyap``, so the function was not idempotent
|
|
31
|
+
and the builder's re-normalization disagreed with the extractor's ``make_id``
|
|
32
|
+
for any Turkish identifier (#2614).
|
|
33
|
+
|
|
34
|
+
Casefolding runs in a FIXPOINT LOOP, not once. A single ``NFKC(casefold(...))``
|
|
35
|
+
left ``normalize_id(s) != normalize_id(s.casefold())`` for some combining-mark
|
|
36
|
+
sequences (Greek ypogegrammeni U+0345 followed by a combining accent):
|
|
37
|
+
pre-casefolding turns U+0345 into ``ι``, which NFKC composes with the accent into
|
|
38
|
+
a precomposed char the single pass never reached. Iterating to a fixpoint —
|
|
39
|
+
casefold first, on the raw input — makes the result caseless-stable regardless of
|
|
40
|
+
how many times the caller has already casefolded.
|
|
41
|
+
"""
|
|
42
|
+
from __future__ import annotations
|
|
43
|
+
|
|
44
|
+
import re
|
|
45
|
+
import unicodedata
|
|
46
|
+
|
|
47
|
+
__all__ = ["normalize_id", "make_id"]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def normalize_id(s: str) -> str:
|
|
51
|
+
r"""Normalize a single ID string to its canonical form.
|
|
52
|
+
|
|
53
|
+
Guarantees, all enforced by tests:
|
|
54
|
+
|
|
55
|
+
- Idempotent: ``normalize_id(normalize_id(s)) == normalize_id(s)``.
|
|
56
|
+
- The result contains only ``\w`` characters and ``_``.
|
|
57
|
+
- Caseless-stable: ``normalize_id(s) == normalize_id(s.casefold())``.
|
|
58
|
+
|
|
59
|
+
casefold and NFKC do not commute, and neither is a fixpoint of the other:
|
|
60
|
+
casefolding a char can expand it into a base letter plus a combining mark
|
|
61
|
+
(``İ`` -> ``i`` + U+0307), and NFKC can then recompose that mark with an
|
|
62
|
+
adjacent one into a different precomposed char. A single ``NFKC(casefold(...))``
|
|
63
|
+
pass therefore left ``normalize_id(s) != normalize_id(s.casefold())`` for some
|
|
64
|
+
combining-mark sequences (e.g. Greek ypogegrammeni U+0345 followed by a
|
|
65
|
+
combining accent): pre-casefolding turned U+0345 into ``ι`` which NFKC then
|
|
66
|
+
composed with the accent, reaching a form the single-pass recipe never saw.
|
|
67
|
+
|
|
68
|
+
So iterate ``casefold`` then ``NFKC`` to a fixpoint (casefold FIRST, on the
|
|
69
|
+
raw input, so a caller that pre-casefolds lands on the same fixpoint). The
|
|
70
|
+
loop is bounded — Unicode caseless folding converges in one or two steps —
|
|
71
|
+
with a hard cap as a termination guard. Only then apply the ``[^\w]+`` filter,
|
|
72
|
+
so every combining mark casefold introduced has been fully normalized before
|
|
73
|
+
it is filtered (#2614 and its combining-mark follow-on).
|
|
74
|
+
"""
|
|
75
|
+
cur = s
|
|
76
|
+
for _ in range(6):
|
|
77
|
+
nxt = unicodedata.normalize("NFKC", cur.casefold())
|
|
78
|
+
if nxt == cur:
|
|
79
|
+
break
|
|
80
|
+
cur = nxt
|
|
81
|
+
cur = re.sub(r"[^\w]+", "_", cur, flags=re.UNICODE)
|
|
82
|
+
cur = re.sub(r"_+", "_", cur)
|
|
83
|
+
return cur.strip("_")
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def make_id(*parts: str) -> str:
|
|
87
|
+
"""Build a canonical node ID from one or more name parts.
|
|
88
|
+
|
|
89
|
+
Parts are joined with ``_`` (after stripping stray ``_``/``.`` edges from each
|
|
90
|
+
part) and then run through :func:`normalize_id`, so the result is identical to
|
|
91
|
+
what the builder produces from the joined string.
|
|
92
|
+
"""
|
|
93
|
+
return normalize_id("_".join(p.strip("_.") for p in parts if p))
|
graphify/ingest.py
ADDED
|
@@ -0,0 +1,358 @@
|
|
|
1
|
+
# fetch URLs (tweet/arxiv/pdf/web) and save as annotated markdown
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
import uuid
|
|
6
|
+
import urllib.error
|
|
7
|
+
import urllib.parse
|
|
8
|
+
from datetime import datetime, timezone
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from graphify.security import safe_fetch, safe_fetch_text, validate_url
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _yaml_str(s: str) -> str:
|
|
15
|
+
"""Escape a string for embedding in a YAML double-quoted scalar.
|
|
16
|
+
|
|
17
|
+
Handles every YAML 1.1/1.2 line-break and control character that could
|
|
18
|
+
let a hostile value (e.g. a fetched page title) break out of the quoted
|
|
19
|
+
scalar and inject sibling YAML keys (F-009 / F-019). The previous
|
|
20
|
+
implementation missed `\\t`, `\\0`, the unicode line-separator U+2028 and
|
|
21
|
+
paragraph-separator U+2029 — all of which YAML treats as line breaks.
|
|
22
|
+
|
|
23
|
+
We intentionally do not depend on PyYAML (not in pyproject deps) and
|
|
24
|
+
instead emit safely-escaped double-quoted scalars by hand: the YAML
|
|
25
|
+
double-quoted form recognises `\\\\`, `\\"`, `\\n`, `\\r`, `\\t`, `\\0`,
|
|
26
|
+
`\\L` (U+2028), `\\P` (U+2029), and `\\xNN`/`\\uNNNN` numeric escapes.
|
|
27
|
+
"""
|
|
28
|
+
if s is None:
|
|
29
|
+
return ""
|
|
30
|
+
out: list[str] = []
|
|
31
|
+
for ch in str(s):
|
|
32
|
+
cp = ord(ch)
|
|
33
|
+
if ch == "\\":
|
|
34
|
+
out.append("\\\\")
|
|
35
|
+
elif ch == '"':
|
|
36
|
+
out.append('\\"')
|
|
37
|
+
elif ch == "\n":
|
|
38
|
+
out.append("\\n")
|
|
39
|
+
elif ch == "\r":
|
|
40
|
+
out.append("\\r")
|
|
41
|
+
elif ch == "\t":
|
|
42
|
+
out.append("\\t")
|
|
43
|
+
elif ch == "\0":
|
|
44
|
+
out.append("\\0")
|
|
45
|
+
elif cp == 0x2028:
|
|
46
|
+
out.append("\\L")
|
|
47
|
+
elif cp == 0x2029:
|
|
48
|
+
out.append("\\P")
|
|
49
|
+
elif cp < 0x20 or cp == 0x7F:
|
|
50
|
+
out.append(f"\\x{cp:02x}")
|
|
51
|
+
else:
|
|
52
|
+
out.append(ch)
|
|
53
|
+
return "".join(out)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _safe_filename(url: str, suffix: str) -> str:
|
|
57
|
+
"""Turn a URL into a safe filename."""
|
|
58
|
+
parsed = urllib.parse.urlparse(url)
|
|
59
|
+
name = parsed.netloc + parsed.path
|
|
60
|
+
name = re.sub(r"[^\w\-]", "_", name).strip("_")
|
|
61
|
+
name = re.sub(r"_+", "_", name)[:80]
|
|
62
|
+
return name + suffix
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _detect_url_type(url: str) -> str:
|
|
66
|
+
"""Classify the URL for targeted extraction."""
|
|
67
|
+
lower = url.lower()
|
|
68
|
+
if "twitter.com" in lower or "x.com" in lower:
|
|
69
|
+
return "tweet"
|
|
70
|
+
if "arxiv.org" in lower:
|
|
71
|
+
return "arxiv"
|
|
72
|
+
if "github.com" in lower:
|
|
73
|
+
return "github"
|
|
74
|
+
if "youtube.com" in lower or "youtu.be" in lower:
|
|
75
|
+
return "youtube"
|
|
76
|
+
parsed = urllib.parse.urlparse(url)
|
|
77
|
+
path = parsed.path.lower()
|
|
78
|
+
if path.endswith(".pdf"):
|
|
79
|
+
return "pdf"
|
|
80
|
+
if any(path.endswith(ext) for ext in (".png", ".jpg", ".jpeg", ".webp", ".gif")):
|
|
81
|
+
return "image"
|
|
82
|
+
return "webpage"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _fetch_html(url: str) -> str:
|
|
86
|
+
return safe_fetch_text(url)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _html_to_markdown(html: str, url: str) -> str:
|
|
90
|
+
"""Convert HTML to clean markdown. Uses markdownify if available, else basic strip."""
|
|
91
|
+
# Always pre-strip script/style so their text content never leaks into output
|
|
92
|
+
html = re.sub(r"<script[^>]*>.*?</script>", "", html, flags=re.DOTALL | re.IGNORECASE)
|
|
93
|
+
html = re.sub(r"<style[^>]*>.*?</style>", "", html, flags=re.DOTALL | re.IGNORECASE)
|
|
94
|
+
try:
|
|
95
|
+
from markdownify import markdownify
|
|
96
|
+
return markdownify(html, heading_style="ATX", bullets="-", strip=["img"])
|
|
97
|
+
except ImportError:
|
|
98
|
+
# Fallback: basic tag strip
|
|
99
|
+
text = re.sub(r"<[^>]+>", " ", html)
|
|
100
|
+
text = re.sub(r"\s+", " ", text).strip()
|
|
101
|
+
return text[:8000]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _fetch_tweet(url: str, author: str | None, contributor: str | None) -> tuple[str, str]:
|
|
105
|
+
"""Fetch a tweet URL. Returns (content, filename)."""
|
|
106
|
+
# Normalize to twitter.com for oEmbed
|
|
107
|
+
oembed_url = url.replace("x.com", "twitter.com")
|
|
108
|
+
oembed_api = f"https://publish.twitter.com/oembed?url={urllib.parse.quote(oembed_url)}&omit_script=true"
|
|
109
|
+
try:
|
|
110
|
+
data = json.loads(safe_fetch_text(oembed_api))
|
|
111
|
+
tweet_text = re.sub(r"<[^>]+>", "", data.get("html", "")).strip()
|
|
112
|
+
tweet_author = data.get("author_name", "unknown")
|
|
113
|
+
except Exception:
|
|
114
|
+
# oEmbed failed - save URL stub
|
|
115
|
+
tweet_text = f"Tweet at {url} (could not fetch content)"
|
|
116
|
+
tweet_author = "unknown"
|
|
117
|
+
|
|
118
|
+
now = datetime.now(timezone.utc).isoformat()
|
|
119
|
+
content = f"""---
|
|
120
|
+
source_url: "{_yaml_str(url)}"
|
|
121
|
+
type: tweet
|
|
122
|
+
author: "{_yaml_str(tweet_author)}"
|
|
123
|
+
captured_at: {now}
|
|
124
|
+
contributor: "{_yaml_str(contributor or author or 'unknown')}"
|
|
125
|
+
---
|
|
126
|
+
|
|
127
|
+
# Tweet by @{tweet_author}
|
|
128
|
+
|
|
129
|
+
{tweet_text}
|
|
130
|
+
|
|
131
|
+
Source: {url}
|
|
132
|
+
"""
|
|
133
|
+
filename = _safe_filename(url, ".md")
|
|
134
|
+
return content, filename
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _fetch_webpage(url: str, author: str | None, contributor: str | None) -> tuple[str, str]:
|
|
138
|
+
"""Fetch a generic webpage and convert to markdown."""
|
|
139
|
+
html = _fetch_html(url)
|
|
140
|
+
# Extract title
|
|
141
|
+
title_match = re.search(r"<title[^>]*>(.*?)</title>", html, re.IGNORECASE | re.DOTALL)
|
|
142
|
+
title = re.sub(r"\s+", " ", title_match.group(1)).strip() if title_match else url
|
|
143
|
+
|
|
144
|
+
markdown = _html_to_markdown(html, url)
|
|
145
|
+
now = datetime.now(timezone.utc).isoformat()
|
|
146
|
+
content = f"""---
|
|
147
|
+
source_url: "{_yaml_str(url)}"
|
|
148
|
+
type: webpage
|
|
149
|
+
title: "{_yaml_str(title)}"
|
|
150
|
+
captured_at: {now}
|
|
151
|
+
contributor: "{_yaml_str(contributor or author or 'unknown')}"
|
|
152
|
+
---
|
|
153
|
+
|
|
154
|
+
# {title}
|
|
155
|
+
|
|
156
|
+
Source: {url}
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
160
|
+
{markdown[:12000]}
|
|
161
|
+
"""
|
|
162
|
+
filename = _safe_filename(url, ".md")
|
|
163
|
+
return content, filename
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _fetch_arxiv(url: str, author: str | None, contributor: str | None) -> tuple[str, str]:
|
|
167
|
+
"""Fetch arXiv abstract page."""
|
|
168
|
+
# Convert /abs/ or /pdf/ to abs for the API
|
|
169
|
+
arxiv_id = re.search(r"(\d{4}\.\d{4,5})", url)
|
|
170
|
+
if arxiv_id:
|
|
171
|
+
api_url = f"https://export.arxiv.org/abs/{arxiv_id.group(1)}"
|
|
172
|
+
try:
|
|
173
|
+
html = _fetch_html(api_url)
|
|
174
|
+
abstract_match = re.search(r'class="abstract[^"]*"[^>]*>(.*?)</blockquote>', html, re.DOTALL | re.IGNORECASE)
|
|
175
|
+
abstract = re.sub(r"<[^>]+>", "", abstract_match.group(1)).strip() if abstract_match else ""
|
|
176
|
+
title_match = re.search(r'class="title[^"]*"[^>]*>(.*?)</h1>', html, re.DOTALL | re.IGNORECASE)
|
|
177
|
+
title = re.sub(r"<[^>]+>", " ", title_match.group(1)).strip() if title_match else arxiv_id.group(1)
|
|
178
|
+
authors_match = re.search(r'class="authors"[^>]*>(.*?)</div>', html, re.DOTALL | re.IGNORECASE)
|
|
179
|
+
paper_authors = re.sub(r"<[^>]+>", "", authors_match.group(1)).strip() if authors_match else ""
|
|
180
|
+
except Exception:
|
|
181
|
+
title, abstract, paper_authors = arxiv_id.group(1), "", ""
|
|
182
|
+
else:
|
|
183
|
+
return _fetch_webpage(url, author, contributor)
|
|
184
|
+
|
|
185
|
+
now = datetime.now(timezone.utc).isoformat()
|
|
186
|
+
content = f"""---
|
|
187
|
+
source_url: "{_yaml_str(url)}"
|
|
188
|
+
arxiv_id: "{_yaml_str(arxiv_id.group(1) if arxiv_id else '')}"
|
|
189
|
+
type: paper
|
|
190
|
+
title: "{_yaml_str(title)}"
|
|
191
|
+
paper_authors: "{_yaml_str(paper_authors)}"
|
|
192
|
+
captured_at: {now}
|
|
193
|
+
contributor: "{_yaml_str(contributor or author or 'unknown')}"
|
|
194
|
+
---
|
|
195
|
+
|
|
196
|
+
# {title}
|
|
197
|
+
|
|
198
|
+
**Authors:** {paper_authors}
|
|
199
|
+
**arXiv:** {arxiv_id.group(1) if arxiv_id else url}
|
|
200
|
+
|
|
201
|
+
## Abstract
|
|
202
|
+
|
|
203
|
+
{abstract}
|
|
204
|
+
|
|
205
|
+
Source: {url}
|
|
206
|
+
"""
|
|
207
|
+
filename = f"arxiv_{arxiv_id.group(1).replace('.', '_')}.md" if arxiv_id else _safe_filename(url, ".md")
|
|
208
|
+
return content, filename
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _download_binary(url: str, suffix: str, target_dir: Path) -> Path:
|
|
212
|
+
"""Download a binary file (PDF, image) directly."""
|
|
213
|
+
filename = _safe_filename(url, suffix)
|
|
214
|
+
out_path = target_dir / filename
|
|
215
|
+
out_path.write_bytes(safe_fetch(url))
|
|
216
|
+
return out_path
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def ingest(url: str, target_dir: Path, author: str | None = None, contributor: str | None = None) -> Path:
|
|
220
|
+
"""
|
|
221
|
+
Fetch a URL and save it into target_dir as a graphify-ready file.
|
|
222
|
+
|
|
223
|
+
Returns the path of the saved file.
|
|
224
|
+
"""
|
|
225
|
+
target_dir.mkdir(parents=True, exist_ok=True)
|
|
226
|
+
url_type = _detect_url_type(url)
|
|
227
|
+
|
|
228
|
+
try:
|
|
229
|
+
validate_url(url)
|
|
230
|
+
except ValueError as exc:
|
|
231
|
+
raise ValueError(f"ingest: {exc}") from exc
|
|
232
|
+
|
|
233
|
+
try:
|
|
234
|
+
if url_type == "pdf":
|
|
235
|
+
out = _download_binary(url, ".pdf", target_dir)
|
|
236
|
+
print(f"Downloaded PDF: {out.name}")
|
|
237
|
+
return out
|
|
238
|
+
|
|
239
|
+
if url_type == "image":
|
|
240
|
+
suffix = Path(urllib.parse.urlparse(url).path).suffix or ".jpg"
|
|
241
|
+
out = _download_binary(url, suffix, target_dir)
|
|
242
|
+
print(f"Downloaded image: {out.name}")
|
|
243
|
+
return out
|
|
244
|
+
|
|
245
|
+
if url_type == "youtube":
|
|
246
|
+
from graphify.transcribe import download_audio
|
|
247
|
+
out = download_audio(url, target_dir)
|
|
248
|
+
print(f"Downloaded audio: {out.name}")
|
|
249
|
+
return out
|
|
250
|
+
|
|
251
|
+
if url_type == "tweet":
|
|
252
|
+
content, filename = _fetch_tweet(url, author, contributor)
|
|
253
|
+
elif url_type == "arxiv":
|
|
254
|
+
content, filename = _fetch_arxiv(url, author, contributor)
|
|
255
|
+
else:
|
|
256
|
+
content, filename = _fetch_webpage(url, author, contributor)
|
|
257
|
+
except (urllib.error.HTTPError, urllib.error.URLError, OSError) as exc:
|
|
258
|
+
raise RuntimeError(f"ingest: failed to fetch {url!r}: {exc}") from exc
|
|
259
|
+
|
|
260
|
+
out_path = target_dir / filename
|
|
261
|
+
# Avoid overwriting - append counter if needed
|
|
262
|
+
counter = 1
|
|
263
|
+
while out_path.exists() and counter < 1000:
|
|
264
|
+
stem = Path(filename).stem
|
|
265
|
+
out_path = target_dir / f"{stem}_{counter}.md"
|
|
266
|
+
counter += 1
|
|
267
|
+
|
|
268
|
+
out_path.write_text(content, encoding="utf-8")
|
|
269
|
+
print(f"Saved {url_type}: {out_path.name}")
|
|
270
|
+
return out_path
|
|
271
|
+
|
|
272
|
+
OUTCOMES = ("useful", "dead_end", "corrected")
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def save_query_result(
|
|
276
|
+
question: str,
|
|
277
|
+
answer: str,
|
|
278
|
+
memory_dir: Path,
|
|
279
|
+
query_type: str = "query",
|
|
280
|
+
source_nodes: list[str] | None = None,
|
|
281
|
+
outcome: str | None = None,
|
|
282
|
+
correction: str | None = None,
|
|
283
|
+
) -> Path:
|
|
284
|
+
"""Save a Q&A result as markdown so it gets extracted into the graph on next --update.
|
|
285
|
+
|
|
286
|
+
Files are stored in memory_dir (typically graphify-out/memory/) with YAML frontmatter
|
|
287
|
+
that graphify's extractor reads as node metadata. This closes the feedback loop:
|
|
288
|
+
the system grows smarter from both what you add AND what you ask.
|
|
289
|
+
|
|
290
|
+
``outcome`` (one of :data:`OUTCOMES`) and ``correction`` are optional work-memory
|
|
291
|
+
signals: they are written both to the frontmatter (so `graphify reflect` can
|
|
292
|
+
aggregate them deterministically) and to an ``## Outcome`` body section (so the
|
|
293
|
+
signal round-trips into the graph on the next semantic re-extraction).
|
|
294
|
+
"""
|
|
295
|
+
if outcome is not None and outcome not in OUTCOMES:
|
|
296
|
+
raise ValueError(f"outcome must be one of {OUTCOMES}, got {outcome!r}")
|
|
297
|
+
|
|
298
|
+
memory_dir = Path(memory_dir)
|
|
299
|
+
memory_dir.mkdir(parents=True, exist_ok=True)
|
|
300
|
+
|
|
301
|
+
now = datetime.now(timezone.utc)
|
|
302
|
+
slug = re.sub(r"[^\w]", "_", question.lower())[:50].strip("_")
|
|
303
|
+
# A second-granularity stamp plus a 50-char slug is not unique: two saves in
|
|
304
|
+
# the same second whose questions share a prefix resolve to one path, and the
|
|
305
|
+
# later write_text silently replaces the earlier one (#3301). The short uuid
|
|
306
|
+
# makes every save its own file; the query_ prefix and .md suffix are kept.
|
|
307
|
+
filename = f"query_{now.strftime('%Y%m%d_%H%M%S')}_{uuid.uuid4().hex[:8]}_{slug}.md"
|
|
308
|
+
|
|
309
|
+
frontmatter_lines = [
|
|
310
|
+
"---",
|
|
311
|
+
f'type: "{query_type}"',
|
|
312
|
+
f'date: "{now.isoformat()}"',
|
|
313
|
+
f'question: "{_yaml_str(question)}"',
|
|
314
|
+
'contributor: "graphify"',
|
|
315
|
+
]
|
|
316
|
+
if outcome:
|
|
317
|
+
frontmatter_lines.append(f'outcome: "{_yaml_str(outcome)}"')
|
|
318
|
+
if correction:
|
|
319
|
+
frontmatter_lines.append(f'correction: "{_yaml_str(correction)}"')
|
|
320
|
+
if source_nodes:
|
|
321
|
+
nodes_str = ", ".join(f'"{_yaml_str(n)}"' for n in source_nodes[:10])
|
|
322
|
+
frontmatter_lines.append(f"source_nodes: [{nodes_str}]")
|
|
323
|
+
frontmatter_lines.append("---")
|
|
324
|
+
|
|
325
|
+
body_lines = [
|
|
326
|
+
"",
|
|
327
|
+
f"# Q: {question}",
|
|
328
|
+
"",
|
|
329
|
+
"## Answer",
|
|
330
|
+
"",
|
|
331
|
+
answer,
|
|
332
|
+
]
|
|
333
|
+
if outcome or correction:
|
|
334
|
+
body_lines += ["", "## Outcome", ""]
|
|
335
|
+
if outcome:
|
|
336
|
+
body_lines.append(f"- Signal: {outcome}")
|
|
337
|
+
if correction:
|
|
338
|
+
body_lines.append(f"- Correction: {correction}")
|
|
339
|
+
if source_nodes:
|
|
340
|
+
body_lines += ["", "## Source Nodes", ""]
|
|
341
|
+
body_lines += [f"- {n}" for n in source_nodes]
|
|
342
|
+
|
|
343
|
+
content = "\n".join(frontmatter_lines + body_lines)
|
|
344
|
+
out_path = memory_dir / filename
|
|
345
|
+
out_path.write_text(content, encoding="utf-8")
|
|
346
|
+
return out_path
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
if __name__ == "__main__":
|
|
350
|
+
import argparse
|
|
351
|
+
parser = argparse.ArgumentParser(description="Fetch a URL into a graphify /raw folder")
|
|
352
|
+
parser.add_argument("url", help="URL to fetch")
|
|
353
|
+
parser.add_argument("target_dir", nargs="?", default="./raw", help="Target directory (default: ./raw)")
|
|
354
|
+
parser.add_argument("--author", help="Your name (stored as node metadata)")
|
|
355
|
+
parser.add_argument("--contributor", help="Contributor name for team graphs")
|
|
356
|
+
args = parser.parse_args()
|
|
357
|
+
out = ingest(args.url, Path(args.target_dir), author=args.author, contributor=args.contributor)
|
|
358
|
+
print(f"Ready for graphify: {out}")
|