graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/ids.py ADDED
@@ -0,0 +1,93 @@
1
+ """Single source of truth for node-ID normalization.
2
+
3
+ Three independent producers must agree on node IDs or the graph splits a single
4
+ entity into disconnected ghost nodes:
5
+
6
+ 1. The AST extractor (``extract._make_id``) — deterministic, per-language.
7
+ 2. The semantic subagents (LLM) — follow the node-ID spec in the skill prompt.
8
+ 3. The graph builder (``build._normalize_id``) — reconciles edge endpoints when
9
+ the LLM emits IDs with slightly different punctuation or casing than the AST.
10
+
11
+ Historically the normalization recipe was copy-pasted into ``extract._make_id``
12
+ and ``build._normalize_id`` and kept in sync only by mirrored docstrings, which
13
+ is exactly how the recurring ID-drift bug class crept in (#811 Unicode collapse,
14
+ #550 same-filename collisions, #1033 AST-vs-LLM file-node mismatch, #1104). This
15
+ module exists so the recipe lives in one place and the two callers can no longer
16
+ diverge.
17
+
18
+ The recipe: iterate ``casefold`` then NFKC-normalize to a fixpoint (casefold can
19
+ *expand* a character into a base letter plus a combining mark — ``İ`` -> ``i`` +
20
+ U+0307 — and NFKC then recomposes what can be recomposed; because the two do not
21
+ commute and neither is a fixpoint of the other, a single pass is not
22
+ caseless-stable), then replace runs of non-word characters with a single
23
+ underscore (``re.UNICODE`` so CJK/Cyrillic/Arabic/accented-Latin letters survive
24
+ instead of collapsing to a per-file node), collapse repeated underscores, then
25
+ strip leading/trailing underscores.
26
+
27
+ Casefolding runs BEFORE the non-word filter, not after. With it last, the
28
+ combining marks casefold introduces were never filtered: ``İslemYap`` produced
29
+ ``i̇slemyap`` — an id containing U+0307, which is not a ``\\w`` character — and a
30
+ second pass collapsed it to ``i_slemyap``, so the function was not idempotent
31
+ and the builder's re-normalization disagreed with the extractor's ``make_id``
32
+ for any Turkish identifier (#2614).
33
+
34
+ Casefolding runs in a FIXPOINT LOOP, not once. A single ``NFKC(casefold(...))``
35
+ left ``normalize_id(s) != normalize_id(s.casefold())`` for some combining-mark
36
+ sequences (Greek ypogegrammeni U+0345 followed by a combining accent):
37
+ pre-casefolding turns U+0345 into ``ι``, which NFKC composes with the accent into
38
+ a precomposed char the single pass never reached. Iterating to a fixpoint —
39
+ casefold first, on the raw input — makes the result caseless-stable regardless of
40
+ how many times the caller has already casefolded.
41
+ """
42
+ from __future__ import annotations
43
+
44
+ import re
45
+ import unicodedata
46
+
47
+ __all__ = ["normalize_id", "make_id"]
48
+
49
+
50
+ def normalize_id(s: str) -> str:
51
+ r"""Normalize a single ID string to its canonical form.
52
+
53
+ Guarantees, all enforced by tests:
54
+
55
+ - Idempotent: ``normalize_id(normalize_id(s)) == normalize_id(s)``.
56
+ - The result contains only ``\w`` characters and ``_``.
57
+ - Caseless-stable: ``normalize_id(s) == normalize_id(s.casefold())``.
58
+
59
+ casefold and NFKC do not commute, and neither is a fixpoint of the other:
60
+ casefolding a char can expand it into a base letter plus a combining mark
61
+ (``İ`` -> ``i`` + U+0307), and NFKC can then recompose that mark with an
62
+ adjacent one into a different precomposed char. A single ``NFKC(casefold(...))``
63
+ pass therefore left ``normalize_id(s) != normalize_id(s.casefold())`` for some
64
+ combining-mark sequences (e.g. Greek ypogegrammeni U+0345 followed by a
65
+ combining accent): pre-casefolding turned U+0345 into ``ι`` which NFKC then
66
+ composed with the accent, reaching a form the single-pass recipe never saw.
67
+
68
+ So iterate ``casefold`` then ``NFKC`` to a fixpoint (casefold FIRST, on the
69
+ raw input, so a caller that pre-casefolds lands on the same fixpoint). The
70
+ loop is bounded — Unicode caseless folding converges in one or two steps —
71
+ with a hard cap as a termination guard. Only then apply the ``[^\w]+`` filter,
72
+ so every combining mark casefold introduced has been fully normalized before
73
+ it is filtered (#2614 and its combining-mark follow-on).
74
+ """
75
+ cur = s
76
+ for _ in range(6):
77
+ nxt = unicodedata.normalize("NFKC", cur.casefold())
78
+ if nxt == cur:
79
+ break
80
+ cur = nxt
81
+ cur = re.sub(r"[^\w]+", "_", cur, flags=re.UNICODE)
82
+ cur = re.sub(r"_+", "_", cur)
83
+ return cur.strip("_")
84
+
85
+
86
+ def make_id(*parts: str) -> str:
87
+ """Build a canonical node ID from one or more name parts.
88
+
89
+ Parts are joined with ``_`` (after stripping stray ``_``/``.`` edges from each
90
+ part) and then run through :func:`normalize_id`, so the result is identical to
91
+ what the builder produces from the joined string.
92
+ """
93
+ return normalize_id("_".join(p.strip("_.") for p in parts if p))
graphify/ingest.py ADDED
@@ -0,0 +1,358 @@
1
+ # fetch URLs (tweet/arxiv/pdf/web) and save as annotated markdown
2
+ from __future__ import annotations
3
+ import json
4
+ import re
5
+ import uuid
6
+ import urllib.error
7
+ import urllib.parse
8
+ from datetime import datetime, timezone
9
+ from pathlib import Path
10
+
11
+ from graphify.security import safe_fetch, safe_fetch_text, validate_url
12
+
13
+
14
+ def _yaml_str(s: str) -> str:
15
+ """Escape a string for embedding in a YAML double-quoted scalar.
16
+
17
+ Handles every YAML 1.1/1.2 line-break and control character that could
18
+ let a hostile value (e.g. a fetched page title) break out of the quoted
19
+ scalar and inject sibling YAML keys (F-009 / F-019). The previous
20
+ implementation missed `\\t`, `\\0`, the unicode line-separator U+2028 and
21
+ paragraph-separator U+2029 — all of which YAML treats as line breaks.
22
+
23
+ We intentionally do not depend on PyYAML (not in pyproject deps) and
24
+ instead emit safely-escaped double-quoted scalars by hand: the YAML
25
+ double-quoted form recognises `\\\\`, `\\"`, `\\n`, `\\r`, `\\t`, `\\0`,
26
+ `\\L` (U+2028), `\\P` (U+2029), and `\\xNN`/`\\uNNNN` numeric escapes.
27
+ """
28
+ if s is None:
29
+ return ""
30
+ out: list[str] = []
31
+ for ch in str(s):
32
+ cp = ord(ch)
33
+ if ch == "\\":
34
+ out.append("\\\\")
35
+ elif ch == '"':
36
+ out.append('\\"')
37
+ elif ch == "\n":
38
+ out.append("\\n")
39
+ elif ch == "\r":
40
+ out.append("\\r")
41
+ elif ch == "\t":
42
+ out.append("\\t")
43
+ elif ch == "\0":
44
+ out.append("\\0")
45
+ elif cp == 0x2028:
46
+ out.append("\\L")
47
+ elif cp == 0x2029:
48
+ out.append("\\P")
49
+ elif cp < 0x20 or cp == 0x7F:
50
+ out.append(f"\\x{cp:02x}")
51
+ else:
52
+ out.append(ch)
53
+ return "".join(out)
54
+
55
+
56
+ def _safe_filename(url: str, suffix: str) -> str:
57
+ """Turn a URL into a safe filename."""
58
+ parsed = urllib.parse.urlparse(url)
59
+ name = parsed.netloc + parsed.path
60
+ name = re.sub(r"[^\w\-]", "_", name).strip("_")
61
+ name = re.sub(r"_+", "_", name)[:80]
62
+ return name + suffix
63
+
64
+
65
+ def _detect_url_type(url: str) -> str:
66
+ """Classify the URL for targeted extraction."""
67
+ lower = url.lower()
68
+ if "twitter.com" in lower or "x.com" in lower:
69
+ return "tweet"
70
+ if "arxiv.org" in lower:
71
+ return "arxiv"
72
+ if "github.com" in lower:
73
+ return "github"
74
+ if "youtube.com" in lower or "youtu.be" in lower:
75
+ return "youtube"
76
+ parsed = urllib.parse.urlparse(url)
77
+ path = parsed.path.lower()
78
+ if path.endswith(".pdf"):
79
+ return "pdf"
80
+ if any(path.endswith(ext) for ext in (".png", ".jpg", ".jpeg", ".webp", ".gif")):
81
+ return "image"
82
+ return "webpage"
83
+
84
+
85
+ def _fetch_html(url: str) -> str:
86
+ return safe_fetch_text(url)
87
+
88
+
89
+ def _html_to_markdown(html: str, url: str) -> str:
90
+ """Convert HTML to clean markdown. Uses markdownify if available, else basic strip."""
91
+ # Always pre-strip script/style so their text content never leaks into output
92
+ html = re.sub(r"<script[^>]*>.*?</script>", "", html, flags=re.DOTALL | re.IGNORECASE)
93
+ html = re.sub(r"<style[^>]*>.*?</style>", "", html, flags=re.DOTALL | re.IGNORECASE)
94
+ try:
95
+ from markdownify import markdownify
96
+ return markdownify(html, heading_style="ATX", bullets="-", strip=["img"])
97
+ except ImportError:
98
+ # Fallback: basic tag strip
99
+ text = re.sub(r"<[^>]+>", " ", html)
100
+ text = re.sub(r"\s+", " ", text).strip()
101
+ return text[:8000]
102
+
103
+
104
+ def _fetch_tweet(url: str, author: str | None, contributor: str | None) -> tuple[str, str]:
105
+ """Fetch a tweet URL. Returns (content, filename)."""
106
+ # Normalize to twitter.com for oEmbed
107
+ oembed_url = url.replace("x.com", "twitter.com")
108
+ oembed_api = f"https://publish.twitter.com/oembed?url={urllib.parse.quote(oembed_url)}&omit_script=true"
109
+ try:
110
+ data = json.loads(safe_fetch_text(oembed_api))
111
+ tweet_text = re.sub(r"<[^>]+>", "", data.get("html", "")).strip()
112
+ tweet_author = data.get("author_name", "unknown")
113
+ except Exception:
114
+ # oEmbed failed - save URL stub
115
+ tweet_text = f"Tweet at {url} (could not fetch content)"
116
+ tweet_author = "unknown"
117
+
118
+ now = datetime.now(timezone.utc).isoformat()
119
+ content = f"""---
120
+ source_url: "{_yaml_str(url)}"
121
+ type: tweet
122
+ author: "{_yaml_str(tweet_author)}"
123
+ captured_at: {now}
124
+ contributor: "{_yaml_str(contributor or author or 'unknown')}"
125
+ ---
126
+
127
+ # Tweet by @{tweet_author}
128
+
129
+ {tweet_text}
130
+
131
+ Source: {url}
132
+ """
133
+ filename = _safe_filename(url, ".md")
134
+ return content, filename
135
+
136
+
137
+ def _fetch_webpage(url: str, author: str | None, contributor: str | None) -> tuple[str, str]:
138
+ """Fetch a generic webpage and convert to markdown."""
139
+ html = _fetch_html(url)
140
+ # Extract title
141
+ title_match = re.search(r"<title[^>]*>(.*?)</title>", html, re.IGNORECASE | re.DOTALL)
142
+ title = re.sub(r"\s+", " ", title_match.group(1)).strip() if title_match else url
143
+
144
+ markdown = _html_to_markdown(html, url)
145
+ now = datetime.now(timezone.utc).isoformat()
146
+ content = f"""---
147
+ source_url: "{_yaml_str(url)}"
148
+ type: webpage
149
+ title: "{_yaml_str(title)}"
150
+ captured_at: {now}
151
+ contributor: "{_yaml_str(contributor or author or 'unknown')}"
152
+ ---
153
+
154
+ # {title}
155
+
156
+ Source: {url}
157
+
158
+ ---
159
+
160
+ {markdown[:12000]}
161
+ """
162
+ filename = _safe_filename(url, ".md")
163
+ return content, filename
164
+
165
+
166
+ def _fetch_arxiv(url: str, author: str | None, contributor: str | None) -> tuple[str, str]:
167
+ """Fetch arXiv abstract page."""
168
+ # Convert /abs/ or /pdf/ to abs for the API
169
+ arxiv_id = re.search(r"(\d{4}\.\d{4,5})", url)
170
+ if arxiv_id:
171
+ api_url = f"https://export.arxiv.org/abs/{arxiv_id.group(1)}"
172
+ try:
173
+ html = _fetch_html(api_url)
174
+ abstract_match = re.search(r'class="abstract[^"]*"[^>]*>(.*?)</blockquote>', html, re.DOTALL | re.IGNORECASE)
175
+ abstract = re.sub(r"<[^>]+>", "", abstract_match.group(1)).strip() if abstract_match else ""
176
+ title_match = re.search(r'class="title[^"]*"[^>]*>(.*?)</h1>', html, re.DOTALL | re.IGNORECASE)
177
+ title = re.sub(r"<[^>]+>", " ", title_match.group(1)).strip() if title_match else arxiv_id.group(1)
178
+ authors_match = re.search(r'class="authors"[^>]*>(.*?)</div>', html, re.DOTALL | re.IGNORECASE)
179
+ paper_authors = re.sub(r"<[^>]+>", "", authors_match.group(1)).strip() if authors_match else ""
180
+ except Exception:
181
+ title, abstract, paper_authors = arxiv_id.group(1), "", ""
182
+ else:
183
+ return _fetch_webpage(url, author, contributor)
184
+
185
+ now = datetime.now(timezone.utc).isoformat()
186
+ content = f"""---
187
+ source_url: "{_yaml_str(url)}"
188
+ arxiv_id: "{_yaml_str(arxiv_id.group(1) if arxiv_id else '')}"
189
+ type: paper
190
+ title: "{_yaml_str(title)}"
191
+ paper_authors: "{_yaml_str(paper_authors)}"
192
+ captured_at: {now}
193
+ contributor: "{_yaml_str(contributor or author or 'unknown')}"
194
+ ---
195
+
196
+ # {title}
197
+
198
+ **Authors:** {paper_authors}
199
+ **arXiv:** {arxiv_id.group(1) if arxiv_id else url}
200
+
201
+ ## Abstract
202
+
203
+ {abstract}
204
+
205
+ Source: {url}
206
+ """
207
+ filename = f"arxiv_{arxiv_id.group(1).replace('.', '_')}.md" if arxiv_id else _safe_filename(url, ".md")
208
+ return content, filename
209
+
210
+
211
+ def _download_binary(url: str, suffix: str, target_dir: Path) -> Path:
212
+ """Download a binary file (PDF, image) directly."""
213
+ filename = _safe_filename(url, suffix)
214
+ out_path = target_dir / filename
215
+ out_path.write_bytes(safe_fetch(url))
216
+ return out_path
217
+
218
+
219
+ def ingest(url: str, target_dir: Path, author: str | None = None, contributor: str | None = None) -> Path:
220
+ """
221
+ Fetch a URL and save it into target_dir as a graphify-ready file.
222
+
223
+ Returns the path of the saved file.
224
+ """
225
+ target_dir.mkdir(parents=True, exist_ok=True)
226
+ url_type = _detect_url_type(url)
227
+
228
+ try:
229
+ validate_url(url)
230
+ except ValueError as exc:
231
+ raise ValueError(f"ingest: {exc}") from exc
232
+
233
+ try:
234
+ if url_type == "pdf":
235
+ out = _download_binary(url, ".pdf", target_dir)
236
+ print(f"Downloaded PDF: {out.name}")
237
+ return out
238
+
239
+ if url_type == "image":
240
+ suffix = Path(urllib.parse.urlparse(url).path).suffix or ".jpg"
241
+ out = _download_binary(url, suffix, target_dir)
242
+ print(f"Downloaded image: {out.name}")
243
+ return out
244
+
245
+ if url_type == "youtube":
246
+ from graphify.transcribe import download_audio
247
+ out = download_audio(url, target_dir)
248
+ print(f"Downloaded audio: {out.name}")
249
+ return out
250
+
251
+ if url_type == "tweet":
252
+ content, filename = _fetch_tweet(url, author, contributor)
253
+ elif url_type == "arxiv":
254
+ content, filename = _fetch_arxiv(url, author, contributor)
255
+ else:
256
+ content, filename = _fetch_webpage(url, author, contributor)
257
+ except (urllib.error.HTTPError, urllib.error.URLError, OSError) as exc:
258
+ raise RuntimeError(f"ingest: failed to fetch {url!r}: {exc}") from exc
259
+
260
+ out_path = target_dir / filename
261
+ # Avoid overwriting - append counter if needed
262
+ counter = 1
263
+ while out_path.exists() and counter < 1000:
264
+ stem = Path(filename).stem
265
+ out_path = target_dir / f"{stem}_{counter}.md"
266
+ counter += 1
267
+
268
+ out_path.write_text(content, encoding="utf-8")
269
+ print(f"Saved {url_type}: {out_path.name}")
270
+ return out_path
271
+
272
+ OUTCOMES = ("useful", "dead_end", "corrected")
273
+
274
+
275
+ def save_query_result(
276
+ question: str,
277
+ answer: str,
278
+ memory_dir: Path,
279
+ query_type: str = "query",
280
+ source_nodes: list[str] | None = None,
281
+ outcome: str | None = None,
282
+ correction: str | None = None,
283
+ ) -> Path:
284
+ """Save a Q&A result as markdown so it gets extracted into the graph on next --update.
285
+
286
+ Files are stored in memory_dir (typically graphify-out/memory/) with YAML frontmatter
287
+ that graphify's extractor reads as node metadata. This closes the feedback loop:
288
+ the system grows smarter from both what you add AND what you ask.
289
+
290
+ ``outcome`` (one of :data:`OUTCOMES`) and ``correction`` are optional work-memory
291
+ signals: they are written both to the frontmatter (so `graphify reflect` can
292
+ aggregate them deterministically) and to an ``## Outcome`` body section (so the
293
+ signal round-trips into the graph on the next semantic re-extraction).
294
+ """
295
+ if outcome is not None and outcome not in OUTCOMES:
296
+ raise ValueError(f"outcome must be one of {OUTCOMES}, got {outcome!r}")
297
+
298
+ memory_dir = Path(memory_dir)
299
+ memory_dir.mkdir(parents=True, exist_ok=True)
300
+
301
+ now = datetime.now(timezone.utc)
302
+ slug = re.sub(r"[^\w]", "_", question.lower())[:50].strip("_")
303
+ # A second-granularity stamp plus a 50-char slug is not unique: two saves in
304
+ # the same second whose questions share a prefix resolve to one path, and the
305
+ # later write_text silently replaces the earlier one (#3301). The short uuid
306
+ # makes every save its own file; the query_ prefix and .md suffix are kept.
307
+ filename = f"query_{now.strftime('%Y%m%d_%H%M%S')}_{uuid.uuid4().hex[:8]}_{slug}.md"
308
+
309
+ frontmatter_lines = [
310
+ "---",
311
+ f'type: "{query_type}"',
312
+ f'date: "{now.isoformat()}"',
313
+ f'question: "{_yaml_str(question)}"',
314
+ 'contributor: "graphify"',
315
+ ]
316
+ if outcome:
317
+ frontmatter_lines.append(f'outcome: "{_yaml_str(outcome)}"')
318
+ if correction:
319
+ frontmatter_lines.append(f'correction: "{_yaml_str(correction)}"')
320
+ if source_nodes:
321
+ nodes_str = ", ".join(f'"{_yaml_str(n)}"' for n in source_nodes[:10])
322
+ frontmatter_lines.append(f"source_nodes: [{nodes_str}]")
323
+ frontmatter_lines.append("---")
324
+
325
+ body_lines = [
326
+ "",
327
+ f"# Q: {question}",
328
+ "",
329
+ "## Answer",
330
+ "",
331
+ answer,
332
+ ]
333
+ if outcome or correction:
334
+ body_lines += ["", "## Outcome", ""]
335
+ if outcome:
336
+ body_lines.append(f"- Signal: {outcome}")
337
+ if correction:
338
+ body_lines.append(f"- Correction: {correction}")
339
+ if source_nodes:
340
+ body_lines += ["", "## Source Nodes", ""]
341
+ body_lines += [f"- {n}" for n in source_nodes]
342
+
343
+ content = "\n".join(frontmatter_lines + body_lines)
344
+ out_path = memory_dir / filename
345
+ out_path.write_text(content, encoding="utf-8")
346
+ return out_path
347
+
348
+
349
+ if __name__ == "__main__":
350
+ import argparse
351
+ parser = argparse.ArgumentParser(description="Fetch a URL into a graphify /raw folder")
352
+ parser.add_argument("url", help="URL to fetch")
353
+ parser.add_argument("target_dir", nargs="?", default="./raw", help="Target directory (default: ./raw)")
354
+ parser.add_argument("--author", help="Your name (stored as node metadata)")
355
+ parser.add_argument("--contributor", help="Contributor name for team graphs")
356
+ args = parser.parse_args()
357
+ out = ingest(args.url, Path(args.target_dir), author=args.author, contributor=args.contributor)
358
+ print(f"Ready for graphify: {out}")