clm-kernel 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (333) hide show
  1. clm/__init__.py +230 -0
  2. clm/__main__.py +25 -0
  3. clm/cross_cutting/__init__.py +50 -0
  4. clm/cross_cutting/_test_runner.py +36 -0
  5. clm/cross_cutting/async_utils.py +25 -0
  6. clm/cross_cutting/cel_builder.py +123 -0
  7. clm/cross_cutting/cel_converter.py +697 -0
  8. clm/cross_cutting/cel_engine.py +252 -0
  9. clm/cross_cutting/cel_ops.py +72 -0
  10. clm/cross_cutting/cel_ops_db.py +157 -0
  11. clm/cross_cutting/cel_ops_files.py +99 -0
  12. clm/cross_cutting/cel_ops_state.py +71 -0
  13. clm/cross_cutting/cel_ops_strings.py +64 -0
  14. clm/cross_cutting/cel_resolver.py +63 -0
  15. clm/cross_cutting/cel_slices.py +56 -0
  16. clm/cross_cutting/cel_store.py +330 -0
  17. clm/cross_cutting/clm_logger.py +140 -0
  18. clm/cross_cutting/config/__init__.py +155 -0
  19. clm/cross_cutting/config/config_constants.py +392 -0
  20. clm/cross_cutting/config/env_parameters.py +90 -0
  21. clm/cross_cutting/config/logging.py +378 -0
  22. clm/cross_cutting/config/settings.py +504 -0
  23. clm/cross_cutting/context.py +5 -0
  24. clm/cross_cutting/domain_models.py +55 -0
  25. clm/cross_cutting/domain_types.py +61 -0
  26. clm/cross_cutting/effects.py +67 -0
  27. clm/cross_cutting/errors.py +153 -0
  28. clm/cross_cutting/governance.py +331 -0
  29. clm/cross_cutting/guards.py +41 -0
  30. clm/cross_cutting/mcard.py +321 -0
  31. clm/cross_cutting/native.py +185 -0
  32. clm/cross_cutting/observability.py +289 -0
  33. clm/cross_cutting/pcard.py +434 -0
  34. clm/cross_cutting/registry.py +71 -0
  35. clm/cross_cutting/resources.py +84 -0
  36. clm/cross_cutting/secrets.py +44 -0
  37. clm/cross_cutting/subprocess_utils.py +118 -0
  38. clm/cross_cutting/telemetry.py +164 -0
  39. clm/cross_cutting/testing/__init__.py +8 -0
  40. clm/cross_cutting/testing/comparator.py +187 -0
  41. clm/cross_cutting/testing/datasets.py +87 -0
  42. clm/cross_cutting/timeutil.py +20 -0
  43. clm/cross_cutting/types/new_type_check.py +0 -0
  44. clm/cross_cutting/utils/__init__.py +1 -0
  45. clm/cross_cutting/utils/url_safety.py +50 -0
  46. clm/cross_cutting/vcard.py +560 -0
  47. clm/cross_cutting/writer.py +118 -0
  48. clm/layer0/README.md +50 -0
  49. clm/layer0/__init__.py +223 -0
  50. clm/layer0/algebra.py +60 -0
  51. clm/layer0/algebra_baldwin.py +133 -0
  52. clm/layer0/algebra_baldwin_generators.py +36 -0
  53. clm/layer0/algebra_equivalence.py +262 -0
  54. clm/layer0/algebra_ingest.py +306 -0
  55. clm/layer0/algebra_manifest_io.py +156 -0
  56. clm/layer0/assembler.py +481 -0
  57. clm/layer0/cel/__init__.py +23 -0
  58. clm/layer0/cel_eval.py +491 -0
  59. clm/layer0/codec.py +106 -0
  60. clm/layer0/context.py +122 -0
  61. clm/layer0/db.py +160 -0
  62. clm/layer0/exceptions.py +73 -0
  63. clm/layer0/fibration.py +106 -0
  64. clm/layer0/gates.py +50 -0
  65. clm/layer0/hash.py +80 -0
  66. clm/layer0/loader.py +826 -0
  67. clm/layer0/math_utils.py +163 -0
  68. clm/layer0/mcard.py +96 -0
  69. clm/layer0/model/__init__.py +0 -0
  70. clm/layer0/model/action.py +223 -0
  71. clm/layer0/model/card.py +280 -0
  72. clm/layer0/model/card_triad.py +427 -0
  73. clm/layer0/model/compatibility.py +60 -0
  74. clm/layer0/model/dictionary.py +46 -0
  75. clm/layer0/model/dots.py +331 -0
  76. clm/layer0/model/event_producer.py +88 -0
  77. clm/layer0/model/g_time.py +89 -0
  78. clm/layer0/model/handle.py +163 -0
  79. clm/layer0/model/hash/__init__.py +6 -0
  80. clm/layer0/model/hash/algorithms/__init__.py +3 -0
  81. clm/layer0/model/hash/algorithms/custom_hash.py +7 -0
  82. clm/layer0/model/hash/algorithms/local_sha256.py +178 -0
  83. clm/layer0/model/hash/constants.py +53 -0
  84. clm/layer0/model/hash/enums.py +60 -0
  85. clm/layer0/model/hash/validator.py +307 -0
  86. clm/layer0/model/interpreter.py +275 -0
  87. clm/layer0/model/pagination.py +45 -0
  88. clm/layer0/model/pcard.py +716 -0
  89. clm/layer0/model/schema.py +44 -0
  90. clm/layer0/model/utils/__init__.py +1 -0
  91. clm/layer0/model/utils/content_analyzer.py +152 -0
  92. clm/layer0/model/validators/__init__.py +1 -0
  93. clm/layer0/model/validators/base_validator.py +44 -0
  94. clm/layer0/model/validators/binary_validator.py +66 -0
  95. clm/layer0/model/validators/text_validator.py +82 -0
  96. clm/layer0/model/validators/validation_registry.py +49 -0
  97. clm/layer0/model/vcard.py +823 -0
  98. clm/layer0/model/vcard_ext/__init__.py +39 -0
  99. clm/layer0/model/vcard_ext/core.py +60 -0
  100. clm/layer0/model/vcard_ext/network.py +23 -0
  101. clm/layer0/model/vcard_ext/observability.py +63 -0
  102. clm/layer0/model/vcard_ext/storage.py +60 -0
  103. clm/layer0/model/vcard_ext/vendors.py +211 -0
  104. clm/layer0/model/vcard_sandwich.py +119 -0
  105. clm/layer0/model/vcard_vocabulary.py +463 -0
  106. clm/layer0/model/workflow.py +274 -0
  107. clm/layer0/narrative.py +431 -0
  108. clm/layer0/ontology/collection.py +125 -0
  109. clm/layer0/parser.py +429 -0
  110. clm/layer0/relations.py +73 -0
  111. clm/layer0/rule_evaluator.py +251 -0
  112. clm/layer0/schema/__init__.py +38 -0
  113. clm/layer0/schema/ddl.py +97 -0
  114. clm/layer0/templates/abstract.yaml +99 -0
  115. clm/layer0/templates/balanced.yaml +253 -0
  116. clm/layer0/templates/concrete.yaml +172 -0
  117. clm/layer0/type_registry.py +223 -0
  118. clm/layer0/types.py +204 -0
  119. clm/layer0/utils.py +886 -0
  120. clm/layer0/verifier_core.py +262 -0
  121. clm/layer1/__init__.py +31 -0
  122. clm/layer1/action_dispatcher.py +275 -0
  123. clm/layer1/adapter_registry.py +233 -0
  124. clm/layer1/algebra_baldwin_generators.py +192 -0
  125. clm/layer1/arrow_compiler.py +525 -0
  126. clm/layer1/batch_processor.py +152 -0
  127. clm/layer1/cel/__init__.py +27 -0
  128. clm/layer1/cel/builder.py +173 -0
  129. clm/layer1/engine.py +759 -0
  130. clm/layer1/engine_async.py +89 -0
  131. clm/layer1/eval_callback.py +49 -0
  132. clm/layer1/gate_composer.py +229 -0
  133. clm/layer1/hoare_sandwich.py +298 -0
  134. clm/layer1/identity_guard.py +348 -0
  135. clm/layer1/matrix.py +138 -0
  136. clm/layer1/neural_runtime.py +221 -0
  137. clm/layer1/operations/__init__.py +123 -0
  138. clm/layer1/operations/augment.py +156 -0
  139. clm/layer1/operations/base.py +109 -0
  140. clm/layer1/operations/builtins.py +338 -0
  141. clm/layer1/operations/dispatch.py +144 -0
  142. clm/layer1/operations/exclude.py +107 -0
  143. clm/layer1/operations/handle.py +197 -0
  144. clm/layer1/operations/identity.py +139 -0
  145. clm/layer1/operations/invert.py +66 -0
  146. clm/layer1/operations/loader.py +116 -0
  147. clm/layer1/operations/manipulators/__init__.py +23 -0
  148. clm/layer1/operations/manipulators/document.py +211 -0
  149. clm/layer1/operations/manipulators/media.py +134 -0
  150. clm/layer1/operations/manipulators/structured.py +159 -0
  151. clm/layer1/operations/manipulators/tabular.py +334 -0
  152. clm/layer1/operations/port.py +65 -0
  153. clm/layer1/operations/services.py +529 -0
  154. clm/layer1/operations/split.py +117 -0
  155. clm/layer1/operations/substitute.py +75 -0
  156. clm/layer1/petri.py +1052 -0
  157. clm/layer1/pocketflow_orchestrator.py +408 -0
  158. clm/layer1/pocketflow_scheduler.py +519 -0
  159. clm/layer1/protocols.py +197 -0
  160. clm/layer1/runtime_adapter.py +718 -0
  161. clm/layer1/runtimes/__init__.py +71 -0
  162. clm/layer1/runtimes/_generated_assertions.py +125 -0
  163. clm/layer1/runtimes/_generated_types.py +42 -0
  164. clm/layer1/runtimes/base.py +265 -0
  165. clm/layer1/runtimes/binary.py +101 -0
  166. clm/layer1/runtimes/factory.py +203 -0
  167. clm/layer1/runtimes/javascript.py +321 -0
  168. clm/layer1/runtimes/javascript_runtime.js +33 -0
  169. clm/layer1/runtimes/js_scripts/isomorphism.mjs +22 -0
  170. clm/layer1/runtimes/js_scripts/loader.mjs +52 -0
  171. clm/layer1/runtimes/lambda_calc.py +126 -0
  172. clm/layer1/runtimes/python.py +288 -0
  173. clm/layer1/runtimes/python_runtime.py +41 -0
  174. clm/layer1/runtimes/script.py +114 -0
  175. clm/layer1/sandbox.py +334 -0
  176. clm/layer1/savepoint_guard.py +136 -0
  177. clm/layer1/sparse_net.py +122 -0
  178. clm/layer1/spec_tester.py +165 -0
  179. clm/layer1/store.py +295 -0
  180. clm/layer1/test_runner.py +498 -0
  181. clm/layer1/type_inspector.py +166 -0
  182. clm/layer1/verifier.py +772 -0
  183. clm/layer1/vm.py +91 -0
  184. clm/layer2/__init__.py +78 -0
  185. clm/layer2/card_collection.py +678 -0
  186. clm/layer2/classifier.py +479 -0
  187. clm/layer2/engine/__init__.py +1 -0
  188. clm/layer2/engine/abstract_sql_engine.py +320 -0
  189. clm/layer2/engine/base.py +193 -0
  190. clm/layer2/engine/duckdb_engine.py +576 -0
  191. clm/layer2/engine/sqlite_engine.py +518 -0
  192. clm/layer2/fiber/__init__.py +31 -0
  193. clm/layer2/fiber/context.py +89 -0
  194. clm/layer2/fiber/disposable.py +163 -0
  195. clm/layer2/fiber/fiber.py +125 -0
  196. clm/layer2/fiber/hoare.py +191 -0
  197. clm/layer2/file_io.py +376 -0
  198. clm/layer2/file_registrar.py +77 -0
  199. clm/layer2/ingest.py +202 -0
  200. clm/layer2/lineage.py +322 -0
  201. clm/layer2/mcard_fs.py +262 -0
  202. clm/layer2/merkle.py +11 -0
  203. clm/layer2/mime.py +244 -0
  204. clm/layer2/mime_detector.py +245 -0
  205. clm/layer2/private_collection.py +118 -0
  206. clm/layer2/shared_store.py +11 -0
  207. clm/layer2/storage/__init__.py +190 -0
  208. clm/layer2/storage/db_bitemporal.py +141 -0
  209. clm/layer2/storage/db_driver.py +331 -0
  210. clm/layer2/storage/db_hyperlinks.py +214 -0
  211. clm/layer2/storage/db_pool.py +139 -0
  212. clm/layer2/storage/db_repository.py +229 -0
  213. clm/layer2/storage/savepoint.py +78 -0
  214. clm/layer2/storage/sqlite_wal.py +74 -0
  215. clm/layer2/tridb.py +245 -0
  216. clm/layer2/vcard_receipt.py +71 -0
  217. clm/layer2/vfs.py +17 -0
  218. clm/layer3/__init__.py +141 -0
  219. clm/layer3/agency/__init__.py +19 -0
  220. clm/layer3/agency/config.py +220 -0
  221. clm/layer3/agency/providers/__init__.py +15 -0
  222. clm/layer3/agency/providers/base.py +91 -0
  223. clm/layer3/agency/providers/mlc_llm.py +169 -0
  224. clm/layer3/agency/providers/ollama.py +203 -0
  225. clm/layer3/agency/router.py +71 -0
  226. clm/layer3/agency/runtime.py +524 -0
  227. clm/layer3/gateway.py +353 -0
  228. clm/layer3/open_interpreter_adapter.py +159 -0
  229. clm/layer3/rag/__init__.py +70 -0
  230. clm/layer3/rag/cli.py +228 -0
  231. clm/layer3/rag/config.py +153 -0
  232. clm/layer3/rag/embeddings/__init__.py +16 -0
  233. clm/layer3/rag/embeddings/base.py +79 -0
  234. clm/layer3/rag/embeddings/ollama.py +211 -0
  235. clm/layer3/rag/embeddings/vision.py +290 -0
  236. clm/layer3/rag/engine.py +276 -0
  237. clm/layer3/rag/graph/__init__.py +22 -0
  238. clm/layer3/rag/graph/community.py +198 -0
  239. clm/layer3/rag/graph/engine.py +433 -0
  240. clm/layer3/rag/graph/extractor.py +255 -0
  241. clm/layer3/rag/graph/schema.py +115 -0
  242. clm/layer3/rag/graph/store.py +641 -0
  243. clm/layer3/rag/indexer.py +333 -0
  244. clm/layer3/rag/llm_providers.json +52 -0
  245. clm/layer3/rag/semantic_versioning.py +313 -0
  246. clm/layer3/rag/vector/__init__.py +28 -0
  247. clm/layer3/rag/vector/handle_vector_store.py +749 -0
  248. clm/layer3/rag/vector/schema.py +151 -0
  249. clm/layer3/rag/vector/store.py +632 -0
  250. clm/layer3/reticulum/__init__.py +55 -0
  251. clm/layer3/reticulum/bridge.py +74 -0
  252. clm/layer3/reticulum/daemon.py +88 -0
  253. clm/layer3/reticulum/discovery.py +164 -0
  254. clm/layer3/reticulum/identity.py +107 -0
  255. clm/layer3/reticulum/media.py +148 -0
  256. clm/layer3/reticulum/transport.py +133 -0
  257. clm/layer3/satori/__init__.py +209 -0
  258. clm/layer3/satori/alpha_conversion.py +168 -0
  259. clm/layer3/satori/beta_reduction.py +359 -0
  260. clm/layer3/satori/continuation.py +38 -0
  261. clm/layer3/satori/eta_conversion.py +205 -0
  262. clm/layer3/satori/free_variables.py +163 -0
  263. clm/layer3/satori/io_effects.py +370 -0
  264. clm/layer3/satori/lambda_runtime.py +726 -0
  265. clm/layer3/satori/lambda_term.py +230 -0
  266. clm/layer3/satori/sheaf_audit.py +261 -0
  267. clm/layer3/satori/speech_act.py +279 -0
  268. clm/layer4/README.md +9 -0
  269. clm/layer4/__init__.py +67 -0
  270. clm/layer4/_sovereign_executor.py +580 -0
  271. clm/layer4/action_commands.py +719 -0
  272. clm/layer4/api.py +236 -0
  273. clm/layer4/bootstrap/__init__.py +13 -0
  274. clm/layer4/bootstrap/bootstrapper.py +118 -0
  275. clm/layer4/bootstrap/cleanup.py +44 -0
  276. clm/layer4/bootstrap/clm_hooks.py +49 -0
  277. clm/layer4/bootstrap/genesis.py +77 -0
  278. clm/layer4/bootstrap/injection.py +125 -0
  279. clm/layer4/cli.py +549 -0
  280. clm/layer4/collection_manager.py +553 -0
  281. clm/layer4/commands/__init__.py +81 -0
  282. clm/layer4/commands/action_cmd.py +42 -0
  283. clm/layer4/commands/check_cmd.py +82 -0
  284. clm/layer4/commands/envelope.py +73 -0
  285. clm/layer4/commands/evaluate_cmd.py +224 -0
  286. clm/layer4/commands/mcard_cmd.py +24 -0
  287. clm/layer4/commands/petri_cmd.py +92 -0
  288. clm/layer4/commands/tridb_cmd.py +319 -0
  289. clm/layer4/commands/version_cmd.py +27 -0
  290. clm/layer4/diff_commands.py +167 -0
  291. clm/layer4/eoa_sponsor.py +232 -0
  292. clm/layer4/events.py +69 -0
  293. clm/layer4/flux_gateway.py +133 -0
  294. clm/layer4/gateway.py +91 -0
  295. clm/layer4/loader.py +307 -0
  296. clm/layer4/mcard_commands.py +166 -0
  297. clm/layer4/membrane.py +273 -0
  298. clm/layer4/navigation.py +50 -0
  299. clm/layer4/protocol/protocol_entry_python.py +108 -0
  300. clm/layer4/protocol/python_sync.py +126 -0
  301. clm/layer4/runner.py +150 -0
  302. clm/layer4/satori.py +262 -0
  303. clm/layer4/server/python_sync_client.py +150 -0
  304. clm/layer4/server/python_sync_server.py +139 -0
  305. clm/layer4/server/service_api.py +315 -0
  306. clm/layer4/server/start_servers.py +107 -0
  307. clm/layer4/server/stop_servers.py +24 -0
  308. clm/layer4/simulated_optimization_fix.py +0 -0
  309. clm/layer4/storage.py +463 -0
  310. clm/layer4/tridb_admin.py +268 -0
  311. clm/layer4/websocket/test_ws_client.py +104 -0
  312. clm/layer4/websocket/test_ws_server.py +138 -0
  313. clm/layer4/websocket/ws_server_local.py +160 -0
  314. clm/layer5/__init__.py +51 -0
  315. clm/layer5/async_batch_writer.py +133 -0
  316. clm/layer5/learning.py +276 -0
  317. clm/layer5/telemetry_hook.py +107 -0
  318. clm/layer5/type_lattice.py +238 -0
  319. clm/py.typed +0 -0
  320. clm/schemas/__init__.py +652 -0
  321. clm/schemas/abstract_face_schema.yaml +77 -0
  322. clm/schemas/cel_schema.yaml +416 -0
  323. clm/schemas/lifecycle/__init__.py +45 -0
  324. clm/schemas/lifecycle/models.py +123 -0
  325. clm/schemas/lifecycle/service.py +364 -0
  326. clm/schemas/mcard_schema.sql +192 -0
  327. clm/schemas/mcard_vector_schema.sql +283 -0
  328. clm/schemas/mlp-context-v1.json +91 -0
  329. clm_kernel-0.0.0.dist-info/METADATA +121 -0
  330. clm_kernel-0.0.0.dist-info/RECORD +333 -0
  331. clm_kernel-0.0.0.dist-info/WHEEL +4 -0
  332. clm_kernel-0.0.0.dist-info/entry_points.txt +3 -0
  333. clm_kernel-0.0.0.dist-info/licenses/LICENSE +21 -0
clm/layer2/mime.py ADDED
@@ -0,0 +1,244 @@
1
+ import json
2
+ import os
3
+ import xml.etree.ElementTree as ET
4
+
5
+ import yaml
6
+
7
+
8
+ def detect_encoding(file_path: str) -> str:
9
+ """
10
+ Detects if a file is binary or text based on content sampling.
11
+
12
+ Args:
13
+ file_path (str): The file to analyze.
14
+
15
+ Returns:
16
+ str: 'binary' or 'text'.
17
+ """
18
+ if not os.path.exists(file_path):
19
+ return "binary"
20
+
21
+ try:
22
+ file_size = os.path.getsize(file_path)
23
+ if file_size == 0:
24
+ return "text"
25
+
26
+ with open(file_path, "rb") as f:
27
+ sample = f.read(4096)
28
+
29
+ if b"\x00" in sample:
30
+ return "binary"
31
+
32
+ # Count non-text characters (chars < 9 or 14-31 or 127+)
33
+ non_text_count = 0
34
+ for byte in sample:
35
+ if (byte < 9) or (13 < byte < 32) or (byte > 127):
36
+ non_text_count += 1
37
+
38
+ # Binary if non-text characters exceed 30% of sample size
39
+ if non_text_count > (len(sample) * 30 // 100):
40
+ return "binary"
41
+
42
+ return "text"
43
+ except Exception:
44
+ return "binary"
45
+
46
+ def is_binary(file_path: str) -> bool:
47
+ """Checks if a file is binary."""
48
+ return detect_encoding(file_path) == "binary"
49
+
50
+ def is_text(file_path: str) -> bool:
51
+ """Checks if a file is text."""
52
+ return detect_encoding(file_path) == "text"
53
+
54
+ def get_magic_rules(clm_root: str) -> dict[str, str]:
55
+ """
56
+ Scans the types/ directory to dynamically compile magic byte rules.
57
+
58
+ Args:
59
+ clm_root (str): The CLM workspace root.
60
+
61
+ Returns:
62
+ Dict[str, str]: A dictionary mapping magic hex strings to MIME types.
63
+ """
64
+ # Scan types/ directory to dynamically compile magic rules from manifests
65
+ magic_rules = {}
66
+ types_dir = os.path.join(clm_root, "types")
67
+ if not os.path.exists(types_dir):
68
+ return magic_rules
69
+
70
+ for root, _, files in os.walk(types_dir):
71
+ for file in files:
72
+ if file.endswith(".yaml") or file.endswith(".clm"):
73
+ try:
74
+ with open(os.path.join(root, file), encoding="utf-8") as f:
75
+ data = yaml.safe_load(f)
76
+ if not data or not isinstance(data, dict):
77
+ continue
78
+ inner = data.get("clm", data)
79
+ concrete = inner.get("concrete_impl") or inner.get("concrete")
80
+ if not concrete or not isinstance(concrete, dict):
81
+ continue
82
+ content_type = concrete.get("content_type")
83
+ if not content_type or not isinstance(content_type, dict):
84
+ continue
85
+
86
+ mime = content_type.get("mime")
87
+ magic_bytes = content_type.get("detection_rules", {}).get("magic_bytes")
88
+ if mime and magic_bytes:
89
+ magic_rules[magic_bytes.lower()] = mime
90
+ except Exception:
91
+ pass
92
+ return magic_rules
93
+
94
+ def detect_mime(clm_root: str, file_path: str, ext_hint: str | None = None) -> str:
95
+ """
96
+ Detects the MIME type of a file using magic bytes, extension, or encoding fallback.
97
+
98
+ Args:
99
+ clm_root (str): The CLM workspace root.
100
+ file_path (str): The file to analyze.
101
+ ext_hint (Optional[str]): An optional file extension hint.
102
+
103
+ Returns:
104
+ str: The detected MIME type.
105
+ """
106
+ if not os.path.exists(file_path) or not os.path.isfile(file_path):
107
+ return "application/octet-stream"
108
+
109
+ try:
110
+ # 1. Check magic bytes
111
+ with open(file_path, "rb") as f:
112
+ header = f.read(16)
113
+ hex_header = header.hex().lower()
114
+
115
+ # Load magic rules dynamically
116
+ magic_rules = get_magic_rules(clm_root)
117
+
118
+ # Add static defaults if not loaded or fallback
119
+ static_magic = {
120
+ "89504e470d0a1a0a": "image/png",
121
+ "ffd8ffe0": "image/jpeg",
122
+ "ffd8ffe1": "image/jpeg",
123
+ "ffd8ffe2": "image/jpeg",
124
+ "ffd8ffe3": "image/jpeg",
125
+ "ffd8ffe8": "image/jpeg",
126
+ "474946383761": "image/gif",
127
+ "474946383961": "image/gif",
128
+ "25504446": "application/pdf",
129
+ "504b0304": "application/zip",
130
+ "53514c69746520666f726d6174203300": "application/sqlite3"
131
+ }
132
+ for k, v in static_magic.items():
133
+ if k not in magic_rules:
134
+ magic_rules[k] = v
135
+
136
+ for magic_hex, mime in sorted(magic_rules.items(), key=lambda x: len(x[0]), reverse=True):
137
+ if hex_header.startswith(magic_hex):
138
+ return mime
139
+
140
+ # 2. Check extension hint
141
+ if ext_hint:
142
+ clean_hint = ext_hint.lstrip(".").lower()
143
+ ext_map = {
144
+ "md": "text/markdown",
145
+ "py": "text/x-python",
146
+ "sh": "text/x-shellscript",
147
+ "js": "application/javascript",
148
+ "yaml": "application/yaml",
149
+ "yml": "application/yaml",
150
+ "json": "application/json",
151
+ "xml": "text/xml",
152
+ "html": "text/html",
153
+ "htm": "text/html",
154
+ "svg": "image/svg+xml",
155
+ "stl": "model/stl",
156
+ "obj": "model/obj",
157
+ "dxf": "image/vnd.dxf",
158
+ "gltf": "model/gltf+json",
159
+ "tex": "text/x-tex",
160
+ "tikz": "text/x-tikz",
161
+ "tikzcd": "text/x-tikzcd",
162
+ "tikz-cd": "text/x-tikzcd",
163
+ "djvu": "image/vnd.djvu",
164
+ "djv": "image/vnd.djvu",
165
+ "geojson": "application/geo+json",
166
+ "kml": "application/vnd.google-earth.kml+xml",
167
+ "shp": "application/x-shapefile",
168
+ "gpkg": "application/geopackage+sqlite3"
169
+ }
170
+ if clean_hint in ext_map:
171
+ return ext_map[clean_hint]
172
+
173
+ # 3. Fallback based on encoding
174
+ if is_binary(file_path):
175
+ return "application/octet-stream"
176
+ else:
177
+ return "text/plain"
178
+ except Exception:
179
+ return "application/octet-stream"
180
+
181
+ def validate(clm_root: str, file_path: str, mime: str | None = None) -> bool:
182
+ """
183
+ Validates a file's content against its detected or provided MIME type.
184
+
185
+ Args:
186
+ clm_root (str): The CLM workspace root.
187
+ file_path (str): The file to validate.
188
+ mime (Optional[str]): The expected MIME type.
189
+
190
+ Returns:
191
+ bool: True if valid, False otherwise.
192
+ """
193
+ if not os.path.exists(file_path) or not os.path.isfile(file_path):
194
+ return False
195
+
196
+ try:
197
+ file_size = os.path.getsize(file_path)
198
+ if file_size == 0:
199
+ return False
200
+
201
+ if not mime:
202
+ mime = detect_mime(clm_root, file_path)
203
+
204
+ if mime == "application/json":
205
+ with open(file_path, encoding="utf-8") as f:
206
+ json.load(f)
207
+ return True
208
+
209
+ elif mime in ("application/yaml", "application/x-yaml"):
210
+ with open(file_path, encoding="utf-8") as f:
211
+ yaml.safe_load(f)
212
+ return True
213
+
214
+ elif mime in ("text/xml", "application/xml", "image/svg+xml"):
215
+ with open(file_path, encoding="utf-8") as f:
216
+ ET.fromstring(f.read())
217
+ return True
218
+
219
+ elif mime.startswith("text/"):
220
+ return is_text(file_path)
221
+
222
+ else:
223
+ # Check magic bytes constraint from corresponding manifest
224
+ safe_mime = "".join(c if c.isalnum() else "_" for c in mime)
225
+ manifest_file = None
226
+ for folder in ("content", "transport", "meta"):
227
+ candidate = os.path.join(clm_root, "types", folder, f"{safe_mime}.yaml")
228
+ if os.path.exists(candidate):
229
+ manifest_file = candidate
230
+ break
231
+
232
+ if manifest_file:
233
+ with open(manifest_file, encoding="utf-8") as f:
234
+ data = yaml.safe_load(f)
235
+ inner = data.get("clm", data)
236
+ magic_bytes = inner.get("concrete_impl", {}).get("content_type", {}).get("detection_rules", {}).get("magic_bytes", "")
237
+ if magic_bytes:
238
+ with open(file_path, "rb") as f:
239
+ header = f.read(len(magic_bytes) // 2)
240
+ if header.hex().lower() != magic_bytes.lower():
241
+ return False
242
+ return True
243
+ except Exception:
244
+ return False
@@ -0,0 +1,245 @@
1
+ """MIME Detector and Structural Metadata Extractor (Sprint 392 / DEL-392-02).
2
+
3
+ Provides robust MIME sniffing and structural metadata extraction for the Universal
4
+ MCard Ingestion Membrane. Detects file formats without relying solely on file extensions.
5
+ Extracts format-specific metadata (columns, keys, headers, dimensions, sizes).
6
+
7
+ Layer 2 module (conforms to AST layer boundaries, only imports L0, L1, L2).
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import csv
13
+ import io
14
+ import json
15
+ import re
16
+ from pathlib import Path
17
+ from typing import Any
18
+
19
+ import yaml
20
+
21
+ # Standard Magic Byte Signatures
22
+ MAGIC_SIGNATURES: list[tuple[bytes, str, str]] = [
23
+ (b"\x89PNG\r\n\x1a\n", "image/png", "image"),
24
+ (b"\xff\xd8\xff", "image/jpeg", "image"),
25
+ (b"%PDF-", "application/pdf", "media"),
26
+ (b"GIF87a", "image/gif", "image"),
27
+ (b"GIF89a", "image/gif", "image"),
28
+ (b"glTF", "model/gltf-binary", "model3d"),
29
+ (b"\x4d\x4d\x00\x2a", "image/tiff", "image"),
30
+ (b"\x49\x49\x2a\x00", "image/tiff", "image"),
31
+ (b"\x4d\x4d", "model/x-3ds", "model3d"),
32
+ ]
33
+
34
+
35
+ def _sniff_magic(data: bytes) -> tuple[str | None, str | None]:
36
+ """Check initial magic bytes for binary media formats."""
37
+ for magic, mime, fmt in MAGIC_SIGNATURES:
38
+ if data.startswith(magic):
39
+ return mime, fmt
40
+
41
+ # RIFF containers: WebP or WAV
42
+ if len(data) >= 12 and data[:4] == b"RIFF":
43
+ riff_type = data[8:12]
44
+ if riff_type == b"WEBP":
45
+ return "image/webp", "image"
46
+ elif riff_type == b"WAVE":
47
+ return "audio/wav", "audio"
48
+
49
+ # MP4 container: ftyp box in first 32 bytes
50
+ if len(data) >= 8 and (b"ftyp" in data[:32]):
51
+ return "video/mp4", "media"
52
+
53
+ # MP3: ID3 tag or sync word
54
+ if data.startswith(b"ID3") or (len(data) >= 2 and data[0] == 0xFF and (data[1] & 0xE0) == 0xE0):
55
+ return "audio/mpeg", "audio"
56
+
57
+ return None, None
58
+
59
+
60
+ def _is_binary(data: bytes) -> bool:
61
+ """Heuristic check for binary content."""
62
+ if b"\x00" in data:
63
+ return True
64
+ # Count non-text bytes in first 4096 bytes
65
+ sample = data[:4096]
66
+ if not sample:
67
+ return False
68
+ non_text = sum(1 for b in sample if (b < 9) or (13 < b < 32) or (b > 127))
69
+ return (non_text / len(sample)) > 0.30
70
+
71
+
72
+ def detect_mime_and_format(data: bytes, filename: str | None = None) -> tuple[str, str]:
73
+ """Detect MIME type and canonical CLM format for byte payload.
74
+
75
+ Returns:
76
+ (mime_type, clm_format) where clm_format is one of:
77
+ csv, json, yaml, markdown, text, binary, media, image, audio, model3d.
78
+ """
79
+ ext = Path(filename).suffix.lower() if filename else ""
80
+
81
+ # 1. Magic byte checks
82
+ magic_mime, magic_fmt = _sniff_magic(data)
83
+ if magic_mime and magic_fmt:
84
+ return magic_mime, magic_fmt
85
+
86
+ # 2. Binary vs Text check
87
+ if _is_binary(data):
88
+ # Specific binary 3D model formats
89
+ if ext in (".stl", ".obj"):
90
+ return "model/stl" if ext == ".stl" else "model/obj", "model3d"
91
+ return "application/octet-stream", "binary"
92
+
93
+ # 3. Text content inspection
94
+ try:
95
+ text = data.decode("utf-8")
96
+ except UnicodeDecodeError:
97
+ try:
98
+ text = data.decode("latin-1")
99
+ except UnicodeDecodeError:
100
+ return "application/octet-stream", "binary"
101
+
102
+ stripped = text.strip()
103
+ if not stripped:
104
+ return "text/plain", "text"
105
+
106
+ # SVG check
107
+ if stripped.startswith("<svg") or (stripped.startswith("<?xml") and "<svg" in stripped[:500]):
108
+ return "image/svg+xml", "image"
109
+
110
+ # JSON check
111
+ if (stripped.startswith("{") and stripped.endswith("}")) or (stripped.startswith("[") and stripped.endswith("]")):
112
+ try:
113
+ parsed = json.loads(stripped)
114
+ # Check if GLTF JSON
115
+ if isinstance(parsed, dict) and "asset" in parsed and isinstance(parsed["asset"], dict) and "version" in parsed["asset"]:
116
+ return "model/gltf+json", "model3d"
117
+ return "application/json", "json"
118
+ except (json.JSONDecodeError, ValueError):
119
+ pass
120
+
121
+ # CSV check: multiple lines with consistent commas/delimiters
122
+ lines = [ln for ln in stripped.splitlines() if ln.strip()]
123
+ if len(lines) >= 2:
124
+ try:
125
+ # Sniff CSV dialect
126
+ sample = "\n".join(lines[:10])
127
+ sniffer = csv.Sniffer()
128
+ dialect = sniffer.sniff(sample, delimiters=",\t|;")
129
+ reader = csv.reader(io.StringIO(sample), dialect)
130
+ rows = list(reader)
131
+ if len(rows) >= 2 and all(len(r) == len(rows[0]) and len(r) >= 2 for r in rows):
132
+ return "text/csv", "csv"
133
+ except Exception:
134
+ pass
135
+
136
+ # Markdown check
137
+ has_md_heading = bool(re.search(r"^#{1,6}\s+\S+", text, re.MULTILINE))
138
+ has_md_table = bool(re.search(r"^\|.+\|\s*\n\|[-:| ]+\|\s*\n\|.+\|", text, re.MULTILINE))
139
+ has_md_links = bool(re.search(r"\[.+\]\(.+\)", text))
140
+ if has_md_heading or has_md_table or (has_md_links and ext in (".md", ".markdown")):
141
+ return "text/markdown", "markdown"
142
+
143
+ # YAML check
144
+ if ":" in stripped and not stripped.startswith("{"):
145
+ try:
146
+ parsed = yaml.safe_load(stripped)
147
+ if isinstance(parsed, (dict, list)):
148
+ return "application/x-yaml", "yaml"
149
+ except Exception:
150
+ pass
151
+
152
+ # Extension fallbacks if plain text
153
+ if ext in (".md", ".markdown"):
154
+ return "text/markdown", "markdown"
155
+ if ext == ".csv":
156
+ return "text/csv", "csv"
157
+ if ext in (".yaml", ".yml"):
158
+ return "application/x-yaml", "yaml"
159
+ if ext == ".json":
160
+ return "application/json", "json"
161
+ if ext == ".obj":
162
+ return "model/obj", "model3d"
163
+ if ext == ".stl":
164
+ return "model/stl", "model3d"
165
+
166
+ return "text/plain", "text"
167
+
168
+
169
+ def extract_structural_metadata(data: bytes, mime_type: str, clm_format: str) -> dict[str, Any]:
170
+ """Extract structural metadata from byte content according to format."""
171
+ meta: dict[str, Any] = {
172
+ "byte_size": len(data),
173
+ "mime_type": mime_type,
174
+ "format": clm_format,
175
+ }
176
+
177
+ if clm_format == "csv":
178
+ try:
179
+ text = data.decode("utf-8")
180
+ lines = [ln for ln in text.splitlines() if ln.strip()]
181
+ if lines:
182
+ sniffer = csv.Sniffer()
183
+ sample = "\n".join(lines[:10])
184
+ try:
185
+ dialect = sniffer.sniff(sample, delimiters=",\t|;")
186
+ delim = dialect.delimiter
187
+ except Exception:
188
+ delim = ","
189
+ reader = csv.reader(io.StringIO(text), delimiter=delim)
190
+ rows = list(reader)
191
+ if rows:
192
+ meta["columns"] = rows[0]
193
+ meta["column_count"] = len(rows[0])
194
+ meta["row_count"] = len(rows) - 1
195
+ meta["delimiter"] = delim
196
+ except Exception:
197
+ pass
198
+
199
+ elif clm_format == "json":
200
+ try:
201
+ parsed = json.loads(data.decode("utf-8"))
202
+ if isinstance(parsed, dict):
203
+ meta["json_type"] = "object"
204
+ meta["keys"] = list(parsed.keys())
205
+ meta["key_count"] = len(parsed)
206
+ elif isinstance(parsed, list):
207
+ meta["json_type"] = "array"
208
+ meta["item_count"] = len(parsed)
209
+ except Exception:
210
+ pass
211
+
212
+ elif clm_format == "yaml":
213
+ try:
214
+ parsed = yaml.safe_load(data.decode("utf-8"))
215
+ if isinstance(parsed, dict):
216
+ meta["yaml_type"] = "mapping"
217
+ meta["keys"] = list(parsed.keys())
218
+ meta["key_count"] = len(parsed)
219
+ elif isinstance(parsed, list):
220
+ meta["yaml_type"] = "sequence"
221
+ meta["item_count"] = len(parsed)
222
+ except Exception:
223
+ pass
224
+
225
+ elif clm_format == "markdown":
226
+ try:
227
+ text = data.decode("utf-8")
228
+ headings = re.findall(r"^(#{1,6})\s+(.+)$", text, re.MULTILINE)
229
+ meta["headings"] = [h[1].strip() for h in headings]
230
+ meta["heading_count"] = len(headings)
231
+ meta["line_count"] = len(text.splitlines())
232
+ meta["word_count"] = len(text.split())
233
+ except Exception:
234
+ pass
235
+
236
+ elif clm_format == "text":
237
+ try:
238
+ text = data.decode("utf-8")
239
+ meta["line_count"] = len(text.splitlines())
240
+ meta["char_count"] = len(text)
241
+ meta["word_count"] = len(text.split())
242
+ except Exception:
243
+ pass
244
+
245
+ return meta
@@ -0,0 +1,118 @@
1
+ """Private collection module for encrypted VCards and private execution logs."""
2
+
3
+ import logging
4
+ import os
5
+ from typing import Any
6
+
7
+ from cryptography.exceptions import InvalidTag
8
+ from cryptography.hazmat.primitives.ciphers.aead import AESGCM
9
+
10
+ from clm.cross_cutting.config.config_constants import DEFAULT_EXECUTION_LOG_PATH
11
+ from clm.layer0.model.card import MCard
12
+ from clm.layer2.card_collection import CardCollection
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+ _VOLATILE_DEV_KEY = AESGCM.generate_key(256)
17
+
18
+
19
+ class EncryptedStorageEngineWrapper:
20
+ """Wraps an existing SQLite/DuckDB engine to transparently encrypt/decrypt content."""
21
+
22
+ def __init__(self, engine: Any, aesgcm: AESGCM):
23
+ self._engine = engine
24
+ self._aesgcm = aesgcm
25
+
26
+ def _encrypt(self, data: bytes) -> bytes:
27
+ nonce = os.urandom(12)
28
+ ciphertext = self._aesgcm.encrypt(nonce, data, None)
29
+ return nonce + ciphertext
30
+
31
+ def _decrypt(self, data: bytes) -> bytes:
32
+ nonce = data[:12]
33
+ ciphertext = data[12:]
34
+ try:
35
+ return self._aesgcm.decrypt(nonce, ciphertext, None)
36
+ except InvalidTag as err:
37
+ logger.error(
38
+ "Failed to decrypt DB content. Key mismatch or data corruption."
39
+ )
40
+ raise ValueError(
41
+ "Decryption failed. Invalid key or corrupted data."
42
+ ) from err
43
+
44
+ def add(self, card: MCard) -> str:
45
+ # Create an encrypted version of the card, but maintain the plaintext hash for addressing
46
+ content_bytes = (
47
+ card.content
48
+ if isinstance(card.content, bytes)
49
+ else str(card.content).encode("utf-8")
50
+ )
51
+ encrypted_bytes = self._encrypt(content_bytes)
52
+
53
+ enc_card = MCard(encrypted_bytes)
54
+ enc_card.hash = card.hash
55
+ enc_card.g_time = card.g_time
56
+ return self._engine.add(enc_card)
57
+
58
+ def get(self, hash_value: str) -> MCard | None:
59
+ enc_card = self._engine.get(hash_value)
60
+ if not enc_card:
61
+ return None
62
+ plaintext_bytes = self._decrypt(enc_card.content)
63
+ return MCard(plaintext_bytes)
64
+
65
+ def delete(self, hash_value: str) -> bool:
66
+ return self._engine.delete(hash_value)
67
+
68
+ # Note: Search by content will not work over encrypted data with AES-GCM
69
+ # Search by string will only work over hash/g_time.
70
+ # We delegate other methods as needed, or raise NotImplementedError for full search
71
+ def __getattr__(self, name):
72
+ return getattr(self._engine, name)
73
+
74
+
75
+ class PrivateCollection(CardCollection):
76
+ """A strictly isolated collection for VCards and execution traces.
77
+
78
+ Ensures that all content stored in this collection is AES-GCM encrypted
79
+ at rest. The encryption key should be provided via the `VCARD_ENCRYPTION_KEY`
80
+ environment variable as a hex string.
81
+ """
82
+
83
+ def __init__(self, db_path: str | None = None, key_hex: str | None = None):
84
+ """Initialize the private collection.
85
+
86
+ Args:
87
+ db_path: Path to the private database.
88
+ key_hex: Optional explicitly provided 256-bit AES key in hex.
89
+ """
90
+ if db_path is None:
91
+ db_path = DEFAULT_EXECUTION_LOG_PATH
92
+
93
+ # Resolve the private key
94
+ if not key_hex:
95
+ key_hex = os.getenv("VCARD_ENCRYPTION_KEY")
96
+
97
+ if key_hex:
98
+ try:
99
+ self.key = bytes.fromhex(key_hex)
100
+ except ValueError:
101
+ logger.error("VCARD_ENCRYPTION_KEY must be valid hex.")
102
+ raise
103
+ else:
104
+ logger.warning(
105
+ "No VCARD_ENCRYPTION_KEY found. Generating a volatile key for this session. "
106
+ "Data written will be unreadable after restart!"
107
+ )
108
+ self.key = _VOLATILE_DEV_KEY
109
+
110
+ self.aesgcm = AESGCM(self.key)
111
+
112
+ # Initialize the underlying standard CardCollection on the target db
113
+ super().__init__(db_path=db_path)
114
+
115
+ # Inject our encryption wrapper into the engine pipeline
116
+ self.engine = EncryptedStorageEngineWrapper(self.engine, self.aesgcm)
117
+
118
+ # The CardCollection methods will now safely call our wrapped engine's `add()` and `get()`
@@ -0,0 +1,11 @@
1
+ """Layer 2: Shared Store & Token State Space.
2
+
3
+ Canonical Path: kernel/scripts/clm/layer2/shared_store.py
4
+ Parity: Rust `src/layer2/storage.rs`, TypeScript `src/layer2/triDb.ts`
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from clm.layer1.store import PTRSharedStore, create_store_from_context
10
+
11
+ __all__ = ["PTRSharedStore", "create_store_from_context"]