system1 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (238) hide show
  1. system1-0.2.2/CHANGELOG.md +24 -0
  2. system1-0.2.2/CONTRIBUTING.md +34 -0
  3. system1-0.2.2/MANIFEST.in +7 -0
  4. system1-0.2.2/PKG-INFO +302 -0
  5. system1-0.2.2/README.md +249 -0
  6. system1-0.2.2/SECURITY.md +13 -0
  7. system1-0.2.2/assets/architecture.png +0 -0
  8. system1-0.2.2/assets/architecture.svg +132 -0
  9. system1-0.2.2/assets/benchmark-chart.png +0 -0
  10. system1-0.2.2/assets/benchmark-chart.svg +99 -0
  11. system1-0.2.2/assets/paperclips-cutover.gif +0 -0
  12. system1-0.2.2/assets/pokemon-system1-60fps.gif +0 -0
  13. system1-0.2.2/assets/privacy-zero-egress.png +0 -0
  14. system1-0.2.2/assets/privacy-zero-egress.svg +184 -0
  15. system1-0.2.2/assets/quality-vs-latency.svg +128 -0
  16. system1-0.2.2/assets/social-preview.png +0 -0
  17. system1-0.2.2/assets/social-preview.svg +84 -0
  18. system1-0.2.2/assets/system1-logo.jpg +0 -0
  19. system1-0.2.2/benchmarks/__init__.py +0 -0
  20. system1-0.2.2/benchmarks/quality/README.md +140 -0
  21. system1-0.2.2/benchmarks/quality/datasets/intent_routing.json +110 -0
  22. system1-0.2.2/benchmarks/quality/datasets/security_triage.json +111 -0
  23. system1-0.2.2/benchmarks/quality/datasets/threat_scoring.json +112 -0
  24. system1-0.2.2/benchmarks/quality/evaluate_teaching.py +128 -0
  25. system1-0.2.2/benchmarks/quality/results/launch_review.json +2788 -0
  26. system1-0.2.2/benchmarks/quality/results/teaching_review.json +8797 -0
  27. system1-0.2.2/benchmarks/quality/run_quality_benchmarks.py +637 -0
  28. system1-0.2.2/benchmarks/results/triple_crown_scorecard.json +139 -0
  29. system1-0.2.2/benchmarks/run_triple_crown_benchmark.py +1105 -0
  30. system1-0.2.2/docs/SPEEDRUN_SHOWDOWN_WORLD_RECORDS.md +180 -0
  31. system1-0.2.2/docs/architecture/technical_specification.md +575 -0
  32. system1-0.2.2/docs/deployment.md +57 -0
  33. system1-0.2.2/docs/examples-review.md +166 -0
  34. system1-0.2.2/docs/guides/training_experts.md +152 -0
  35. system1-0.2.2/docs/launch-review.md +103 -0
  36. system1-0.2.2/docs/observability/README.md +195 -0
  37. system1-0.2.2/docs/observability/grafana-dashboard.json +186 -0
  38. system1-0.2.2/docs/paper/conformal_gating.md +193 -0
  39. system1-0.2.2/docs/paper/system1_technical_brief.md +206 -0
  40. system1-0.2.2/docs/paper/system1_whitepaper.md +582 -0
  41. system1-0.2.2/docs/typesafe.md +132 -0
  42. system1-0.2.2/examples/_teaching_demo.py +126 -0
  43. system1-0.2.2/examples/agent_guard.py +74 -0
  44. system1-0.2.2/examples/auto_cutover_showcase.py +915 -0
  45. system1-0.2.2/examples/autonomous_agent_firewall_showcase.py +1455 -0
  46. system1-0.2.2/examples/core_standalone_evaluator.py +145 -0
  47. system1-0.2.2/examples/deep_jev_benchmark.py +414 -0
  48. system1-0.2.2/examples/enterprise_stress_showcase.py +29 -0
  49. system1-0.2.2/examples/four_levers_benchmark.py +290 -0
  50. system1-0.2.2/examples/gaming/__init__.py +1 -0
  51. system1-0.2.2/examples/gaming/paperclips_speedrun.py +1216 -0
  52. system1-0.2.2/examples/gaming/pokemon_all_games_benchmark.py +837 -0
  53. system1-0.2.2/examples/gaming/pokemon_battle_system1.py +2208 -0
  54. system1-0.2.2/examples/gaming/pokemon_full_campaign_speedrun.py +2438 -0
  55. system1-0.2.2/examples/gaming/pokemon_gameboy_gui.py +939 -0
  56. system1-0.2.2/examples/gaming/pokemon_kaizo_speedrun.py +516 -0
  57. system1-0.2.2/examples/gaming/pokemon_showdown_system1.py +851 -0
  58. system1-0.2.2/examples/jev_comparison_demos.py +303 -0
  59. system1-0.2.2/examples/killer_use_cases_live_test.py +539 -0
  60. system1-0.2.2/examples/model_routing.py +33 -0
  61. system1-0.2.2/examples/observe_routing.py +143 -0
  62. system1-0.2.2/examples/paperclips_typesafe_dropin.py +703 -0
  63. system1-0.2.2/examples/support_triage.py +34 -0
  64. system1-0.2.2/examples/teach_skill.py +50 -0
  65. system1-0.2.2/examples/teaching/README.md +123 -0
  66. system1-0.2.2/examples/teaching/agent_guard.json +646 -0
  67. system1-0.2.2/examples/teaching/model_routing.json +646 -0
  68. system1-0.2.2/examples/teaching/results/agent_guard.json +390 -0
  69. system1-0.2.2/examples/teaching/results/model_routing.json +392 -0
  70. system1-0.2.2/examples/teaching/results/observed_routing.json +341 -0
  71. system1-0.2.2/examples/teaching/results/support_triage.json +421 -0
  72. system1-0.2.2/examples/teaching/support_triage.json +697 -0
  73. system1-0.2.2/examples/train_expert.py +348 -0
  74. system1-0.2.2/examples/typesafe_sdk_dropin_showcase.py +295 -0
  75. {system1-0.2.0 → system1-0.2.2}/pyproject.toml +9 -6
  76. system1-0.2.2/scripts/generate_audit_bundle.py +251 -0
  77. system1-0.2.2/scripts/generate_paperclips_gif.py +208 -0
  78. system1-0.2.2/scripts/generate_pokemon_gif.py +160 -0
  79. system1-0.2.2/scripts/run_paperclips_speedrun.py +33 -0
  80. system1-0.2.2/scripts/run_pokemon_kaizo.py +33 -0
  81. system1-0.2.2/scripts/run_pokemon_showdown.py +33 -0
  82. {system1-0.2.0 → system1-0.2.2}/src/reflex/__init__.py +6 -6
  83. {system1-0.2.0 → system1-0.2.2}/src/reflex/core/__init__.py +1 -1
  84. system1-0.2.2/src/reflex/proto/__init__.py +17 -0
  85. system1-0.2.2/src/reflex/proto/system1_pb2.py +10 -0
  86. system1-0.2.2/src/reflex/proto/system1_pb2_grpc.py +10 -0
  87. {system1-0.2.0 → system1-0.2.2}/src/system1/__init__.py +7 -7
  88. {system1-0.2.0 → system1-0.2.2}/src/system1/calibration.py +12 -2
  89. {system1-0.2.0 → system1-0.2.2}/src/system1/cli.py +84 -50
  90. {system1-0.2.0 → system1-0.2.2}/src/system1/compat/typesafe.py +305 -270
  91. {system1-0.2.0 → system1-0.2.2}/src/system1/compiler.py +260 -118
  92. {system1-0.2.0 → system1-0.2.2}/src/system1/core/__init__.py +1 -1
  93. {system1-0.2.0 → system1-0.2.2}/src/system1/engine.py +8 -6
  94. {system1-0.2.0 → system1-0.2.2}/src/system1/grpc_server.py +1 -1
  95. {system1-0.2.0 → system1-0.2.2}/src/system1/guard.py +4 -4
  96. {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/__init__.py +6 -1
  97. {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/fastapi.py +26 -4
  98. {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/langchain.py +9 -6
  99. {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/mcp.py +4 -4
  100. {system1-0.2.0 → system1-0.2.2}/src/system1/proto/__init__.py +9 -3
  101. system1-0.2.2/src/system1/proto/system1.proto +171 -0
  102. system1-0.2.2/src/system1.egg-info/PKG-INFO +302 -0
  103. {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/SOURCES.txt +106 -1
  104. {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/requires.txt +5 -2
  105. system1-0.2.2/tests/conftest.py +18 -0
  106. system1-0.2.2/tests/e2e/__init__.py +1 -0
  107. system1-0.2.2/tests/e2e/conftest.py +88 -0
  108. system1-0.2.2/tests/e2e/test_tier1_cli.py +182 -0
  109. system1-0.2.2/tests/e2e/test_tier1_conformal.py +197 -0
  110. system1-0.2.2/tests/e2e/test_tier1_dual_imports.py +166 -0
  111. system1-0.2.2/tests/e2e/test_tier1_latency.py +176 -0
  112. system1-0.2.2/tests/e2e/test_tier1_ledger.py +138 -0
  113. system1-0.2.2/tests/e2e/test_tier1_receipts.py +175 -0
  114. system1-0.2.2/tests/e2e/test_tier1_zero_network.py +99 -0
  115. system1-0.2.2/tests/e2e/test_tier2_boundaries.py +177 -0
  116. system1-0.2.2/tests/e2e/test_tier3_combinations.py +185 -0
  117. system1-0.2.2/tests/e2e/test_tier4_real_world.py +212 -0
  118. system1-0.2.2/tests/review_acceptance/__init__.py +1 -0
  119. system1-0.2.2/tests/review_acceptance/test_postfix_boundaries.py +127 -0
  120. system1-0.2.2/tests/review_acceptance/test_previous_contracts_adapted.py +241 -0
  121. system1-0.2.2/tests/review_acceptance/test_round4_variants.py +76 -0
  122. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_crypto_compat_cli.py +3 -3
  123. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_d.py +6 -0
  124. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_f.py +4 -3
  125. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_invariants_6_to_10.py +2 -2
  126. {system1-0.2.0 → system1-0.2.2}/tests/test_auto_cutover.py +7 -0
  127. {system1-0.2.0 → system1-0.2.2}/tests/test_autonomous_agent_firewall_showcase.py +4 -2
  128. {system1-0.2.0 → system1-0.2.2}/tests/test_cli.py +5 -5
  129. {system1-0.2.0 → system1-0.2.2}/tests/test_conformal_invariants.py +3 -3
  130. {system1-0.2.0 → system1-0.2.2}/tests/test_cutover_invariants.py +5 -1
  131. system1-0.2.2/tests/test_gateway_boundaries.py +109 -0
  132. {system1-0.2.0 → system1-0.2.2}/tests/test_grpc_external_generated_client.py +1 -1
  133. {system1-0.2.0 → system1-0.2.2}/tests/test_grpc_polyglot.py +2 -2
  134. {system1-0.2.0 → system1-0.2.2}/tests/test_grpc_server.py +2 -2
  135. system1-0.2.2/tests/test_langchain_dispatch.py +76 -0
  136. system1-0.2.2/tests/test_observation_contract.py +228 -0
  137. {system1-0.2.0 → system1-0.2.2}/tests/test_paperclips_speedrun.py +1 -2
  138. {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_battle_system1.py +1 -1
  139. {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_full_campaign_speedrun.py +35 -0
  140. system1-0.2.2/tests/test_primary_examples.py +71 -0
  141. system1-0.2.2/tests/test_release_examples.py +70 -0
  142. {system1-0.2.0 → system1-0.2.2}/tests/test_round2_gates_c_d.py +3 -0
  143. {system1-0.2.0 → system1-0.2.2}/tests/test_round2_gates_e_f.py +1 -1
  144. system1-0.2.2/tests/test_teaching.py +143 -0
  145. {system1-0.2.0 → system1-0.2.2}/tests/test_typesafe_compat.py +14 -14
  146. system1-0.2.2/uv.lock +1952 -0
  147. system1-0.2.0/PKG-INFO +0 -546
  148. system1-0.2.0/README.md +0 -495
  149. system1-0.2.0/src/reflex/proto/__init__.py +0 -5
  150. system1-0.2.0/src/system1.egg-info/PKG-INFO +0 -546
  151. {system1-0.2.0 → system1-0.2.2}/LICENSE +0 -0
  152. {system1-0.2.0 → system1-0.2.2}/setup.cfg +0 -0
  153. {system1-0.2.0 → system1-0.2.2}/src/reflex/cache.py +0 -0
  154. {system1-0.2.0 → system1-0.2.2}/src/reflex/calibration.py +0 -0
  155. {system1-0.2.0 → system1-0.2.2}/src/reflex/cli.py +0 -0
  156. {system1-0.2.0 → system1-0.2.2}/src/reflex/compat/__init__.py +0 -0
  157. {system1-0.2.0 → system1-0.2.2}/src/reflex/compat/typesafe.py +0 -0
  158. {system1-0.2.0 → system1-0.2.2}/src/reflex/compiler.py +0 -0
  159. {system1-0.2.0 → system1-0.2.2}/src/reflex/core/embeddings.py +0 -0
  160. {system1-0.2.0 → system1-0.2.2}/src/reflex/core/model.py +0 -0
  161. {system1-0.2.0 → system1-0.2.2}/src/reflex/core/neural.py +0 -0
  162. {system1-0.2.0 → system1-0.2.2}/src/reflex/core/schema.py +0 -0
  163. {system1-0.2.0 → system1-0.2.2}/src/reflex/core/telemetry.py +0 -0
  164. {system1-0.2.0 → system1-0.2.2}/src/reflex/embeddings.py +0 -0
  165. {system1-0.2.0 → system1-0.2.2}/src/reflex/engine.py +0 -0
  166. {system1-0.2.0 → system1-0.2.2}/src/reflex/grpc_server.py +0 -0
  167. {system1-0.2.0 → system1-0.2.2}/src/reflex/guard.py +0 -0
  168. {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/__init__.py +0 -0
  169. {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/fastapi.py +0 -0
  170. {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/langchain.py +0 -0
  171. {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/mcp.py +0 -0
  172. {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/observability.py +0 -0
  173. {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/otel.py +0 -0
  174. {system1-0.2.0 → system1-0.2.2}/src/reflex/ledger.py +0 -0
  175. {system1-0.2.0 → system1-0.2.2}/src/reflex/model.py +0 -0
  176. {system1-0.2.0 → system1-0.2.2}/src/reflex/neural.py +0 -0
  177. {system1-0.2.0/src/system1 → system1-0.2.2/src/reflex}/proto/system1.proto +0 -0
  178. {system1-0.2.0 → system1-0.2.2}/src/reflex/receipt.py +0 -0
  179. {system1-0.2.0 → system1-0.2.2}/src/reflex/schema.py +0 -0
  180. {system1-0.2.0 → system1-0.2.2}/src/reflex/telemetry.py +0 -0
  181. {system1-0.2.0 → system1-0.2.2}/src/system1/cache.py +0 -0
  182. {system1-0.2.0 → system1-0.2.2}/src/system1/compat/__init__.py +0 -0
  183. {system1-0.2.0 → system1-0.2.2}/src/system1/core/embeddings.py +0 -0
  184. {system1-0.2.0 → system1-0.2.2}/src/system1/core/model.py +0 -0
  185. {system1-0.2.0 → system1-0.2.2}/src/system1/core/neural.py +0 -0
  186. {system1-0.2.0 → system1-0.2.2}/src/system1/core/schema.py +0 -0
  187. {system1-0.2.0 → system1-0.2.2}/src/system1/core/telemetry.py +0 -0
  188. {system1-0.2.0 → system1-0.2.2}/src/system1/embeddings.py +0 -0
  189. {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/observability.py +0 -0
  190. {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/otel.py +0 -0
  191. {system1-0.2.0 → system1-0.2.2}/src/system1/ledger.py +0 -0
  192. {system1-0.2.0 → system1-0.2.2}/src/system1/model.py +0 -0
  193. {system1-0.2.0 → system1-0.2.2}/src/system1/neural.py +0 -0
  194. {system1-0.2.0 → system1-0.2.2}/src/system1/proto/system1_pb2.py +0 -0
  195. {system1-0.2.0 → system1-0.2.2}/src/system1/proto/system1_pb2_grpc.py +0 -0
  196. {system1-0.2.0 → system1-0.2.2}/src/system1/receipt.py +0 -0
  197. {system1-0.2.0 → system1-0.2.2}/src/system1/schema.py +0 -0
  198. {system1-0.2.0 → system1-0.2.2}/src/system1/telemetry.py +0 -0
  199. {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/dependency_links.txt +0 -0
  200. {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/entry_points.txt +0 -0
  201. {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/top_level.txt +0 -0
  202. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_a.py +0 -0
  203. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_b.py +0 -0
  204. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_c.py +0 -0
  205. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_e.py +0 -0
  206. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_invariants_challenger_1.py +0 -0
  207. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_m1_parity.py +0 -0
  208. {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_tier5_engine.py +0 -0
  209. {system1-0.2.0 → system1-0.2.2}/tests/test_attestation_invariants.py +0 -0
  210. {system1-0.2.0 → system1-0.2.2}/tests/test_audit_architectural_upgrades.py +0 -0
  211. {system1-0.2.0 → system1-0.2.2}/tests/test_authorization_invariants.py +0 -0
  212. {system1-0.2.0 → system1-0.2.2}/tests/test_auto_cutover_showcase.py +0 -0
  213. {system1-0.2.0 → system1-0.2.2}/tests/test_cache_invariants.py +0 -0
  214. {system1-0.2.0 → system1-0.2.2}/tests/test_calibration.py +0 -0
  215. {system1-0.2.0 → system1-0.2.2}/tests/test_cli_compile.py +0 -0
  216. {system1-0.2.0 → system1-0.2.2}/tests/test_compiler.py +0 -0
  217. {system1-0.2.0 → system1-0.2.2}/tests/test_conformal.py +0 -0
  218. {system1-0.2.0 → system1-0.2.2}/tests/test_core_isolation.py +0 -0
  219. {system1-0.2.0 → system1-0.2.2}/tests/test_engine.py +0 -0
  220. {system1-0.2.0 → system1-0.2.2}/tests/test_four_levers.py +0 -0
  221. {system1-0.2.0 → system1-0.2.2}/tests/test_guard.py +0 -0
  222. {system1-0.2.0 → system1-0.2.2}/tests/test_hybrid_embeddings.py +0 -0
  223. {system1-0.2.0 → system1-0.2.2}/tests/test_integrations.py +0 -0
  224. {system1-0.2.0 → system1-0.2.2}/tests/test_model.py +0 -0
  225. {system1-0.2.0 → system1-0.2.2}/tests/test_neural_projector.py +0 -0
  226. {system1-0.2.0 → system1-0.2.2}/tests/test_observability.py +0 -0
  227. {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_benchmark.py +0 -0
  228. {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_kaizo_speedrun.py +0 -0
  229. {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_showdown_system1.py +0 -0
  230. {system1-0.2.0 → system1-0.2.2}/tests/test_previous_contracts_adapted.py +0 -0
  231. {system1-0.2.0 → system1-0.2.2}/tests/test_receipt.py +0 -0
  232. {system1-0.2.0 → system1-0.2.2}/tests/test_round2_gates_a_b.py +0 -0
  233. {system1-0.2.0 → system1-0.2.2}/tests/test_round4_variants.py +0 -0
  234. {system1-0.2.0 → system1-0.2.2}/tests/test_schema.py +0 -0
  235. {system1-0.2.0 → system1-0.2.2}/tests/test_system1_exports.py +0 -0
  236. {system1-0.2.0 → system1-0.2.2}/tests/test_train_expert.py +0 -0
  237. {system1-0.2.0 → system1-0.2.2}/tests/test_triple_crown_benchmark.py +0 -0
  238. {system1-0.2.0 → system1-0.2.2}/tests/test_whitening_and_exemplars.py +0 -0
@@ -0,0 +1,24 @@
1
+ # Changelog
2
+
3
+ ## 0.2.2 — 2026-09-19
4
+
5
+ - Preserve the complete validated skill through takeover and reload: projector settings, schema identity, calibration, distributions and review behavior. Gate normal promotion on useful local acceptance as well as teacher agreement.
6
+ - Make observed-only teaching and strict review the TypeSafe adapter defaults. Group related observations across all supplied lineage identifiers; insufficient evidence continues using the teacher.
7
+ - Support SDK context managers, JSON object/array state, typed response accessors and ordinal score distributions. Document the supported contract and reject unsupported transport/response options.
8
+ - Correct strict conformal sets to invert their calibrated cumulative-probability scores; retain conservative review when calibration is insufficient.
9
+ - Add the complete offline observation-to-local-reuse example, with optional real Jev observation. Remove fabricated speedups after HTTP failures, counter-only campaign takeover, reconstructed probability distributions and unsupported example claims.
10
+
11
+ - Teach a skill from supplied examples with `compile(..., augment=False)` or CLI `--dataset`, rejecting malformed labels instead of inventing replacements. Keep repeated prompts out of separate calibration partitions and preserve uncalibrated status when no calibration examples exist.
12
+ - Load saved skills directly with `system1 decide --model skill.s1m`, using strict uncertainty gating. Add a small teaching example and reproducible comparison on unseen examples.
13
+ - Replace the three primary showcases with focused, offline teaching demonstrations: explicit labeled datasets, separate calibration and evaluation cases, saved skills, and measured quality, review frequency, size, and speed. Keep agent permissions under deterministic policy rules.
14
+ - Propagate LangChain callback denials and approval requirements through synchronous and asynchronous tool dispatch. Preserve the configured principal instead of substituting a run ID.
15
+ - Handle non-object JSON, malformed multipart text, disconnects, and one-time body replay in the ASGI gateway; respect structured escalation signals.
16
+ - Return actual receipt digests in gateway headers and MCP errors.
17
+ - Add real LangChain dispatch regression tests, built-distribution checks, Python 3.14 CI coverage, and release-tag validation.
18
+ - Keep legacy proto assets accessible without optional gRPC/protobuf dependencies in minimal wheel installs.
19
+ - Update package build requirements and the lockfile; include examples and test support files in source distributions.
20
+ - Replace unsupported launch claims with reproducible benchmark results, executable quickstarts, and explicit deployment boundaries.
21
+
22
+ ## 0.2.1
23
+
24
+ Previous published baseline. See the [repository history](https://github.com/steph4n-gh/system1/commits/main/) for earlier changes.
@@ -0,0 +1,34 @@
1
+ # Contributing
2
+
3
+ Report reproducible bugs and feature requests through the [issue tracker](https://github.com/steph4n-gh/system1/issues). For vulnerabilities, follow [SECURITY.md](SECURITY.md).
4
+
5
+ ## Set up
6
+
7
+ Use Python 3.11 or newer:
8
+
9
+ ```bash
10
+ git clone https://github.com/steph4n-gh/system1.git
11
+ cd system1
12
+ python -m venv .venv
13
+ source .venv/bin/activate
14
+ python -m pip install -e '.[dev,langchain]'
15
+ python -m pytest -q
16
+ ```
17
+
18
+ With uv, use `uv sync --locked --extra dev --extra langchain` and `uv run --no-sync pytest -q`. LangChain integration tests use the real callback dispatcher. MLX and emulator coverage requires optional extras and suitable hardware/assets; skipped optional tests are reported by pytest.
19
+
20
+ ## Changes
21
+
22
+ Keep changes focused and prefer the simplest implementation that addresses a reproduced problem. Add regression coverage for behavior changes, preserve `system1`/`reflex` API parity, and update runnable documentation when the API changes. Do not include private keys, tokens, ledger data, or proprietary ROMs.
23
+
24
+ Include the problem, resulting behavior, and relevant validation in pull requests. Benchmark claims should include the command, source revision, environment, dataset, overall quality metrics, and raw results. Simulations and cloud measurements must be clearly distinguished.
25
+
26
+ ## Package checks
27
+
28
+ ```bash
29
+ uv lock --check
30
+ uv build
31
+ uvx twine check --strict dist/*
32
+ ```
33
+
34
+ CI also installs the wheel into a clean environment and runs the CLI outside the checkout. A release tag must match the version in `pyproject.toml`. PyPI publishing requires the repository's `pypi` environment and trusted-publisher configuration; tagging and publishing are separate release actions.
@@ -0,0 +1,7 @@
1
+ include LICENSE README.md CONTRIBUTING.md SECURITY.md CHANGELOG.md uv.lock
2
+ recursive-include tests *.py
3
+ recursive-include examples *.py *.json *.md
4
+ recursive-include scripts *.py
5
+ recursive-include benchmarks *.py *.json *.md
6
+ recursive-include docs *.md *.json
7
+ recursive-include assets *.svg *.png *.jpg *.gif
system1-0.2.2/PKG-INFO ADDED
@@ -0,0 +1,302 @@
1
+ Metadata-Version: 2.4
2
+ Name: system1
3
+ Version: 0.2.2
4
+ Summary: Local structured decisions, deterministic tool policies, conformal uncertainty gating, and signed audit receipts.
5
+ Author: System 1 Authors
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Homepage, https://github.com/steph4n-gh/system1
8
+ Project-URL: Repository, https://github.com/steph4n-gh/system1.git
9
+ Project-URL: Documentation, https://github.com/steph4n-gh/system1#readme
10
+ Project-URL: Issue Tracker, https://github.com/steph4n-gh/system1/issues
11
+ Keywords: reflex,system1,decision-runtime,sub-millisecond,apple-silicon,mlx,conformal-prediction,dual-process,zero-egress,ed25519,audit-receipts
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Programming Language :: Python :: 3.14
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: Security :: Cryptography
21
+ Requires-Python: >=3.11
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Requires-Dist: numpy>=1.24.0
25
+ Requires-Dist: cryptography>=41.0.0
26
+ Provides-Extra: metal
27
+ Requires-Dist: mlx>=0.10.0; extra == "metal"
28
+ Provides-Extra: neural
29
+ Requires-Dist: mlx>=0.10.0; extra == "neural"
30
+ Provides-Extra: gameboy
31
+ Requires-Dist: pyboy>=2.0.0; extra == "gameboy"
32
+ Provides-Extra: observability
33
+ Requires-Dist: prometheus_client>=0.17.0; extra == "observability"
34
+ Provides-Extra: otel
35
+ Requires-Dist: opentelemetry-api>=1.20.0; extra == "otel"
36
+ Requires-Dist: opentelemetry-sdk>=1.20.0; extra == "otel"
37
+ Provides-Extra: dev
38
+ Requires-Dist: pytest>=7.0.0; extra == "dev"
39
+ Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
40
+ Requires-Dist: grpcio>=1.80.0; extra == "dev"
41
+ Requires-Dist: grpcio-tools>=1.80.0; extra == "dev"
42
+ Requires-Dist: protobuf>=7.35.1; extra == "dev"
43
+ Requires-Dist: prometheus_client>=0.17.0; extra == "dev"
44
+ Requires-Dist: opentelemetry-api>=1.20.0; extra == "dev"
45
+ Requires-Dist: opentelemetry-sdk>=1.20.0; extra == "dev"
46
+ Provides-Extra: grpc
47
+ Requires-Dist: grpcio>=1.80.0; extra == "grpc"
48
+ Requires-Dist: grpcio-tools>=1.80.0; extra == "grpc"
49
+ Requires-Dist: protobuf>=7.35.1; extra == "grpc"
50
+ Provides-Extra: langchain
51
+ Requires-Dist: langchain-core<2,>=1.0; extra == "langchain"
52
+ Dynamic: license-file
53
+
54
+ <p align="center">
55
+ <img src="https://raw.githubusercontent.com/steph4n-gh/system1/main/assets/system1-logo.jpg" alt="System 1 logo" width="128" />
56
+ </p>
57
+
58
+ # System 1: Local Decision Runtime for AI Agents
59
+
60
+ Teach a repeatable decision skill from examples or by observing a teacher such as Jev. Validate it, take over locally, and save the skill for reuse.
61
+
62
+ [![CI](https://github.com/steph4n-gh/system1/actions/workflows/ci.yml/badge.svg)](https://github.com/steph4n-gh/system1/actions/workflows/ci.yml)
63
+ [![PyPI](https://img.shields.io/pypi/v/system1.svg)](https://pypi.org/project/system1/)
64
+ [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)](pyproject.toml)
65
+ [![License](https://img.shields.io/badge/license-Apache_2.0-blue.svg)](LICENSE)
66
+
67
+ **Status: beta.** System 1 provides a small local classifier and a separate deterministic policy guard. The included seed models need evaluation and calibration on your workload. They are not a general-purpose detector of malicious actions or prompt injection.
68
+
69
+ [Quickstart](#quickstart) · [Policy guard](#policy-guard) · [Integrations](#integrations) · [Benchmarks](#benchmarks) · [Limits and deployment](docs/deployment.md) · [Contributing](CONTRIBUTING.md)
70
+
71
+ ## What it does
72
+
73
+ - **Observe → teach → validate → run locally:** keep the teacher answering until the observed skill passes held-out checks, then use the same call site locally. Save and reload the validated skill with its uncertainty behavior intact.
74
+ - **Structured classification:** choice, boolean, multi-choice, and score fields using local NumPy projections, with optional MLX acceleration.
75
+ - **Uncertainty handling:** calibration and conformal prediction sets for routing uncertain decisions to application-defined review or fallback paths.
76
+ - **Deterministic permissions:** `PolicyEngine` evaluates explicit rules; `SystemOneGuard(enforcement_profile=True)` requires a matching permission grant, signing key, and durable ledger before returning `ALLOW`.
77
+ - **Audit evidence:** Ed25519 software signatures (RFC 8032) on decision receipts and a SHA-256 hash-chained SQLite ledger. Applications remain responsible for authenticating callers and enforcing decisions at the tool boundary.
78
+ - **Local execution:** the core decision path requires no network or cloud API. Optional fallback clients, telemetry exporters, and application tools can use the network.
79
+
80
+ ## Quickstart
81
+
82
+ The project is packaged on PyPI as `system1`. Use Python 3.11 or newer:
83
+
84
+ ```bash
85
+ python -m pip install system1
86
+ ```
87
+
88
+ To try the code in this repository:
89
+
90
+ ```bash
91
+ git clone https://github.com/steph4n-gh/system1.git
92
+ cd system1
93
+ python -m venv .venv
94
+ source .venv/bin/activate
95
+ python -m pip install -e '.[dev]'
96
+ ```
97
+
98
+ Define the outputs you need, then evaluate a prompt:
99
+
100
+ ```python
101
+ from system1 import ChoiceField, DecisionSchema, System1Engine
102
+
103
+ class SupportRoute(DecisionSchema):
104
+ team = ChoiceField(
105
+ options=["billing", "technical_support"],
106
+ descriptions={
107
+ "billing": "Invoices, payments, subscription plans, and refunds",
108
+ "technical_support": "Software installation, errors, and debugging",
109
+ },
110
+ )
111
+
112
+ engine = System1Engine(SupportRoute)
113
+ decision = engine.decide("I need a refund for my subscription")
114
+ print(decision.values)
115
+ print(f"Latency: {decision.latency_ms:.2f} ms")
116
+ print(f"Needs review: {decision.is_ambiguous}")
117
+ print(f"Prediction sets: {decision.conformal_sets}")
118
+ ```
119
+
120
+ This starts with schema-derived seed prototypes. Outputs and latency depend on the schema, data, and hardware. A predicted value is not permission to execute a tool.
121
+
122
+ ## Observe a teacher, then take over
123
+
124
+ ```bash
125
+ python examples/observe_routing.py
126
+ ```
127
+
128
+ This short offline example shows the complete journey: observe answers, teach one routing skill, pass the normal promotion gates, disconnect the teacher, evaluate fresh requests locally, then save and reopen the skill. In the recorded run it needed **190 observations**, answered **40/40 fresh synthetic tickets correctly**, accepted all 40, and saved a **5 KB** skill. Reloading preserved answers, probabilities, and review decisions. Local median latency was about **0.45 ms** on the review machine.
129
+
130
+ The default teacher is a rule over a small structured-ticket vocabulary, clearly labeled as a simulation. It demonstrates the lifecycle, not general language understanding or Jev quality parity. `--teacher jev` observes actual Jev responses with your API key; another API can use the existing teacher callback. See the [adapter guide and compatibility contract](docs/typesafe.md) and [recorded evidence](examples/teaching/results/observed_routing.json).
131
+
132
+ Promotion depends on evidence, not a fixed turn count. Insufficient evidence keeps the teacher active, and uncertain local responses still request review.
133
+
134
+ ## Teach a skill
135
+
136
+ Give System 1 labeled examples of one task, save the resulting `.s1m` skill, and reuse it locally. A person or System 2 can supply the examples; teaching does not require an LLM. The existing compiler fits a small decision head using NumPy.
137
+
138
+ ```bash
139
+ python examples/support_triage.py
140
+ python examples/model_routing.py
141
+ python examples/agent_guard.py
142
+ system1 decide "Please correct the invoice address" --model .system1/examples/support_triage/skill.s1m --json
143
+ ```
144
+
145
+ Each primary example supplies labeled teaching cases, separate calibration cases, and 24 unseen evaluation cases. One command teaches, saves, reloads, and reports the results. On the review machine, teaching plus calibration took 127–159 ms, saved skills were 25–32 KiB, and uncached decisions took about 0.5 ms. Preparing good labeled examples takes additional work.
146
+
147
+ The authored demonstration cases reached 87.5% support-routing accuracy, 91.7% model-routing accuracy, and 100% operation-triage accuracy. Strict uncertainty checks requested review on 23/24 support cases, all 24 routing cases, and 22/24 operation cases. The three accepted responses were correct. See the [complete results, data, and limits](examples/teaching/README.md); these are demonstrations, not production quality guarantees.
148
+
149
+ For the smallest API example, see [teach one skill](examples/teach_skill.py). The [teaching guide](docs/guides/training_experts.md) covers your own data and evaluation. In Python, use `compile(examples, augment=False)` and `System1Engine(..., strict_mode=True)` for this workflow.
150
+
151
+ ## Policy guard
152
+
153
+ This executable example permits one configuration lookup for one application-supplied principal, records the actual lookup result, and verifies the ledger. Other keys or tools have no grant. It persists a local demonstration key across runs.
154
+
155
+ ```python
156
+ from pathlib import Path
157
+ from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
158
+ from system1 import (
159
+ ActionLedger, ActionProposal, DecisionOutcome, PolicyEngine, PolicyRule,
160
+ RiskLevel, SystemOneGuard, load_private_key, save_keypair,
161
+ )
162
+
163
+ key_path = Path(".system1/demo-identity/identity.key")
164
+ if not key_path.exists():
165
+ save_keypair(Ed25519PrivateKey.generate(), key_path.parent)
166
+ signing_key = load_private_key(key_path)
167
+
168
+ policy = PolicyEngine(rules=[PolicyRule(
169
+ rule_id="read_service_name",
170
+ tools=["read_config"],
171
+ allowed_principals=["agent-worker"],
172
+ allowed_tenants=["demo"],
173
+ allowed_scopes=["config:read"],
174
+ argument_limits={"key": ["service_name"]},
175
+ outcome=DecisionOutcome.ALLOW,
176
+ risk=RiskLevel.READ_ONLY,
177
+ )])
178
+
179
+ with ActionLedger(".system1/demo-audit.sqlite", require_durable=True) as ledger:
180
+ guard = SystemOneGuard(
181
+ policy=policy, ledger=ledger, signing_key=signing_key,
182
+ enforcement_profile=True,
183
+ )
184
+ proposal = ActionProposal.create(
185
+ tenant_id="demo", principal_id="agent-worker", scope="config:read",
186
+ tool="read_config", arguments={"key": "service_name"},
187
+ canonical_target="config:service_name", purpose="Inspect service name",
188
+ )
189
+ auth = guard.evaluate_proposal(proposal)
190
+ if auth.outcome != DecisionOutcome.ALLOW:
191
+ raise PermissionError(auth.reason)
192
+
193
+ # Execute exactly the authorized operation and arguments.
194
+ result = {"service_name": "system1-demo"}[proposal.arguments["key"]]
195
+ ledger.record_execution_outcome(
196
+ action_id=proposal.action_id,
197
+ receipt_digest=auth.receipt.compute_digest(),
198
+ status="SUCCEEDED", result_payload={"value": result},
199
+ tenant_id=proposal.tenant_id, principal_id=proposal.principal_id,
200
+ scope=proposal.scope, trusted_public_key=signing_key.public_key(),
201
+ )
202
+ assert ledger.verify_integrity(trusted_public_key=signing_key.public_key())
203
+ print(result)
204
+ ```
205
+
206
+ The application must derive identity from its authenticated session, constrain tool arguments and targets, and keep agent code from bypassing the guard. Policy correctness is the operator's responsibility. The default guard without `enforcement_profile=True` can use statistical classification to allow actions. See [deployment boundaries](docs/deployment.md) before granting consequential permissions.
207
+
208
+ ## Integrations
209
+
210
+ ### LangChain
211
+
212
+ Install `python -m pip install 'system1[langchain]'`. Attach a configured guard to actual tool invocations:
213
+
214
+ ```python
215
+ from system1.integrations import SystemOneGuardCallbackHandler
216
+
217
+ # guard is your configured SystemOneGuard, with a live ledger and signing key.
218
+ handler = SystemOneGuardCallbackHandler(
219
+ guard=guard, tenant_id="demo", principal_id="agent-worker",
220
+ )
221
+ result = your_tool.invoke(tool_arguments, config={"callbacks": [handler]})
222
+ ```
223
+
224
+ The callback uses scope `langchain:tools:exec` and the serialized tool input in its proposal. Match policy rules to that contract. `SystemOneGuardBlockedException` propagates to the caller on a deny or approval requirement. `wrap_langchain_tool` also supports guarding Python callables directly.
225
+
226
+ ### MCP
227
+
228
+ `System1MCPProxy` intercepts JSON-RPC tool calls. Supply your configured guard and authenticated identity, then use `handle_call(request, executor)` or `async_handle_call(request, executor)` to connect it to your application's executor. See [integration tests](tests/test_integrations.py) for the dispatch contract. Diagnostic mode does not produce trusted signed enforcement evidence by default.
229
+
230
+ ### FastAPI / ASGI
231
+
232
+ `add_system1_gateway(app, schema=YourSchema)` can respond to confident classification requests locally and pass uncertain requests to the downstream application. Install your ASGI framework separately. Configure authentication and request-size limits **outside** this middleware: local responses bypass downstream endpoint dependencies. This is a classification gateway, not a tool authorization boundary. See [deployment guidance](docs/deployment.md#asgi-gateway).
233
+
234
+ ### gRPC
235
+
236
+ ```bash
237
+ python -m pip install 'system1[grpc]'
238
+ system1 serve --grpc --host 127.0.0.1 --port 50051
239
+ ```
240
+
241
+ The bundled [protobuf contract](src/system1/proto/system1.proto) supports clients generated for other languages. The CLI server is a local diagnostic service with insecure gRPC transport. Configure signing, ledger, and policy through the Python `serve(...)` API for a custom deployment. The repository does not supply a published Docker image or Kubernetes manifests.
242
+
243
+ ### Observability
244
+
245
+ Install `system1[observability]` for Prometheus or `system1[otel]` for OpenTelemetry. See the [observability guide](docs/observability/README.md) and [Grafana dashboard](docs/observability/grafana-dashboard.json). Starting a metrics server or configuring an exporter changes the application's network behavior.
246
+
247
+ ## Benchmarks
248
+
249
+ Reproduce the included seed-model benchmarks from the repository root:
250
+
251
+ ```bash
252
+ python benchmarks/quality/run_quality_benchmarks.py
253
+ system1 bench --schema triage --iterations 200 --json
254
+ ```
255
+
256
+ The 100-example datasets are small, repository-authored evaluations, not independent security certifications. The launch-review run measured:
257
+
258
+ | Task | Overall quality | Additional metric | Median latency |
259
+ |---|---|---|---|
260
+ | Security triage | 50% accuracy; 0.486 macro-F1 | BLOCK recall: 0.371 | 0.410 ms |
261
+ | Intent routing | 51% accuracy; 0.495 macro-F1 | Billing F1: 0.653 | 0.488 ms |
262
+ | Threat scoring, 0–10 | MAE: 3.068; RMSE: 3.480 | Pearson r: 0.309 | 0.469 ms |
263
+
264
+ See the [recorded results and environment](benchmarks/quality/results/launch_review.json) and [methodology](benchmarks/quality/README.md). These are raw classification results, not the accuracy of authorized tool actions. Latency is workload- and hardware-dependent; these measurements do not establish durable end-to-end authorization latency or a service-level guarantee. No cloud providers were measured in this run.
265
+
266
+ Teaching from the existing examples improves raw accuracy in a separate five-fold development check: intent routing reaches 68% and security triage 71% with 2048 features. Both still require review on every case under strict uncertainty gating; their calibration sets are too small. This is evidence that examples help, not release acceptance evidence. See the [teaching comparison](benchmarks/quality/README.md#teaching-comparison) for default-dimension results, protocol, and limitations.
267
+
268
+ Conformal coverage applies to prediction sets under exchangeability and appropriate held-out calibration. It does not guarantee that a singleton prediction is safe, control the error rate conditional on local acceptance, or imply a particular local-retention percentage. Distribution shift and model updates require reevaluation. See [limits](docs/deployment.md#statistical-limits) and the [conformal prediction introduction](https://arxiv.org/abs/2107.07511).
269
+
270
+ ## Examples and research
271
+
272
+ - [Support triage](examples/support_triage.py), [model routing](examples/model_routing.py), and [agent guard](examples/agent_guard.py) teach and check one skill each. Their [datasets and measured results](examples/teaching/README.md) are included.
273
+ - [Teach one skill](examples/teach_skill.py), [advanced expert examples](examples/train_expert.py), and [observed teaching and takeover](examples/observe_routing.py). Cloud modes require explicit configuration.
274
+ - [Gaming examples](examples/gaming/) explore simulated environments and optional local emulation; they are not independently verified world records. ROMs are not included.
275
+ - [Research manuscripts](docs/paper/) describe the design and earlier experiments. Their historical timing and quality claims are not release acceptance criteria; use the reproducible measurements above.
276
+
277
+ Both `import system1` and the legacy `import reflex` expose the same API. The `reflex` namespace can conflict with the separate Reflex web-framework package, so use separate environments when needed. The [TypeSafe adapter](docs/typesafe.md) supports the basic sync/async decision API, structured state, typed response accessors, and ordinal score distributions. Its documented contract does not include the entire SDK transport/Pydantic surface.
278
+
279
+ ## CLI
280
+
281
+ ```bash
282
+ system1 decide "How do I reset my password?" --schema triage --json
283
+ system1 decide "Read documentation" --schema guard --sign --ledger audit.sqlite --json
284
+ system1 verify-receipt receipt.json --public-key identity.pub
285
+ system1 calibrate --dataset data.json --schema triage --bins 10
286
+ system1 compile --schema triage --output triage.s1m --json
287
+ ```
288
+
289
+ Use `system1 --help` or `system1 <command> --help` for options. Receipt verification requires a trusted public key supplied independently of the receipt.
290
+
291
+ ## Development
292
+
293
+ ```bash
294
+ python -m pip install -e '.[dev,langchain]'
295
+ python -m pytest tests/ -q
296
+ ```
297
+
298
+ Or use the committed lockfile with `uv sync --locked --extra dev --extra langchain` and `uv run --no-sync pytest -q`. CI tests Linux and macOS, checks built distributions, and exercises real LangChain dispatch. Optional MLX and emulator tests require their extras and suitable hardware/assets. See [contributing](CONTRIBUTING.md) and [security reporting](SECURITY.md).
299
+
300
+ ## License
301
+
302
+ [Apache License 2.0](LICENSE).
@@ -0,0 +1,249 @@
1
+ <p align="center">
2
+ <img src="https://raw.githubusercontent.com/steph4n-gh/system1/main/assets/system1-logo.jpg" alt="System 1 logo" width="128" />
3
+ </p>
4
+
5
+ # System 1: Local Decision Runtime for AI Agents
6
+
7
+ Teach a repeatable decision skill from examples or by observing a teacher such as Jev. Validate it, take over locally, and save the skill for reuse.
8
+
9
+ [![CI](https://github.com/steph4n-gh/system1/actions/workflows/ci.yml/badge.svg)](https://github.com/steph4n-gh/system1/actions/workflows/ci.yml)
10
+ [![PyPI](https://img.shields.io/pypi/v/system1.svg)](https://pypi.org/project/system1/)
11
+ [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)](pyproject.toml)
12
+ [![License](https://img.shields.io/badge/license-Apache_2.0-blue.svg)](LICENSE)
13
+
14
+ **Status: beta.** System 1 provides a small local classifier and a separate deterministic policy guard. The included seed models need evaluation and calibration on your workload. They are not a general-purpose detector of malicious actions or prompt injection.
15
+
16
+ [Quickstart](#quickstart) · [Policy guard](#policy-guard) · [Integrations](#integrations) · [Benchmarks](#benchmarks) · [Limits and deployment](docs/deployment.md) · [Contributing](CONTRIBUTING.md)
17
+
18
+ ## What it does
19
+
20
+ - **Observe → teach → validate → run locally:** keep the teacher answering until the observed skill passes held-out checks, then use the same call site locally. Save and reload the validated skill with its uncertainty behavior intact.
21
+ - **Structured classification:** choice, boolean, multi-choice, and score fields using local NumPy projections, with optional MLX acceleration.
22
+ - **Uncertainty handling:** calibration and conformal prediction sets for routing uncertain decisions to application-defined review or fallback paths.
23
+ - **Deterministic permissions:** `PolicyEngine` evaluates explicit rules; `SystemOneGuard(enforcement_profile=True)` requires a matching permission grant, signing key, and durable ledger before returning `ALLOW`.
24
+ - **Audit evidence:** Ed25519 software signatures (RFC 8032) on decision receipts and a SHA-256 hash-chained SQLite ledger. Applications remain responsible for authenticating callers and enforcing decisions at the tool boundary.
25
+ - **Local execution:** the core decision path requires no network or cloud API. Optional fallback clients, telemetry exporters, and application tools can use the network.
26
+
27
+ ## Quickstart
28
+
29
+ The project is packaged on PyPI as `system1`. Use Python 3.11 or newer:
30
+
31
+ ```bash
32
+ python -m pip install system1
33
+ ```
34
+
35
+ To try the code in this repository:
36
+
37
+ ```bash
38
+ git clone https://github.com/steph4n-gh/system1.git
39
+ cd system1
40
+ python -m venv .venv
41
+ source .venv/bin/activate
42
+ python -m pip install -e '.[dev]'
43
+ ```
44
+
45
+ Define the outputs you need, then evaluate a prompt:
46
+
47
+ ```python
48
+ from system1 import ChoiceField, DecisionSchema, System1Engine
49
+
50
+ class SupportRoute(DecisionSchema):
51
+ team = ChoiceField(
52
+ options=["billing", "technical_support"],
53
+ descriptions={
54
+ "billing": "Invoices, payments, subscription plans, and refunds",
55
+ "technical_support": "Software installation, errors, and debugging",
56
+ },
57
+ )
58
+
59
+ engine = System1Engine(SupportRoute)
60
+ decision = engine.decide("I need a refund for my subscription")
61
+ print(decision.values)
62
+ print(f"Latency: {decision.latency_ms:.2f} ms")
63
+ print(f"Needs review: {decision.is_ambiguous}")
64
+ print(f"Prediction sets: {decision.conformal_sets}")
65
+ ```
66
+
67
+ This starts with schema-derived seed prototypes. Outputs and latency depend on the schema, data, and hardware. A predicted value is not permission to execute a tool.
68
+
69
+ ## Observe a teacher, then take over
70
+
71
+ ```bash
72
+ python examples/observe_routing.py
73
+ ```
74
+
75
+ This short offline example shows the complete journey: observe answers, teach one routing skill, pass the normal promotion gates, disconnect the teacher, evaluate fresh requests locally, then save and reopen the skill. In the recorded run it needed **190 observations**, answered **40/40 fresh synthetic tickets correctly**, accepted all 40, and saved a **5 KB** skill. Reloading preserved answers, probabilities, and review decisions. Local median latency was about **0.45 ms** on the review machine.
76
+
77
+ The default teacher is a rule over a small structured-ticket vocabulary, clearly labeled as a simulation. It demonstrates the lifecycle, not general language understanding or Jev quality parity. `--teacher jev` observes actual Jev responses with your API key; another API can use the existing teacher callback. See the [adapter guide and compatibility contract](docs/typesafe.md) and [recorded evidence](examples/teaching/results/observed_routing.json).
78
+
79
+ Promotion depends on evidence, not a fixed turn count. Insufficient evidence keeps the teacher active, and uncertain local responses still request review.
80
+
81
+ ## Teach a skill
82
+
83
+ Give System 1 labeled examples of one task, save the resulting `.s1m` skill, and reuse it locally. A person or System 2 can supply the examples; teaching does not require an LLM. The existing compiler fits a small decision head using NumPy.
84
+
85
+ ```bash
86
+ python examples/support_triage.py
87
+ python examples/model_routing.py
88
+ python examples/agent_guard.py
89
+ system1 decide "Please correct the invoice address" --model .system1/examples/support_triage/skill.s1m --json
90
+ ```
91
+
92
+ Each primary example supplies labeled teaching cases, separate calibration cases, and 24 unseen evaluation cases. One command teaches, saves, reloads, and reports the results. On the review machine, teaching plus calibration took 127–159 ms, saved skills were 25–32 KiB, and uncached decisions took about 0.5 ms. Preparing good labeled examples takes additional work.
93
+
94
+ The authored demonstration cases reached 87.5% support-routing accuracy, 91.7% model-routing accuracy, and 100% operation-triage accuracy. Strict uncertainty checks requested review on 23/24 support cases, all 24 routing cases, and 22/24 operation cases. The three accepted responses were correct. See the [complete results, data, and limits](examples/teaching/README.md); these are demonstrations, not production quality guarantees.
95
+
96
+ For the smallest API example, see [teach one skill](examples/teach_skill.py). The [teaching guide](docs/guides/training_experts.md) covers your own data and evaluation. In Python, use `compile(examples, augment=False)` and `System1Engine(..., strict_mode=True)` for this workflow.
97
+
98
+ ## Policy guard
99
+
100
+ This executable example permits one configuration lookup for one application-supplied principal, records the actual lookup result, and verifies the ledger. Other keys or tools have no grant. It persists a local demonstration key across runs.
101
+
102
+ ```python
103
+ from pathlib import Path
104
+ from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
105
+ from system1 import (
106
+ ActionLedger, ActionProposal, DecisionOutcome, PolicyEngine, PolicyRule,
107
+ RiskLevel, SystemOneGuard, load_private_key, save_keypair,
108
+ )
109
+
110
+ key_path = Path(".system1/demo-identity/identity.key")
111
+ if not key_path.exists():
112
+ save_keypair(Ed25519PrivateKey.generate(), key_path.parent)
113
+ signing_key = load_private_key(key_path)
114
+
115
+ policy = PolicyEngine(rules=[PolicyRule(
116
+ rule_id="read_service_name",
117
+ tools=["read_config"],
118
+ allowed_principals=["agent-worker"],
119
+ allowed_tenants=["demo"],
120
+ allowed_scopes=["config:read"],
121
+ argument_limits={"key": ["service_name"]},
122
+ outcome=DecisionOutcome.ALLOW,
123
+ risk=RiskLevel.READ_ONLY,
124
+ )])
125
+
126
+ with ActionLedger(".system1/demo-audit.sqlite", require_durable=True) as ledger:
127
+ guard = SystemOneGuard(
128
+ policy=policy, ledger=ledger, signing_key=signing_key,
129
+ enforcement_profile=True,
130
+ )
131
+ proposal = ActionProposal.create(
132
+ tenant_id="demo", principal_id="agent-worker", scope="config:read",
133
+ tool="read_config", arguments={"key": "service_name"},
134
+ canonical_target="config:service_name", purpose="Inspect service name",
135
+ )
136
+ auth = guard.evaluate_proposal(proposal)
137
+ if auth.outcome != DecisionOutcome.ALLOW:
138
+ raise PermissionError(auth.reason)
139
+
140
+ # Execute exactly the authorized operation and arguments.
141
+ result = {"service_name": "system1-demo"}[proposal.arguments["key"]]
142
+ ledger.record_execution_outcome(
143
+ action_id=proposal.action_id,
144
+ receipt_digest=auth.receipt.compute_digest(),
145
+ status="SUCCEEDED", result_payload={"value": result},
146
+ tenant_id=proposal.tenant_id, principal_id=proposal.principal_id,
147
+ scope=proposal.scope, trusted_public_key=signing_key.public_key(),
148
+ )
149
+ assert ledger.verify_integrity(trusted_public_key=signing_key.public_key())
150
+ print(result)
151
+ ```
152
+
153
+ The application must derive identity from its authenticated session, constrain tool arguments and targets, and keep agent code from bypassing the guard. Policy correctness is the operator's responsibility. The default guard without `enforcement_profile=True` can use statistical classification to allow actions. See [deployment boundaries](docs/deployment.md) before granting consequential permissions.
154
+
155
+ ## Integrations
156
+
157
+ ### LangChain
158
+
159
+ Install `python -m pip install 'system1[langchain]'`. Attach a configured guard to actual tool invocations:
160
+
161
+ ```python
162
+ from system1.integrations import SystemOneGuardCallbackHandler
163
+
164
+ # guard is your configured SystemOneGuard, with a live ledger and signing key.
165
+ handler = SystemOneGuardCallbackHandler(
166
+ guard=guard, tenant_id="demo", principal_id="agent-worker",
167
+ )
168
+ result = your_tool.invoke(tool_arguments, config={"callbacks": [handler]})
169
+ ```
170
+
171
+ The callback uses scope `langchain:tools:exec` and the serialized tool input in its proposal. Match policy rules to that contract. `SystemOneGuardBlockedException` propagates to the caller on a deny or approval requirement. `wrap_langchain_tool` also supports guarding Python callables directly.
172
+
173
+ ### MCP
174
+
175
+ `System1MCPProxy` intercepts JSON-RPC tool calls. Supply your configured guard and authenticated identity, then use `handle_call(request, executor)` or `async_handle_call(request, executor)` to connect it to your application's executor. See [integration tests](tests/test_integrations.py) for the dispatch contract. Diagnostic mode does not produce trusted signed enforcement evidence by default.
176
+
177
+ ### FastAPI / ASGI
178
+
179
+ `add_system1_gateway(app, schema=YourSchema)` can respond to confident classification requests locally and pass uncertain requests to the downstream application. Install your ASGI framework separately. Configure authentication and request-size limits **outside** this middleware: local responses bypass downstream endpoint dependencies. This is a classification gateway, not a tool authorization boundary. See [deployment guidance](docs/deployment.md#asgi-gateway).
180
+
181
+ ### gRPC
182
+
183
+ ```bash
184
+ python -m pip install 'system1[grpc]'
185
+ system1 serve --grpc --host 127.0.0.1 --port 50051
186
+ ```
187
+
188
+ The bundled [protobuf contract](src/system1/proto/system1.proto) supports clients generated for other languages. The CLI server is a local diagnostic service with insecure gRPC transport. Configure signing, ledger, and policy through the Python `serve(...)` API for a custom deployment. The repository does not supply a published Docker image or Kubernetes manifests.
189
+
190
+ ### Observability
191
+
192
+ Install `system1[observability]` for Prometheus or `system1[otel]` for OpenTelemetry. See the [observability guide](docs/observability/README.md) and [Grafana dashboard](docs/observability/grafana-dashboard.json). Starting a metrics server or configuring an exporter changes the application's network behavior.
193
+
194
+ ## Benchmarks
195
+
196
+ Reproduce the included seed-model benchmarks from the repository root:
197
+
198
+ ```bash
199
+ python benchmarks/quality/run_quality_benchmarks.py
200
+ system1 bench --schema triage --iterations 200 --json
201
+ ```
202
+
203
+ The 100-example datasets are small, repository-authored evaluations, not independent security certifications. The launch-review run measured:
204
+
205
+ | Task | Overall quality | Additional metric | Median latency |
206
+ |---|---|---|---|
207
+ | Security triage | 50% accuracy; 0.486 macro-F1 | BLOCK recall: 0.371 | 0.410 ms |
208
+ | Intent routing | 51% accuracy; 0.495 macro-F1 | Billing F1: 0.653 | 0.488 ms |
209
+ | Threat scoring, 0–10 | MAE: 3.068; RMSE: 3.480 | Pearson r: 0.309 | 0.469 ms |
210
+
211
+ See the [recorded results and environment](benchmarks/quality/results/launch_review.json) and [methodology](benchmarks/quality/README.md). These are raw classification results, not the accuracy of authorized tool actions. Latency is workload- and hardware-dependent; these measurements do not establish durable end-to-end authorization latency or a service-level guarantee. No cloud providers were measured in this run.
212
+
213
+ Teaching from the existing examples improves raw accuracy in a separate five-fold development check: intent routing reaches 68% and security triage 71% with 2048 features. Both still require review on every case under strict uncertainty gating; their calibration sets are too small. This is evidence that examples help, not release acceptance evidence. See the [teaching comparison](benchmarks/quality/README.md#teaching-comparison) for default-dimension results, protocol, and limitations.
214
+
215
+ Conformal coverage applies to prediction sets under exchangeability and appropriate held-out calibration. It does not guarantee that a singleton prediction is safe, control the error rate conditional on local acceptance, or imply a particular local-retention percentage. Distribution shift and model updates require reevaluation. See [limits](docs/deployment.md#statistical-limits) and the [conformal prediction introduction](https://arxiv.org/abs/2107.07511).
216
+
217
+ ## Examples and research
218
+
219
+ - [Support triage](examples/support_triage.py), [model routing](examples/model_routing.py), and [agent guard](examples/agent_guard.py) teach and check one skill each. Their [datasets and measured results](examples/teaching/README.md) are included.
220
+ - [Teach one skill](examples/teach_skill.py), [advanced expert examples](examples/train_expert.py), and [observed teaching and takeover](examples/observe_routing.py). Cloud modes require explicit configuration.
221
+ - [Gaming examples](examples/gaming/) explore simulated environments and optional local emulation; they are not independently verified world records. ROMs are not included.
222
+ - [Research manuscripts](docs/paper/) describe the design and earlier experiments. Their historical timing and quality claims are not release acceptance criteria; use the reproducible measurements above.
223
+
224
+ Both `import system1` and the legacy `import reflex` expose the same API. The `reflex` namespace can conflict with the separate Reflex web-framework package, so use separate environments when needed. The [TypeSafe adapter](docs/typesafe.md) supports the basic sync/async decision API, structured state, typed response accessors, and ordinal score distributions. Its documented contract does not include the entire SDK transport/Pydantic surface.
225
+
226
+ ## CLI
227
+
228
+ ```bash
229
+ system1 decide "How do I reset my password?" --schema triage --json
230
+ system1 decide "Read documentation" --schema guard --sign --ledger audit.sqlite --json
231
+ system1 verify-receipt receipt.json --public-key identity.pub
232
+ system1 calibrate --dataset data.json --schema triage --bins 10
233
+ system1 compile --schema triage --output triage.s1m --json
234
+ ```
235
+
236
+ Use `system1 --help` or `system1 <command> --help` for options. Receipt verification requires a trusted public key supplied independently of the receipt.
237
+
238
+ ## Development
239
+
240
+ ```bash
241
+ python -m pip install -e '.[dev,langchain]'
242
+ python -m pytest tests/ -q
243
+ ```
244
+
245
+ Or use the committed lockfile with `uv sync --locked --extra dev --extra langchain` and `uv run --no-sync pytest -q`. CI tests Linux and macOS, checks built distributions, and exercises real LangChain dispatch. Optional MLX and emulator tests require their extras and suitable hardware/assets. See [contributing](CONTRIBUTING.md) and [security reporting](SECURITY.md).
246
+
247
+ ## License
248
+
249
+ [Apache License 2.0](LICENSE).
@@ -0,0 +1,13 @@
1
+ # Security policy
2
+
3
+ ## Reporting a vulnerability
4
+
5
+ Report vulnerabilities privately through [GitHub's private vulnerability reporting form](https://github.com/steph4n-gh/system1/security/advisories/new). If private reporting is unavailable, open an issue requesting a private contact without posting exploit details, credentials, or sensitive data.
6
+
7
+ Include the affected version, configuration, expected boundary, reproduction steps using harmless sentinel tools, and observed result. Please avoid testing against services or data you do not own.
8
+
9
+ ## Supported scope
10
+
11
+ System 1 is in beta. Fixes target the latest release; older versions are not maintained as separate branches. Report failures in deterministic policy enforcement, signed receipt verification, ledger integrity, model loading, or integration dispatch.
12
+
13
+ The statistical classifier is not a general-purpose security detector. The application supplies authenticated identity, isolates tools, protects policy and signing keys, and enforces returned decisions. Read [deployment boundaries](docs/deployment.md) before using the library for consequential actions.
Binary file