system1 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- system1-0.2.2/CHANGELOG.md +24 -0
- system1-0.2.2/CONTRIBUTING.md +34 -0
- system1-0.2.2/MANIFEST.in +7 -0
- system1-0.2.2/PKG-INFO +302 -0
- system1-0.2.2/README.md +249 -0
- system1-0.2.2/SECURITY.md +13 -0
- system1-0.2.2/assets/architecture.png +0 -0
- system1-0.2.2/assets/architecture.svg +132 -0
- system1-0.2.2/assets/benchmark-chart.png +0 -0
- system1-0.2.2/assets/benchmark-chart.svg +99 -0
- system1-0.2.2/assets/paperclips-cutover.gif +0 -0
- system1-0.2.2/assets/pokemon-system1-60fps.gif +0 -0
- system1-0.2.2/assets/privacy-zero-egress.png +0 -0
- system1-0.2.2/assets/privacy-zero-egress.svg +184 -0
- system1-0.2.2/assets/quality-vs-latency.svg +128 -0
- system1-0.2.2/assets/social-preview.png +0 -0
- system1-0.2.2/assets/social-preview.svg +84 -0
- system1-0.2.2/assets/system1-logo.jpg +0 -0
- system1-0.2.2/benchmarks/__init__.py +0 -0
- system1-0.2.2/benchmarks/quality/README.md +140 -0
- system1-0.2.2/benchmarks/quality/datasets/intent_routing.json +110 -0
- system1-0.2.2/benchmarks/quality/datasets/security_triage.json +111 -0
- system1-0.2.2/benchmarks/quality/datasets/threat_scoring.json +112 -0
- system1-0.2.2/benchmarks/quality/evaluate_teaching.py +128 -0
- system1-0.2.2/benchmarks/quality/results/launch_review.json +2788 -0
- system1-0.2.2/benchmarks/quality/results/teaching_review.json +8797 -0
- system1-0.2.2/benchmarks/quality/run_quality_benchmarks.py +637 -0
- system1-0.2.2/benchmarks/results/triple_crown_scorecard.json +139 -0
- system1-0.2.2/benchmarks/run_triple_crown_benchmark.py +1105 -0
- system1-0.2.2/docs/SPEEDRUN_SHOWDOWN_WORLD_RECORDS.md +180 -0
- system1-0.2.2/docs/architecture/technical_specification.md +575 -0
- system1-0.2.2/docs/deployment.md +57 -0
- system1-0.2.2/docs/examples-review.md +166 -0
- system1-0.2.2/docs/guides/training_experts.md +152 -0
- system1-0.2.2/docs/launch-review.md +103 -0
- system1-0.2.2/docs/observability/README.md +195 -0
- system1-0.2.2/docs/observability/grafana-dashboard.json +186 -0
- system1-0.2.2/docs/paper/conformal_gating.md +193 -0
- system1-0.2.2/docs/paper/system1_technical_brief.md +206 -0
- system1-0.2.2/docs/paper/system1_whitepaper.md +582 -0
- system1-0.2.2/docs/typesafe.md +132 -0
- system1-0.2.2/examples/_teaching_demo.py +126 -0
- system1-0.2.2/examples/agent_guard.py +74 -0
- system1-0.2.2/examples/auto_cutover_showcase.py +915 -0
- system1-0.2.2/examples/autonomous_agent_firewall_showcase.py +1455 -0
- system1-0.2.2/examples/core_standalone_evaluator.py +145 -0
- system1-0.2.2/examples/deep_jev_benchmark.py +414 -0
- system1-0.2.2/examples/enterprise_stress_showcase.py +29 -0
- system1-0.2.2/examples/four_levers_benchmark.py +290 -0
- system1-0.2.2/examples/gaming/__init__.py +1 -0
- system1-0.2.2/examples/gaming/paperclips_speedrun.py +1216 -0
- system1-0.2.2/examples/gaming/pokemon_all_games_benchmark.py +837 -0
- system1-0.2.2/examples/gaming/pokemon_battle_system1.py +2208 -0
- system1-0.2.2/examples/gaming/pokemon_full_campaign_speedrun.py +2438 -0
- system1-0.2.2/examples/gaming/pokemon_gameboy_gui.py +939 -0
- system1-0.2.2/examples/gaming/pokemon_kaizo_speedrun.py +516 -0
- system1-0.2.2/examples/gaming/pokemon_showdown_system1.py +851 -0
- system1-0.2.2/examples/jev_comparison_demos.py +303 -0
- system1-0.2.2/examples/killer_use_cases_live_test.py +539 -0
- system1-0.2.2/examples/model_routing.py +33 -0
- system1-0.2.2/examples/observe_routing.py +143 -0
- system1-0.2.2/examples/paperclips_typesafe_dropin.py +703 -0
- system1-0.2.2/examples/support_triage.py +34 -0
- system1-0.2.2/examples/teach_skill.py +50 -0
- system1-0.2.2/examples/teaching/README.md +123 -0
- system1-0.2.2/examples/teaching/agent_guard.json +646 -0
- system1-0.2.2/examples/teaching/model_routing.json +646 -0
- system1-0.2.2/examples/teaching/results/agent_guard.json +390 -0
- system1-0.2.2/examples/teaching/results/model_routing.json +392 -0
- system1-0.2.2/examples/teaching/results/observed_routing.json +341 -0
- system1-0.2.2/examples/teaching/results/support_triage.json +421 -0
- system1-0.2.2/examples/teaching/support_triage.json +697 -0
- system1-0.2.2/examples/train_expert.py +348 -0
- system1-0.2.2/examples/typesafe_sdk_dropin_showcase.py +295 -0
- {system1-0.2.0 → system1-0.2.2}/pyproject.toml +9 -6
- system1-0.2.2/scripts/generate_audit_bundle.py +251 -0
- system1-0.2.2/scripts/generate_paperclips_gif.py +208 -0
- system1-0.2.2/scripts/generate_pokemon_gif.py +160 -0
- system1-0.2.2/scripts/run_paperclips_speedrun.py +33 -0
- system1-0.2.2/scripts/run_pokemon_kaizo.py +33 -0
- system1-0.2.2/scripts/run_pokemon_showdown.py +33 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/__init__.py +6 -6
- {system1-0.2.0 → system1-0.2.2}/src/reflex/core/__init__.py +1 -1
- system1-0.2.2/src/reflex/proto/__init__.py +17 -0
- system1-0.2.2/src/reflex/proto/system1_pb2.py +10 -0
- system1-0.2.2/src/reflex/proto/system1_pb2_grpc.py +10 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/__init__.py +7 -7
- {system1-0.2.0 → system1-0.2.2}/src/system1/calibration.py +12 -2
- {system1-0.2.0 → system1-0.2.2}/src/system1/cli.py +84 -50
- {system1-0.2.0 → system1-0.2.2}/src/system1/compat/typesafe.py +305 -270
- {system1-0.2.0 → system1-0.2.2}/src/system1/compiler.py +260 -118
- {system1-0.2.0 → system1-0.2.2}/src/system1/core/__init__.py +1 -1
- {system1-0.2.0 → system1-0.2.2}/src/system1/engine.py +8 -6
- {system1-0.2.0 → system1-0.2.2}/src/system1/grpc_server.py +1 -1
- {system1-0.2.0 → system1-0.2.2}/src/system1/guard.py +4 -4
- {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/__init__.py +6 -1
- {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/fastapi.py +26 -4
- {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/langchain.py +9 -6
- {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/mcp.py +4 -4
- {system1-0.2.0 → system1-0.2.2}/src/system1/proto/__init__.py +9 -3
- system1-0.2.2/src/system1/proto/system1.proto +171 -0
- system1-0.2.2/src/system1.egg-info/PKG-INFO +302 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/SOURCES.txt +106 -1
- {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/requires.txt +5 -2
- system1-0.2.2/tests/conftest.py +18 -0
- system1-0.2.2/tests/e2e/__init__.py +1 -0
- system1-0.2.2/tests/e2e/conftest.py +88 -0
- system1-0.2.2/tests/e2e/test_tier1_cli.py +182 -0
- system1-0.2.2/tests/e2e/test_tier1_conformal.py +197 -0
- system1-0.2.2/tests/e2e/test_tier1_dual_imports.py +166 -0
- system1-0.2.2/tests/e2e/test_tier1_latency.py +176 -0
- system1-0.2.2/tests/e2e/test_tier1_ledger.py +138 -0
- system1-0.2.2/tests/e2e/test_tier1_receipts.py +175 -0
- system1-0.2.2/tests/e2e/test_tier1_zero_network.py +99 -0
- system1-0.2.2/tests/e2e/test_tier2_boundaries.py +177 -0
- system1-0.2.2/tests/e2e/test_tier3_combinations.py +185 -0
- system1-0.2.2/tests/e2e/test_tier4_real_world.py +212 -0
- system1-0.2.2/tests/review_acceptance/__init__.py +1 -0
- system1-0.2.2/tests/review_acceptance/test_postfix_boundaries.py +127 -0
- system1-0.2.2/tests/review_acceptance/test_previous_contracts_adapted.py +241 -0
- system1-0.2.2/tests/review_acceptance/test_round4_variants.py +76 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_crypto_compat_cli.py +3 -3
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_d.py +6 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_f.py +4 -3
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_invariants_6_to_10.py +2 -2
- {system1-0.2.0 → system1-0.2.2}/tests/test_auto_cutover.py +7 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_autonomous_agent_firewall_showcase.py +4 -2
- {system1-0.2.0 → system1-0.2.2}/tests/test_cli.py +5 -5
- {system1-0.2.0 → system1-0.2.2}/tests/test_conformal_invariants.py +3 -3
- {system1-0.2.0 → system1-0.2.2}/tests/test_cutover_invariants.py +5 -1
- system1-0.2.2/tests/test_gateway_boundaries.py +109 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_grpc_external_generated_client.py +1 -1
- {system1-0.2.0 → system1-0.2.2}/tests/test_grpc_polyglot.py +2 -2
- {system1-0.2.0 → system1-0.2.2}/tests/test_grpc_server.py +2 -2
- system1-0.2.2/tests/test_langchain_dispatch.py +76 -0
- system1-0.2.2/tests/test_observation_contract.py +228 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_paperclips_speedrun.py +1 -2
- {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_battle_system1.py +1 -1
- {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_full_campaign_speedrun.py +35 -0
- system1-0.2.2/tests/test_primary_examples.py +71 -0
- system1-0.2.2/tests/test_release_examples.py +70 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_round2_gates_c_d.py +3 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_round2_gates_e_f.py +1 -1
- system1-0.2.2/tests/test_teaching.py +143 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_typesafe_compat.py +14 -14
- system1-0.2.2/uv.lock +1952 -0
- system1-0.2.0/PKG-INFO +0 -546
- system1-0.2.0/README.md +0 -495
- system1-0.2.0/src/reflex/proto/__init__.py +0 -5
- system1-0.2.0/src/system1.egg-info/PKG-INFO +0 -546
- {system1-0.2.0 → system1-0.2.2}/LICENSE +0 -0
- {system1-0.2.0 → system1-0.2.2}/setup.cfg +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/cache.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/calibration.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/cli.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/compat/__init__.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/compat/typesafe.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/compiler.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/core/embeddings.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/core/model.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/core/neural.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/core/schema.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/core/telemetry.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/embeddings.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/engine.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/grpc_server.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/guard.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/__init__.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/fastapi.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/langchain.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/mcp.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/observability.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/integrations/otel.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/ledger.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/model.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/neural.py +0 -0
- {system1-0.2.0/src/system1 → system1-0.2.2/src/reflex}/proto/system1.proto +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/receipt.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/schema.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/reflex/telemetry.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/cache.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/compat/__init__.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/core/embeddings.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/core/model.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/core/neural.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/core/schema.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/core/telemetry.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/embeddings.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/observability.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/integrations/otel.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/ledger.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/model.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/neural.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/proto/system1_pb2.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/proto/system1_pb2_grpc.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/receipt.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/schema.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1/telemetry.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/dependency_links.txt +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/entry_points.txt +0 -0
- {system1-0.2.0 → system1-0.2.2}/src/system1.egg-info/top_level.txt +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_a.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_b.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_c.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_gate_e.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_invariants_challenger_1.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_m1_parity.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_adversarial_tier5_engine.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_attestation_invariants.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_audit_architectural_upgrades.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_authorization_invariants.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_auto_cutover_showcase.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_cache_invariants.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_calibration.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_cli_compile.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_compiler.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_conformal.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_core_isolation.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_engine.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_four_levers.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_guard.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_hybrid_embeddings.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_integrations.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_model.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_neural_projector.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_observability.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_benchmark.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_kaizo_speedrun.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_pokemon_showdown_system1.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_previous_contracts_adapted.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_receipt.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_round2_gates_a_b.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_round4_variants.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_schema.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_system1_exports.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_train_expert.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_triple_crown_benchmark.py +0 -0
- {system1-0.2.0 → system1-0.2.2}/tests/test_whitening_and_exemplars.py +0 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.2.2 — 2026-09-19
|
|
4
|
+
|
|
5
|
+
- Preserve the complete validated skill through takeover and reload: projector settings, schema identity, calibration, distributions and review behavior. Gate normal promotion on useful local acceptance as well as teacher agreement.
|
|
6
|
+
- Make observed-only teaching and strict review the TypeSafe adapter defaults. Group related observations across all supplied lineage identifiers; insufficient evidence continues using the teacher.
|
|
7
|
+
- Support SDK context managers, JSON object/array state, typed response accessors and ordinal score distributions. Document the supported contract and reject unsupported transport/response options.
|
|
8
|
+
- Correct strict conformal sets to invert their calibrated cumulative-probability scores; retain conservative review when calibration is insufficient.
|
|
9
|
+
- Add the complete offline observation-to-local-reuse example, with optional real Jev observation. Remove fabricated speedups after HTTP failures, counter-only campaign takeover, reconstructed probability distributions and unsupported example claims.
|
|
10
|
+
|
|
11
|
+
- Teach a skill from supplied examples with `compile(..., augment=False)` or CLI `--dataset`, rejecting malformed labels instead of inventing replacements. Keep repeated prompts out of separate calibration partitions and preserve uncalibrated status when no calibration examples exist.
|
|
12
|
+
- Load saved skills directly with `system1 decide --model skill.s1m`, using strict uncertainty gating. Add a small teaching example and reproducible comparison on unseen examples.
|
|
13
|
+
- Replace the three primary showcases with focused, offline teaching demonstrations: explicit labeled datasets, separate calibration and evaluation cases, saved skills, and measured quality, review frequency, size, and speed. Keep agent permissions under deterministic policy rules.
|
|
14
|
+
- Propagate LangChain callback denials and approval requirements through synchronous and asynchronous tool dispatch. Preserve the configured principal instead of substituting a run ID.
|
|
15
|
+
- Handle non-object JSON, malformed multipart text, disconnects, and one-time body replay in the ASGI gateway; respect structured escalation signals.
|
|
16
|
+
- Return actual receipt digests in gateway headers and MCP errors.
|
|
17
|
+
- Add real LangChain dispatch regression tests, built-distribution checks, Python 3.14 CI coverage, and release-tag validation.
|
|
18
|
+
- Keep legacy proto assets accessible without optional gRPC/protobuf dependencies in minimal wheel installs.
|
|
19
|
+
- Update package build requirements and the lockfile; include examples and test support files in source distributions.
|
|
20
|
+
- Replace unsupported launch claims with reproducible benchmark results, executable quickstarts, and explicit deployment boundaries.
|
|
21
|
+
|
|
22
|
+
## 0.2.1
|
|
23
|
+
|
|
24
|
+
Previous published baseline. See the [repository history](https://github.com/steph4n-gh/system1/commits/main/) for earlier changes.
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Report reproducible bugs and feature requests through the [issue tracker](https://github.com/steph4n-gh/system1/issues). For vulnerabilities, follow [SECURITY.md](SECURITY.md).
|
|
4
|
+
|
|
5
|
+
## Set up
|
|
6
|
+
|
|
7
|
+
Use Python 3.11 or newer:
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
git clone https://github.com/steph4n-gh/system1.git
|
|
11
|
+
cd system1
|
|
12
|
+
python -m venv .venv
|
|
13
|
+
source .venv/bin/activate
|
|
14
|
+
python -m pip install -e '.[dev,langchain]'
|
|
15
|
+
python -m pytest -q
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
With uv, use `uv sync --locked --extra dev --extra langchain` and `uv run --no-sync pytest -q`. LangChain integration tests use the real callback dispatcher. MLX and emulator coverage requires optional extras and suitable hardware/assets; skipped optional tests are reported by pytest.
|
|
19
|
+
|
|
20
|
+
## Changes
|
|
21
|
+
|
|
22
|
+
Keep changes focused and prefer the simplest implementation that addresses a reproduced problem. Add regression coverage for behavior changes, preserve `system1`/`reflex` API parity, and update runnable documentation when the API changes. Do not include private keys, tokens, ledger data, or proprietary ROMs.
|
|
23
|
+
|
|
24
|
+
Include the problem, resulting behavior, and relevant validation in pull requests. Benchmark claims should include the command, source revision, environment, dataset, overall quality metrics, and raw results. Simulations and cloud measurements must be clearly distinguished.
|
|
25
|
+
|
|
26
|
+
## Package checks
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
uv lock --check
|
|
30
|
+
uv build
|
|
31
|
+
uvx twine check --strict dist/*
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
CI also installs the wheel into a clean environment and runs the CLI outside the checkout. A release tag must match the version in `pyproject.toml`. PyPI publishing requires the repository's `pypi` environment and trusted-publisher configuration; tagging and publishing are separate release actions.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
include LICENSE README.md CONTRIBUTING.md SECURITY.md CHANGELOG.md uv.lock
|
|
2
|
+
recursive-include tests *.py
|
|
3
|
+
recursive-include examples *.py *.json *.md
|
|
4
|
+
recursive-include scripts *.py
|
|
5
|
+
recursive-include benchmarks *.py *.json *.md
|
|
6
|
+
recursive-include docs *.md *.json
|
|
7
|
+
recursive-include assets *.svg *.png *.jpg *.gif
|
system1-0.2.2/PKG-INFO
ADDED
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: system1
|
|
3
|
+
Version: 0.2.2
|
|
4
|
+
Summary: Local structured decisions, deterministic tool policies, conformal uncertainty gating, and signed audit receipts.
|
|
5
|
+
Author: System 1 Authors
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/steph4n-gh/system1
|
|
8
|
+
Project-URL: Repository, https://github.com/steph4n-gh/system1.git
|
|
9
|
+
Project-URL: Documentation, https://github.com/steph4n-gh/system1#readme
|
|
10
|
+
Project-URL: Issue Tracker, https://github.com/steph4n-gh/system1/issues
|
|
11
|
+
Keywords: reflex,system1,decision-runtime,sub-millisecond,apple-silicon,mlx,conformal-prediction,dual-process,zero-egress,ed25519,audit-receipts
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Security :: Cryptography
|
|
21
|
+
Requires-Python: >=3.11
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: numpy>=1.24.0
|
|
25
|
+
Requires-Dist: cryptography>=41.0.0
|
|
26
|
+
Provides-Extra: metal
|
|
27
|
+
Requires-Dist: mlx>=0.10.0; extra == "metal"
|
|
28
|
+
Provides-Extra: neural
|
|
29
|
+
Requires-Dist: mlx>=0.10.0; extra == "neural"
|
|
30
|
+
Provides-Extra: gameboy
|
|
31
|
+
Requires-Dist: pyboy>=2.0.0; extra == "gameboy"
|
|
32
|
+
Provides-Extra: observability
|
|
33
|
+
Requires-Dist: prometheus_client>=0.17.0; extra == "observability"
|
|
34
|
+
Provides-Extra: otel
|
|
35
|
+
Requires-Dist: opentelemetry-api>=1.20.0; extra == "otel"
|
|
36
|
+
Requires-Dist: opentelemetry-sdk>=1.20.0; extra == "otel"
|
|
37
|
+
Provides-Extra: dev
|
|
38
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
39
|
+
Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
|
|
40
|
+
Requires-Dist: grpcio>=1.80.0; extra == "dev"
|
|
41
|
+
Requires-Dist: grpcio-tools>=1.80.0; extra == "dev"
|
|
42
|
+
Requires-Dist: protobuf>=7.35.1; extra == "dev"
|
|
43
|
+
Requires-Dist: prometheus_client>=0.17.0; extra == "dev"
|
|
44
|
+
Requires-Dist: opentelemetry-api>=1.20.0; extra == "dev"
|
|
45
|
+
Requires-Dist: opentelemetry-sdk>=1.20.0; extra == "dev"
|
|
46
|
+
Provides-Extra: grpc
|
|
47
|
+
Requires-Dist: grpcio>=1.80.0; extra == "grpc"
|
|
48
|
+
Requires-Dist: grpcio-tools>=1.80.0; extra == "grpc"
|
|
49
|
+
Requires-Dist: protobuf>=7.35.1; extra == "grpc"
|
|
50
|
+
Provides-Extra: langchain
|
|
51
|
+
Requires-Dist: langchain-core<2,>=1.0; extra == "langchain"
|
|
52
|
+
Dynamic: license-file
|
|
53
|
+
|
|
54
|
+
<p align="center">
|
|
55
|
+
<img src="https://raw.githubusercontent.com/steph4n-gh/system1/main/assets/system1-logo.jpg" alt="System 1 logo" width="128" />
|
|
56
|
+
</p>
|
|
57
|
+
|
|
58
|
+
# System 1: Local Decision Runtime for AI Agents
|
|
59
|
+
|
|
60
|
+
Teach a repeatable decision skill from examples or by observing a teacher such as Jev. Validate it, take over locally, and save the skill for reuse.
|
|
61
|
+
|
|
62
|
+
[](https://github.com/steph4n-gh/system1/actions/workflows/ci.yml)
|
|
63
|
+
[](https://pypi.org/project/system1/)
|
|
64
|
+
[](pyproject.toml)
|
|
65
|
+
[](LICENSE)
|
|
66
|
+
|
|
67
|
+
**Status: beta.** System 1 provides a small local classifier and a separate deterministic policy guard. The included seed models need evaluation and calibration on your workload. They are not a general-purpose detector of malicious actions or prompt injection.
|
|
68
|
+
|
|
69
|
+
[Quickstart](#quickstart) · [Policy guard](#policy-guard) · [Integrations](#integrations) · [Benchmarks](#benchmarks) · [Limits and deployment](docs/deployment.md) · [Contributing](CONTRIBUTING.md)
|
|
70
|
+
|
|
71
|
+
## What it does
|
|
72
|
+
|
|
73
|
+
- **Observe → teach → validate → run locally:** keep the teacher answering until the observed skill passes held-out checks, then use the same call site locally. Save and reload the validated skill with its uncertainty behavior intact.
|
|
74
|
+
- **Structured classification:** choice, boolean, multi-choice, and score fields using local NumPy projections, with optional MLX acceleration.
|
|
75
|
+
- **Uncertainty handling:** calibration and conformal prediction sets for routing uncertain decisions to application-defined review or fallback paths.
|
|
76
|
+
- **Deterministic permissions:** `PolicyEngine` evaluates explicit rules; `SystemOneGuard(enforcement_profile=True)` requires a matching permission grant, signing key, and durable ledger before returning `ALLOW`.
|
|
77
|
+
- **Audit evidence:** Ed25519 software signatures (RFC 8032) on decision receipts and a SHA-256 hash-chained SQLite ledger. Applications remain responsible for authenticating callers and enforcing decisions at the tool boundary.
|
|
78
|
+
- **Local execution:** the core decision path requires no network or cloud API. Optional fallback clients, telemetry exporters, and application tools can use the network.
|
|
79
|
+
|
|
80
|
+
## Quickstart
|
|
81
|
+
|
|
82
|
+
The project is packaged on PyPI as `system1`. Use Python 3.11 or newer:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
python -m pip install system1
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
To try the code in this repository:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
git clone https://github.com/steph4n-gh/system1.git
|
|
92
|
+
cd system1
|
|
93
|
+
python -m venv .venv
|
|
94
|
+
source .venv/bin/activate
|
|
95
|
+
python -m pip install -e '.[dev]'
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Define the outputs you need, then evaluate a prompt:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
from system1 import ChoiceField, DecisionSchema, System1Engine
|
|
102
|
+
|
|
103
|
+
class SupportRoute(DecisionSchema):
|
|
104
|
+
team = ChoiceField(
|
|
105
|
+
options=["billing", "technical_support"],
|
|
106
|
+
descriptions={
|
|
107
|
+
"billing": "Invoices, payments, subscription plans, and refunds",
|
|
108
|
+
"technical_support": "Software installation, errors, and debugging",
|
|
109
|
+
},
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
engine = System1Engine(SupportRoute)
|
|
113
|
+
decision = engine.decide("I need a refund for my subscription")
|
|
114
|
+
print(decision.values)
|
|
115
|
+
print(f"Latency: {decision.latency_ms:.2f} ms")
|
|
116
|
+
print(f"Needs review: {decision.is_ambiguous}")
|
|
117
|
+
print(f"Prediction sets: {decision.conformal_sets}")
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
This starts with schema-derived seed prototypes. Outputs and latency depend on the schema, data, and hardware. A predicted value is not permission to execute a tool.
|
|
121
|
+
|
|
122
|
+
## Observe a teacher, then take over
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
python examples/observe_routing.py
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
This short offline example shows the complete journey: observe answers, teach one routing skill, pass the normal promotion gates, disconnect the teacher, evaluate fresh requests locally, then save and reopen the skill. In the recorded run it needed **190 observations**, answered **40/40 fresh synthetic tickets correctly**, accepted all 40, and saved a **5 KB** skill. Reloading preserved answers, probabilities, and review decisions. Local median latency was about **0.45 ms** on the review machine.
|
|
129
|
+
|
|
130
|
+
The default teacher is a rule over a small structured-ticket vocabulary, clearly labeled as a simulation. It demonstrates the lifecycle, not general language understanding or Jev quality parity. `--teacher jev` observes actual Jev responses with your API key; another API can use the existing teacher callback. See the [adapter guide and compatibility contract](docs/typesafe.md) and [recorded evidence](examples/teaching/results/observed_routing.json).
|
|
131
|
+
|
|
132
|
+
Promotion depends on evidence, not a fixed turn count. Insufficient evidence keeps the teacher active, and uncertain local responses still request review.
|
|
133
|
+
|
|
134
|
+
## Teach a skill
|
|
135
|
+
|
|
136
|
+
Give System 1 labeled examples of one task, save the resulting `.s1m` skill, and reuse it locally. A person or System 2 can supply the examples; teaching does not require an LLM. The existing compiler fits a small decision head using NumPy.
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
python examples/support_triage.py
|
|
140
|
+
python examples/model_routing.py
|
|
141
|
+
python examples/agent_guard.py
|
|
142
|
+
system1 decide "Please correct the invoice address" --model .system1/examples/support_triage/skill.s1m --json
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
Each primary example supplies labeled teaching cases, separate calibration cases, and 24 unseen evaluation cases. One command teaches, saves, reloads, and reports the results. On the review machine, teaching plus calibration took 127–159 ms, saved skills were 25–32 KiB, and uncached decisions took about 0.5 ms. Preparing good labeled examples takes additional work.
|
|
146
|
+
|
|
147
|
+
The authored demonstration cases reached 87.5% support-routing accuracy, 91.7% model-routing accuracy, and 100% operation-triage accuracy. Strict uncertainty checks requested review on 23/24 support cases, all 24 routing cases, and 22/24 operation cases. The three accepted responses were correct. See the [complete results, data, and limits](examples/teaching/README.md); these are demonstrations, not production quality guarantees.
|
|
148
|
+
|
|
149
|
+
For the smallest API example, see [teach one skill](examples/teach_skill.py). The [teaching guide](docs/guides/training_experts.md) covers your own data and evaluation. In Python, use `compile(examples, augment=False)` and `System1Engine(..., strict_mode=True)` for this workflow.
|
|
150
|
+
|
|
151
|
+
## Policy guard
|
|
152
|
+
|
|
153
|
+
This executable example permits one configuration lookup for one application-supplied principal, records the actual lookup result, and verifies the ledger. Other keys or tools have no grant. It persists a local demonstration key across runs.
|
|
154
|
+
|
|
155
|
+
```python
|
|
156
|
+
from pathlib import Path
|
|
157
|
+
from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
|
|
158
|
+
from system1 import (
|
|
159
|
+
ActionLedger, ActionProposal, DecisionOutcome, PolicyEngine, PolicyRule,
|
|
160
|
+
RiskLevel, SystemOneGuard, load_private_key, save_keypair,
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
key_path = Path(".system1/demo-identity/identity.key")
|
|
164
|
+
if not key_path.exists():
|
|
165
|
+
save_keypair(Ed25519PrivateKey.generate(), key_path.parent)
|
|
166
|
+
signing_key = load_private_key(key_path)
|
|
167
|
+
|
|
168
|
+
policy = PolicyEngine(rules=[PolicyRule(
|
|
169
|
+
rule_id="read_service_name",
|
|
170
|
+
tools=["read_config"],
|
|
171
|
+
allowed_principals=["agent-worker"],
|
|
172
|
+
allowed_tenants=["demo"],
|
|
173
|
+
allowed_scopes=["config:read"],
|
|
174
|
+
argument_limits={"key": ["service_name"]},
|
|
175
|
+
outcome=DecisionOutcome.ALLOW,
|
|
176
|
+
risk=RiskLevel.READ_ONLY,
|
|
177
|
+
)])
|
|
178
|
+
|
|
179
|
+
with ActionLedger(".system1/demo-audit.sqlite", require_durable=True) as ledger:
|
|
180
|
+
guard = SystemOneGuard(
|
|
181
|
+
policy=policy, ledger=ledger, signing_key=signing_key,
|
|
182
|
+
enforcement_profile=True,
|
|
183
|
+
)
|
|
184
|
+
proposal = ActionProposal.create(
|
|
185
|
+
tenant_id="demo", principal_id="agent-worker", scope="config:read",
|
|
186
|
+
tool="read_config", arguments={"key": "service_name"},
|
|
187
|
+
canonical_target="config:service_name", purpose="Inspect service name",
|
|
188
|
+
)
|
|
189
|
+
auth = guard.evaluate_proposal(proposal)
|
|
190
|
+
if auth.outcome != DecisionOutcome.ALLOW:
|
|
191
|
+
raise PermissionError(auth.reason)
|
|
192
|
+
|
|
193
|
+
# Execute exactly the authorized operation and arguments.
|
|
194
|
+
result = {"service_name": "system1-demo"}[proposal.arguments["key"]]
|
|
195
|
+
ledger.record_execution_outcome(
|
|
196
|
+
action_id=proposal.action_id,
|
|
197
|
+
receipt_digest=auth.receipt.compute_digest(),
|
|
198
|
+
status="SUCCEEDED", result_payload={"value": result},
|
|
199
|
+
tenant_id=proposal.tenant_id, principal_id=proposal.principal_id,
|
|
200
|
+
scope=proposal.scope, trusted_public_key=signing_key.public_key(),
|
|
201
|
+
)
|
|
202
|
+
assert ledger.verify_integrity(trusted_public_key=signing_key.public_key())
|
|
203
|
+
print(result)
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
The application must derive identity from its authenticated session, constrain tool arguments and targets, and keep agent code from bypassing the guard. Policy correctness is the operator's responsibility. The default guard without `enforcement_profile=True` can use statistical classification to allow actions. See [deployment boundaries](docs/deployment.md) before granting consequential permissions.
|
|
207
|
+
|
|
208
|
+
## Integrations
|
|
209
|
+
|
|
210
|
+
### LangChain
|
|
211
|
+
|
|
212
|
+
Install `python -m pip install 'system1[langchain]'`. Attach a configured guard to actual tool invocations:
|
|
213
|
+
|
|
214
|
+
```python
|
|
215
|
+
from system1.integrations import SystemOneGuardCallbackHandler
|
|
216
|
+
|
|
217
|
+
# guard is your configured SystemOneGuard, with a live ledger and signing key.
|
|
218
|
+
handler = SystemOneGuardCallbackHandler(
|
|
219
|
+
guard=guard, tenant_id="demo", principal_id="agent-worker",
|
|
220
|
+
)
|
|
221
|
+
result = your_tool.invoke(tool_arguments, config={"callbacks": [handler]})
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
The callback uses scope `langchain:tools:exec` and the serialized tool input in its proposal. Match policy rules to that contract. `SystemOneGuardBlockedException` propagates to the caller on a deny or approval requirement. `wrap_langchain_tool` also supports guarding Python callables directly.
|
|
225
|
+
|
|
226
|
+
### MCP
|
|
227
|
+
|
|
228
|
+
`System1MCPProxy` intercepts JSON-RPC tool calls. Supply your configured guard and authenticated identity, then use `handle_call(request, executor)` or `async_handle_call(request, executor)` to connect it to your application's executor. See [integration tests](tests/test_integrations.py) for the dispatch contract. Diagnostic mode does not produce trusted signed enforcement evidence by default.
|
|
229
|
+
|
|
230
|
+
### FastAPI / ASGI
|
|
231
|
+
|
|
232
|
+
`add_system1_gateway(app, schema=YourSchema)` can respond to confident classification requests locally and pass uncertain requests to the downstream application. Install your ASGI framework separately. Configure authentication and request-size limits **outside** this middleware: local responses bypass downstream endpoint dependencies. This is a classification gateway, not a tool authorization boundary. See [deployment guidance](docs/deployment.md#asgi-gateway).
|
|
233
|
+
|
|
234
|
+
### gRPC
|
|
235
|
+
|
|
236
|
+
```bash
|
|
237
|
+
python -m pip install 'system1[grpc]'
|
|
238
|
+
system1 serve --grpc --host 127.0.0.1 --port 50051
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
The bundled [protobuf contract](src/system1/proto/system1.proto) supports clients generated for other languages. The CLI server is a local diagnostic service with insecure gRPC transport. Configure signing, ledger, and policy through the Python `serve(...)` API for a custom deployment. The repository does not supply a published Docker image or Kubernetes manifests.
|
|
242
|
+
|
|
243
|
+
### Observability
|
|
244
|
+
|
|
245
|
+
Install `system1[observability]` for Prometheus or `system1[otel]` for OpenTelemetry. See the [observability guide](docs/observability/README.md) and [Grafana dashboard](docs/observability/grafana-dashboard.json). Starting a metrics server or configuring an exporter changes the application's network behavior.
|
|
246
|
+
|
|
247
|
+
## Benchmarks
|
|
248
|
+
|
|
249
|
+
Reproduce the included seed-model benchmarks from the repository root:
|
|
250
|
+
|
|
251
|
+
```bash
|
|
252
|
+
python benchmarks/quality/run_quality_benchmarks.py
|
|
253
|
+
system1 bench --schema triage --iterations 200 --json
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
The 100-example datasets are small, repository-authored evaluations, not independent security certifications. The launch-review run measured:
|
|
257
|
+
|
|
258
|
+
| Task | Overall quality | Additional metric | Median latency |
|
|
259
|
+
|---|---|---|---|
|
|
260
|
+
| Security triage | 50% accuracy; 0.486 macro-F1 | BLOCK recall: 0.371 | 0.410 ms |
|
|
261
|
+
| Intent routing | 51% accuracy; 0.495 macro-F1 | Billing F1: 0.653 | 0.488 ms |
|
|
262
|
+
| Threat scoring, 0–10 | MAE: 3.068; RMSE: 3.480 | Pearson r: 0.309 | 0.469 ms |
|
|
263
|
+
|
|
264
|
+
See the [recorded results and environment](benchmarks/quality/results/launch_review.json) and [methodology](benchmarks/quality/README.md). These are raw classification results, not the accuracy of authorized tool actions. Latency is workload- and hardware-dependent; these measurements do not establish durable end-to-end authorization latency or a service-level guarantee. No cloud providers were measured in this run.
|
|
265
|
+
|
|
266
|
+
Teaching from the existing examples improves raw accuracy in a separate five-fold development check: intent routing reaches 68% and security triage 71% with 2048 features. Both still require review on every case under strict uncertainty gating; their calibration sets are too small. This is evidence that examples help, not release acceptance evidence. See the [teaching comparison](benchmarks/quality/README.md#teaching-comparison) for default-dimension results, protocol, and limitations.
|
|
267
|
+
|
|
268
|
+
Conformal coverage applies to prediction sets under exchangeability and appropriate held-out calibration. It does not guarantee that a singleton prediction is safe, control the error rate conditional on local acceptance, or imply a particular local-retention percentage. Distribution shift and model updates require reevaluation. See [limits](docs/deployment.md#statistical-limits) and the [conformal prediction introduction](https://arxiv.org/abs/2107.07511).
|
|
269
|
+
|
|
270
|
+
## Examples and research
|
|
271
|
+
|
|
272
|
+
- [Support triage](examples/support_triage.py), [model routing](examples/model_routing.py), and [agent guard](examples/agent_guard.py) teach and check one skill each. Their [datasets and measured results](examples/teaching/README.md) are included.
|
|
273
|
+
- [Teach one skill](examples/teach_skill.py), [advanced expert examples](examples/train_expert.py), and [observed teaching and takeover](examples/observe_routing.py). Cloud modes require explicit configuration.
|
|
274
|
+
- [Gaming examples](examples/gaming/) explore simulated environments and optional local emulation; they are not independently verified world records. ROMs are not included.
|
|
275
|
+
- [Research manuscripts](docs/paper/) describe the design and earlier experiments. Their historical timing and quality claims are not release acceptance criteria; use the reproducible measurements above.
|
|
276
|
+
|
|
277
|
+
Both `import system1` and the legacy `import reflex` expose the same API. The `reflex` namespace can conflict with the separate Reflex web-framework package, so use separate environments when needed. The [TypeSafe adapter](docs/typesafe.md) supports the basic sync/async decision API, structured state, typed response accessors, and ordinal score distributions. Its documented contract does not include the entire SDK transport/Pydantic surface.
|
|
278
|
+
|
|
279
|
+
## CLI
|
|
280
|
+
|
|
281
|
+
```bash
|
|
282
|
+
system1 decide "How do I reset my password?" --schema triage --json
|
|
283
|
+
system1 decide "Read documentation" --schema guard --sign --ledger audit.sqlite --json
|
|
284
|
+
system1 verify-receipt receipt.json --public-key identity.pub
|
|
285
|
+
system1 calibrate --dataset data.json --schema triage --bins 10
|
|
286
|
+
system1 compile --schema triage --output triage.s1m --json
|
|
287
|
+
```
|
|
288
|
+
|
|
289
|
+
Use `system1 --help` or `system1 <command> --help` for options. Receipt verification requires a trusted public key supplied independently of the receipt.
|
|
290
|
+
|
|
291
|
+
## Development
|
|
292
|
+
|
|
293
|
+
```bash
|
|
294
|
+
python -m pip install -e '.[dev,langchain]'
|
|
295
|
+
python -m pytest tests/ -q
|
|
296
|
+
```
|
|
297
|
+
|
|
298
|
+
Or use the committed lockfile with `uv sync --locked --extra dev --extra langchain` and `uv run --no-sync pytest -q`. CI tests Linux and macOS, checks built distributions, and exercises real LangChain dispatch. Optional MLX and emulator tests require their extras and suitable hardware/assets. See [contributing](CONTRIBUTING.md) and [security reporting](SECURITY.md).
|
|
299
|
+
|
|
300
|
+
## License
|
|
301
|
+
|
|
302
|
+
[Apache License 2.0](LICENSE).
|
system1-0.2.2/README.md
ADDED
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<img src="https://raw.githubusercontent.com/steph4n-gh/system1/main/assets/system1-logo.jpg" alt="System 1 logo" width="128" />
|
|
3
|
+
</p>
|
|
4
|
+
|
|
5
|
+
# System 1: Local Decision Runtime for AI Agents
|
|
6
|
+
|
|
7
|
+
Teach a repeatable decision skill from examples or by observing a teacher such as Jev. Validate it, take over locally, and save the skill for reuse.
|
|
8
|
+
|
|
9
|
+
[](https://github.com/steph4n-gh/system1/actions/workflows/ci.yml)
|
|
10
|
+
[](https://pypi.org/project/system1/)
|
|
11
|
+
[](pyproject.toml)
|
|
12
|
+
[](LICENSE)
|
|
13
|
+
|
|
14
|
+
**Status: beta.** System 1 provides a small local classifier and a separate deterministic policy guard. The included seed models need evaluation and calibration on your workload. They are not a general-purpose detector of malicious actions or prompt injection.
|
|
15
|
+
|
|
16
|
+
[Quickstart](#quickstart) · [Policy guard](#policy-guard) · [Integrations](#integrations) · [Benchmarks](#benchmarks) · [Limits and deployment](docs/deployment.md) · [Contributing](CONTRIBUTING.md)
|
|
17
|
+
|
|
18
|
+
## What it does
|
|
19
|
+
|
|
20
|
+
- **Observe → teach → validate → run locally:** keep the teacher answering until the observed skill passes held-out checks, then use the same call site locally. Save and reload the validated skill with its uncertainty behavior intact.
|
|
21
|
+
- **Structured classification:** choice, boolean, multi-choice, and score fields using local NumPy projections, with optional MLX acceleration.
|
|
22
|
+
- **Uncertainty handling:** calibration and conformal prediction sets for routing uncertain decisions to application-defined review or fallback paths.
|
|
23
|
+
- **Deterministic permissions:** `PolicyEngine` evaluates explicit rules; `SystemOneGuard(enforcement_profile=True)` requires a matching permission grant, signing key, and durable ledger before returning `ALLOW`.
|
|
24
|
+
- **Audit evidence:** Ed25519 software signatures (RFC 8032) on decision receipts and a SHA-256 hash-chained SQLite ledger. Applications remain responsible for authenticating callers and enforcing decisions at the tool boundary.
|
|
25
|
+
- **Local execution:** the core decision path requires no network or cloud API. Optional fallback clients, telemetry exporters, and application tools can use the network.
|
|
26
|
+
|
|
27
|
+
## Quickstart
|
|
28
|
+
|
|
29
|
+
The project is packaged on PyPI as `system1`. Use Python 3.11 or newer:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
python -m pip install system1
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
To try the code in this repository:
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
git clone https://github.com/steph4n-gh/system1.git
|
|
39
|
+
cd system1
|
|
40
|
+
python -m venv .venv
|
|
41
|
+
source .venv/bin/activate
|
|
42
|
+
python -m pip install -e '.[dev]'
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
Define the outputs you need, then evaluate a prompt:
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
from system1 import ChoiceField, DecisionSchema, System1Engine
|
|
49
|
+
|
|
50
|
+
class SupportRoute(DecisionSchema):
|
|
51
|
+
team = ChoiceField(
|
|
52
|
+
options=["billing", "technical_support"],
|
|
53
|
+
descriptions={
|
|
54
|
+
"billing": "Invoices, payments, subscription plans, and refunds",
|
|
55
|
+
"technical_support": "Software installation, errors, and debugging",
|
|
56
|
+
},
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
engine = System1Engine(SupportRoute)
|
|
60
|
+
decision = engine.decide("I need a refund for my subscription")
|
|
61
|
+
print(decision.values)
|
|
62
|
+
print(f"Latency: {decision.latency_ms:.2f} ms")
|
|
63
|
+
print(f"Needs review: {decision.is_ambiguous}")
|
|
64
|
+
print(f"Prediction sets: {decision.conformal_sets}")
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
This starts with schema-derived seed prototypes. Outputs and latency depend on the schema, data, and hardware. A predicted value is not permission to execute a tool.
|
|
68
|
+
|
|
69
|
+
## Observe a teacher, then take over
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
python examples/observe_routing.py
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
This short offline example shows the complete journey: observe answers, teach one routing skill, pass the normal promotion gates, disconnect the teacher, evaluate fresh requests locally, then save and reopen the skill. In the recorded run it needed **190 observations**, answered **40/40 fresh synthetic tickets correctly**, accepted all 40, and saved a **5 KB** skill. Reloading preserved answers, probabilities, and review decisions. Local median latency was about **0.45 ms** on the review machine.
|
|
76
|
+
|
|
77
|
+
The default teacher is a rule over a small structured-ticket vocabulary, clearly labeled as a simulation. It demonstrates the lifecycle, not general language understanding or Jev quality parity. `--teacher jev` observes actual Jev responses with your API key; another API can use the existing teacher callback. See the [adapter guide and compatibility contract](docs/typesafe.md) and [recorded evidence](examples/teaching/results/observed_routing.json).
|
|
78
|
+
|
|
79
|
+
Promotion depends on evidence, not a fixed turn count. Insufficient evidence keeps the teacher active, and uncertain local responses still request review.
|
|
80
|
+
|
|
81
|
+
## Teach a skill
|
|
82
|
+
|
|
83
|
+
Give System 1 labeled examples of one task, save the resulting `.s1m` skill, and reuse it locally. A person or System 2 can supply the examples; teaching does not require an LLM. The existing compiler fits a small decision head using NumPy.
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
python examples/support_triage.py
|
|
87
|
+
python examples/model_routing.py
|
|
88
|
+
python examples/agent_guard.py
|
|
89
|
+
system1 decide "Please correct the invoice address" --model .system1/examples/support_triage/skill.s1m --json
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Each primary example supplies labeled teaching cases, separate calibration cases, and 24 unseen evaluation cases. One command teaches, saves, reloads, and reports the results. On the review machine, teaching plus calibration took 127–159 ms, saved skills were 25–32 KiB, and uncached decisions took about 0.5 ms. Preparing good labeled examples takes additional work.
|
|
93
|
+
|
|
94
|
+
The authored demonstration cases reached 87.5% support-routing accuracy, 91.7% model-routing accuracy, and 100% operation-triage accuracy. Strict uncertainty checks requested review on 23/24 support cases, all 24 routing cases, and 22/24 operation cases. The three accepted responses were correct. See the [complete results, data, and limits](examples/teaching/README.md); these are demonstrations, not production quality guarantees.
|
|
95
|
+
|
|
96
|
+
For the smallest API example, see [teach one skill](examples/teach_skill.py). The [teaching guide](docs/guides/training_experts.md) covers your own data and evaluation. In Python, use `compile(examples, augment=False)` and `System1Engine(..., strict_mode=True)` for this workflow.
|
|
97
|
+
|
|
98
|
+
## Policy guard
|
|
99
|
+
|
|
100
|
+
This executable example permits one configuration lookup for one application-supplied principal, records the actual lookup result, and verifies the ledger. Other keys or tools have no grant. It persists a local demonstration key across runs.
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
from pathlib import Path
|
|
104
|
+
from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PrivateKey
|
|
105
|
+
from system1 import (
|
|
106
|
+
ActionLedger, ActionProposal, DecisionOutcome, PolicyEngine, PolicyRule,
|
|
107
|
+
RiskLevel, SystemOneGuard, load_private_key, save_keypair,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
key_path = Path(".system1/demo-identity/identity.key")
|
|
111
|
+
if not key_path.exists():
|
|
112
|
+
save_keypair(Ed25519PrivateKey.generate(), key_path.parent)
|
|
113
|
+
signing_key = load_private_key(key_path)
|
|
114
|
+
|
|
115
|
+
policy = PolicyEngine(rules=[PolicyRule(
|
|
116
|
+
rule_id="read_service_name",
|
|
117
|
+
tools=["read_config"],
|
|
118
|
+
allowed_principals=["agent-worker"],
|
|
119
|
+
allowed_tenants=["demo"],
|
|
120
|
+
allowed_scopes=["config:read"],
|
|
121
|
+
argument_limits={"key": ["service_name"]},
|
|
122
|
+
outcome=DecisionOutcome.ALLOW,
|
|
123
|
+
risk=RiskLevel.READ_ONLY,
|
|
124
|
+
)])
|
|
125
|
+
|
|
126
|
+
with ActionLedger(".system1/demo-audit.sqlite", require_durable=True) as ledger:
|
|
127
|
+
guard = SystemOneGuard(
|
|
128
|
+
policy=policy, ledger=ledger, signing_key=signing_key,
|
|
129
|
+
enforcement_profile=True,
|
|
130
|
+
)
|
|
131
|
+
proposal = ActionProposal.create(
|
|
132
|
+
tenant_id="demo", principal_id="agent-worker", scope="config:read",
|
|
133
|
+
tool="read_config", arguments={"key": "service_name"},
|
|
134
|
+
canonical_target="config:service_name", purpose="Inspect service name",
|
|
135
|
+
)
|
|
136
|
+
auth = guard.evaluate_proposal(proposal)
|
|
137
|
+
if auth.outcome != DecisionOutcome.ALLOW:
|
|
138
|
+
raise PermissionError(auth.reason)
|
|
139
|
+
|
|
140
|
+
# Execute exactly the authorized operation and arguments.
|
|
141
|
+
result = {"service_name": "system1-demo"}[proposal.arguments["key"]]
|
|
142
|
+
ledger.record_execution_outcome(
|
|
143
|
+
action_id=proposal.action_id,
|
|
144
|
+
receipt_digest=auth.receipt.compute_digest(),
|
|
145
|
+
status="SUCCEEDED", result_payload={"value": result},
|
|
146
|
+
tenant_id=proposal.tenant_id, principal_id=proposal.principal_id,
|
|
147
|
+
scope=proposal.scope, trusted_public_key=signing_key.public_key(),
|
|
148
|
+
)
|
|
149
|
+
assert ledger.verify_integrity(trusted_public_key=signing_key.public_key())
|
|
150
|
+
print(result)
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
The application must derive identity from its authenticated session, constrain tool arguments and targets, and keep agent code from bypassing the guard. Policy correctness is the operator's responsibility. The default guard without `enforcement_profile=True` can use statistical classification to allow actions. See [deployment boundaries](docs/deployment.md) before granting consequential permissions.
|
|
154
|
+
|
|
155
|
+
## Integrations
|
|
156
|
+
|
|
157
|
+
### LangChain
|
|
158
|
+
|
|
159
|
+
Install `python -m pip install 'system1[langchain]'`. Attach a configured guard to actual tool invocations:
|
|
160
|
+
|
|
161
|
+
```python
|
|
162
|
+
from system1.integrations import SystemOneGuardCallbackHandler
|
|
163
|
+
|
|
164
|
+
# guard is your configured SystemOneGuard, with a live ledger and signing key.
|
|
165
|
+
handler = SystemOneGuardCallbackHandler(
|
|
166
|
+
guard=guard, tenant_id="demo", principal_id="agent-worker",
|
|
167
|
+
)
|
|
168
|
+
result = your_tool.invoke(tool_arguments, config={"callbacks": [handler]})
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
The callback uses scope `langchain:tools:exec` and the serialized tool input in its proposal. Match policy rules to that contract. `SystemOneGuardBlockedException` propagates to the caller on a deny or approval requirement. `wrap_langchain_tool` also supports guarding Python callables directly.
|
|
172
|
+
|
|
173
|
+
### MCP
|
|
174
|
+
|
|
175
|
+
`System1MCPProxy` intercepts JSON-RPC tool calls. Supply your configured guard and authenticated identity, then use `handle_call(request, executor)` or `async_handle_call(request, executor)` to connect it to your application's executor. See [integration tests](tests/test_integrations.py) for the dispatch contract. Diagnostic mode does not produce trusted signed enforcement evidence by default.
|
|
176
|
+
|
|
177
|
+
### FastAPI / ASGI
|
|
178
|
+
|
|
179
|
+
`add_system1_gateway(app, schema=YourSchema)` can respond to confident classification requests locally and pass uncertain requests to the downstream application. Install your ASGI framework separately. Configure authentication and request-size limits **outside** this middleware: local responses bypass downstream endpoint dependencies. This is a classification gateway, not a tool authorization boundary. See [deployment guidance](docs/deployment.md#asgi-gateway).
|
|
180
|
+
|
|
181
|
+
### gRPC
|
|
182
|
+
|
|
183
|
+
```bash
|
|
184
|
+
python -m pip install 'system1[grpc]'
|
|
185
|
+
system1 serve --grpc --host 127.0.0.1 --port 50051
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
The bundled [protobuf contract](src/system1/proto/system1.proto) supports clients generated for other languages. The CLI server is a local diagnostic service with insecure gRPC transport. Configure signing, ledger, and policy through the Python `serve(...)` API for a custom deployment. The repository does not supply a published Docker image or Kubernetes manifests.
|
|
189
|
+
|
|
190
|
+
### Observability
|
|
191
|
+
|
|
192
|
+
Install `system1[observability]` for Prometheus or `system1[otel]` for OpenTelemetry. See the [observability guide](docs/observability/README.md) and [Grafana dashboard](docs/observability/grafana-dashboard.json). Starting a metrics server or configuring an exporter changes the application's network behavior.
|
|
193
|
+
|
|
194
|
+
## Benchmarks
|
|
195
|
+
|
|
196
|
+
Reproduce the included seed-model benchmarks from the repository root:
|
|
197
|
+
|
|
198
|
+
```bash
|
|
199
|
+
python benchmarks/quality/run_quality_benchmarks.py
|
|
200
|
+
system1 bench --schema triage --iterations 200 --json
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
The 100-example datasets are small, repository-authored evaluations, not independent security certifications. The launch-review run measured:
|
|
204
|
+
|
|
205
|
+
| Task | Overall quality | Additional metric | Median latency |
|
|
206
|
+
|---|---|---|---|
|
|
207
|
+
| Security triage | 50% accuracy; 0.486 macro-F1 | BLOCK recall: 0.371 | 0.410 ms |
|
|
208
|
+
| Intent routing | 51% accuracy; 0.495 macro-F1 | Billing F1: 0.653 | 0.488 ms |
|
|
209
|
+
| Threat scoring, 0–10 | MAE: 3.068; RMSE: 3.480 | Pearson r: 0.309 | 0.469 ms |
|
|
210
|
+
|
|
211
|
+
See the [recorded results and environment](benchmarks/quality/results/launch_review.json) and [methodology](benchmarks/quality/README.md). These are raw classification results, not the accuracy of authorized tool actions. Latency is workload- and hardware-dependent; these measurements do not establish durable end-to-end authorization latency or a service-level guarantee. No cloud providers were measured in this run.
|
|
212
|
+
|
|
213
|
+
Teaching from the existing examples improves raw accuracy in a separate five-fold development check: intent routing reaches 68% and security triage 71% with 2048 features. Both still require review on every case under strict uncertainty gating; their calibration sets are too small. This is evidence that examples help, not release acceptance evidence. See the [teaching comparison](benchmarks/quality/README.md#teaching-comparison) for default-dimension results, protocol, and limitations.
|
|
214
|
+
|
|
215
|
+
Conformal coverage applies to prediction sets under exchangeability and appropriate held-out calibration. It does not guarantee that a singleton prediction is safe, control the error rate conditional on local acceptance, or imply a particular local-retention percentage. Distribution shift and model updates require reevaluation. See [limits](docs/deployment.md#statistical-limits) and the [conformal prediction introduction](https://arxiv.org/abs/2107.07511).
|
|
216
|
+
|
|
217
|
+
## Examples and research
|
|
218
|
+
|
|
219
|
+
- [Support triage](examples/support_triage.py), [model routing](examples/model_routing.py), and [agent guard](examples/agent_guard.py) teach and check one skill each. Their [datasets and measured results](examples/teaching/README.md) are included.
|
|
220
|
+
- [Teach one skill](examples/teach_skill.py), [advanced expert examples](examples/train_expert.py), and [observed teaching and takeover](examples/observe_routing.py). Cloud modes require explicit configuration.
|
|
221
|
+
- [Gaming examples](examples/gaming/) explore simulated environments and optional local emulation; they are not independently verified world records. ROMs are not included.
|
|
222
|
+
- [Research manuscripts](docs/paper/) describe the design and earlier experiments. Their historical timing and quality claims are not release acceptance criteria; use the reproducible measurements above.
|
|
223
|
+
|
|
224
|
+
Both `import system1` and the legacy `import reflex` expose the same API. The `reflex` namespace can conflict with the separate Reflex web-framework package, so use separate environments when needed. The [TypeSafe adapter](docs/typesafe.md) supports the basic sync/async decision API, structured state, typed response accessors, and ordinal score distributions. Its documented contract does not include the entire SDK transport/Pydantic surface.
|
|
225
|
+
|
|
226
|
+
## CLI
|
|
227
|
+
|
|
228
|
+
```bash
|
|
229
|
+
system1 decide "How do I reset my password?" --schema triage --json
|
|
230
|
+
system1 decide "Read documentation" --schema guard --sign --ledger audit.sqlite --json
|
|
231
|
+
system1 verify-receipt receipt.json --public-key identity.pub
|
|
232
|
+
system1 calibrate --dataset data.json --schema triage --bins 10
|
|
233
|
+
system1 compile --schema triage --output triage.s1m --json
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
Use `system1 --help` or `system1 <command> --help` for options. Receipt verification requires a trusted public key supplied independently of the receipt.
|
|
237
|
+
|
|
238
|
+
## Development
|
|
239
|
+
|
|
240
|
+
```bash
|
|
241
|
+
python -m pip install -e '.[dev,langchain]'
|
|
242
|
+
python -m pytest tests/ -q
|
|
243
|
+
```
|
|
244
|
+
|
|
245
|
+
Or use the committed lockfile with `uv sync --locked --extra dev --extra langchain` and `uv run --no-sync pytest -q`. CI tests Linux and macOS, checks built distributions, and exercises real LangChain dispatch. Optional MLX and emulator tests require their extras and suitable hardware/assets. See [contributing](CONTRIBUTING.md) and [security reporting](SECURITY.md).
|
|
246
|
+
|
|
247
|
+
## License
|
|
248
|
+
|
|
249
|
+
[Apache License 2.0](LICENSE).
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Security policy
|
|
2
|
+
|
|
3
|
+
## Reporting a vulnerability
|
|
4
|
+
|
|
5
|
+
Report vulnerabilities privately through [GitHub's private vulnerability reporting form](https://github.com/steph4n-gh/system1/security/advisories/new). If private reporting is unavailable, open an issue requesting a private contact without posting exploit details, credentials, or sensitive data.
|
|
6
|
+
|
|
7
|
+
Include the affected version, configuration, expected boundary, reproduction steps using harmless sentinel tools, and observed result. Please avoid testing against services or data you do not own.
|
|
8
|
+
|
|
9
|
+
## Supported scope
|
|
10
|
+
|
|
11
|
+
System 1 is in beta. Fixes target the latest release; older versions are not maintained as separate branches. Report failures in deterministic policy enforcement, signed receipt verification, ledger integrity, model loading, or integration dispatch.
|
|
12
|
+
|
|
13
|
+
The statistical classifier is not a general-purpose security detector. The application supplies authenticated identity, isolates tools, protects policy and signing keys, and enforces returned decisions. Read [deployment boundaries](docs/deployment.md) before using the library for consequential actions.
|
|
Binary file
|