pen-stack 4.5.0__tar.gz → 5.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pen_stack-4.5.0 → pen_stack-5.0.0}/CHANGELOG.md +36 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/CITATION.cff +1 -1
- {pen_stack-4.5.0 → pen_stack-5.0.0}/PKG-INFO +22 -5
- {pen_stack-4.5.0 → pen_stack-5.0.0}/README.md +21 -4
- {pen_stack-4.5.0 → pen_stack-5.0.0}/benchmarks/genome_writing_bench/LEADERBOARD.md +6 -5
- {pen_stack-4.5.0 → pen_stack-5.0.0}/benchmarks/genome_writing_bench/tasks.yaml +18 -1
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/cell_types.yaml +2 -2
- pen_stack-5.0.0/docs/co_scientist.md +31 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/__init__.py +1 -1
- pen_stack-5.0.0/pen_stack/agent/cite.py +118 -0
- pen_stack-5.0.0/pen_stack/agent/co_scientist.py +232 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/graph/build.py +2 -2
- pen_stack-5.0.0/pen_stack/validate/bench_coscientist_tasks.py +60 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack.egg-info/PKG-INFO +22 -5
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack.egg-info/SOURCES.txt +10 -0
- pen_stack-5.0.0/prereg/SHA256_LOCK_ws_cite.json +8 -0
- pen_stack-5.0.0/prereg/SHA256_LOCK_ws_crit.json +8 -0
- pen_stack-5.0.0/prereg/SHA256_LOCK_ws_plan.json +8 -0
- pen_stack-5.0.0/prereg/ws_cite.yaml +17 -0
- pen_stack-5.0.0/prereg/ws_crit.yaml +16 -0
- pen_stack-5.0.0/prereg/ws_plan.yaml +18 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pyproject.toml +1 -1
- {pen_stack-4.5.0 → pen_stack-5.0.0}/LICENSE +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/MANIFEST.in +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/bench/run.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/benchmarks/genome_writing_bench/README.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/benchmarks/genome_writing_bench/SHA256SUMS +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/benchmarks/genome_writing_bench/SUBMISSIONS.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/atlas_families.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/bridge_offtarget_profile.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/cargo_polish.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/datasets.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/delivery_constraints.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/delivery_rules.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/delivery_vehicles.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/gates_v3.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/gsh_validated_heldout.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/intent_weights.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/known_unknowns.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/llm.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/monitor_queries.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/oracles/scope_cards.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/rules/delivery.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/rules/fold.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/rules/multiplex.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/rules/payload.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/rules/reachability.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/score_axes.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/target_sites.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/universe_crosswalk.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/write_types.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/configs/wtkb_curated.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/data/curated/bridge_offtarget_energetics.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/data/curated/bridge_offtarget_profile_measured.parquet +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/data/curated/gene_coords.parquet +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/data/curated/unified_editor_universe.parquet +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/BACKLOG.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/DEPLOY.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/INFRA.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/MCP.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/RELEASING.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/REPRO.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/agent.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/alphagenome_feasibility.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/benchmark_circularity.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/cards/atlas.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/cards/durability.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/cards/safety.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/delivery.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/dissemination.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/environment.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/index.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/mechanistic_constraints.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/oracles.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/positioning.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/private_data_formats.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/quickstart.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/rules.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/scope.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/scorecard.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/tutorials/compare-families.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/tutorials/score-deliverability.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/tutorials/where-can-i-write.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/tutorials/which-writer-reaches-locus.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/uncertainty.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/verify.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/world_model.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/writer_verification.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/docs/wtkb.md +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/_resources.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/adapt/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/adapt/finetune.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/adapt/ingest.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/adapt/pipeline.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/adapt/recalibrate.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/adapt/report.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/agent/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/agent/epistemic.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/agent/guardrails.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/agent/mcp_server.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/agent/orchestrator.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/agent/pen_agent.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/agent/scope.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/agent/tools.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/atlas/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/atlas/build_wtkb.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/atlas/crosslink.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/atlas/expand.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/atlas/schema.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/atlas/scorecard.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/atlas/universe.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/atlas/variant_propose.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/atlas/writer_verify.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/activity.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/cli.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/fold_qc.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/guide_qc.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/ingest.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/offtarget.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/offtarget_energetics.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/ortholog_screen.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/bridge/pipeline.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/cli.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/data/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/data/encode.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/data/genome.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/data/ingest_chromatin.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/data/ingest_integration.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/data/ingest_safety_annot.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/data/ingest_trip.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/env/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/env/genome_writing_env.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/env/policies.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/graph/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/graph/cell_types.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/graph/ingest.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/graph/query.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/graph/schema.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/mech/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/mech/classify_atlas.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/mech/whitelist.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/monitor/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/monitor/europepmc.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/monitor/run.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/monitor/triage.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/oracles/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/oracles/cache.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/oracles/energetics.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/oracles/genome.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/oracles/protein_design.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/oracles/rna.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/oracles/schema.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/oracles/structure.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/cargo.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/cargo_polish.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/delivery.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/delivery_constraints.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/delivery_vehicles.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/multiplex.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/optimize.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/pipeline.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/report.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/router.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/planner/target_site.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/rag/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/rag/index.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/rag/llm.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/rag/qa.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/rules/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/rules/evaluators.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/rules/loader.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/rules/schema.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/rules/solver.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/score/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/score/recalibrate.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/score/therapeutic.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/server/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/server/api.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/ui/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/ui/app.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/adapt_demo.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/agent_eval.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/bench_adversarial_tasks.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/bench_graph_tasks.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/bench_rule_tasks.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/bench_trust_tasks.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/bench_writetype_tasks.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/blind_gsh_discovery.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/cargo_directionality.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/durability_baselines.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/forward_hypotheses.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/guide_qc_demo.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/intent_specification.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/offtarget_energetics_eval.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/out_of_scope_refusal.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/outcome_calibration.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/paper3_benchmark.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/paper4_real_validation.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/paper4_validation.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/selective_prediction.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/seq_vs_measured.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/target_site_controls.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/uncertainty_eval.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/ungrounded_baseline.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/within_locus_ranking.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/validate/writer_recovery.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/verify/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/verify/schema.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/verify/service.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/__init__.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/chromatin_seq.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/durability.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/export_tracks.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/features.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/gsh_baseline.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/mesh_features.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/ood.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/providers.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/safety.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/structure3d.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/uncertainty.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack/wgenome/writability.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack.egg-info/dependency_links.txt +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack.egg-info/entry_points.txt +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack.egg-info/requires.txt +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/pen_stack.egg-info/top_level.txt +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_phase0.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_phase1_5.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_phase2.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_phase3.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_a.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_atlas.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_b.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_ba.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_ba_v33.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_ba_v45.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_bench.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_c.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_cal.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_ct.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_d.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_e.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_env.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_ep.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_f.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_g.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_graph.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_h.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_mc.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_mon.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_o.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_r.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_route.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_uq.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_v.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/SHA256_LOCK_ws_wv.json +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/paper1.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/paper2.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/paper3.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/paper4.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/phase0.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_a.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_atlas.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_b.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_ba.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_ba_v33.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_ba_v45.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_bench.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_c.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_cal.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_ct.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_d.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_e.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_env.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_ep.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_f.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_g.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_graph.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_h.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_mc.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_mon.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_o.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_r.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_route.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_uq.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_v.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/prereg/ws_wv.yaml +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/p1_build_atlas.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/p1_build_durability.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/p1_export_tracks.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/p1_safety_concordance.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/p1_train_safety.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/p1_validation_report.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/p2_build_atlas.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/p3_benchmark_report.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/p4_genome_scan.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/ws_b_report.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/scripts/ws_c_report.py +0 -0
- {pen_stack-4.5.0 → pen_stack-5.0.0}/setup.cfg +0 -0
|
@@ -3,6 +3,42 @@
|
|
|
3
3
|
All notable changes to PEN-STACK are documented here. This file follows
|
|
4
4
|
[Keep a Changelog](https://keepachangelog.com/) and the program's phase structure.
|
|
5
5
|
|
|
6
|
+
## [5.0.0] - 2026-06-09 - v5.0 release: the Co-Scientist (capstone — smart because it is grounded)
|
|
7
|
+
|
|
8
|
+
The reasoning ceiling rises while the grounding floor stays fixed: a co-scientist that proposes multiple
|
|
9
|
+
distinct strategies, critiques and revises its own plans, cites its reasoning, and itemises what it cannot
|
|
10
|
+
assess — with **no-fabrication holding across the full reasoning stack** (the central gate). Workstreams
|
|
11
|
+
WS-{PLAN,MULTI,CRIT,SCOPE2,CITE,GEN}, each SHA-locked.
|
|
12
|
+
|
|
13
|
+
### Added
|
|
14
|
+
- **WS-PLAN + WS-MULTI** — `pen_stack/agent/co_scientist.py`: `propose_strategies()` returns 2–3 **materially
|
|
15
|
+
distinct** strategies (≥2 design axes differ — measured by `distinctness()`, not reworded), each
|
|
16
|
+
independently legal + confidence-tagged; `deliberate()` benchmarks the deliberative planner vs the
|
|
17
|
+
deterministic baseline. `prereg/ws_plan.yaml`.
|
|
18
|
+
- **WS-CRIT + WS-SCOPE2** — `critique()` / `critique_and_revise()` (the critic only flags + swaps a design
|
|
19
|
+
choice, never invents a number; revisions re-verified) + `critique_falsifiability()` (improves flawed plans
|
|
20
|
+
illegal→legal, 0 spurious revisions on clean) + `scope_ledger()` (per-recommendation: what was/ wasn't
|
|
21
|
+
assessed, the known-unknowns itemised). `prereg/ws_crit.yaml`.
|
|
22
|
+
- **WS-CITE + WS-GEN** — `pen_stack/agent/cite.py`: `cited_rationale()` (citations drawn from the curated
|
|
23
|
+
world-model → resolve by construction) + `citations_grounded()` guard (rejects any DOI not in the curated
|
|
24
|
+
set) + `generalise()` (adjacent tasks grounded-or-refused). `prereg/ws_cite.yaml`.
|
|
25
|
+
- **Bench v0.3.2** — `co_scientist_grounded` reference-solver task: grounded rate 1.0 vs ungrounded 0.0;
|
|
26
|
+
no-fabrication across the full stack. `docs/co_scientist.md`.
|
|
27
|
+
|
|
28
|
+
### Changed
|
|
29
|
+
- Version 4.5.1 -> 5.0.0 (major — the substrate matured into a grounded co-scientist); bench 0.3.1 -> 0.3.2.
|
|
30
|
+
|
|
31
|
+
## [4.5.1] - 2026-06-09 - ID-correctness patch: cell-type ontology IDs
|
|
32
|
+
|
|
33
|
+
### Fixed
|
|
34
|
+
Two of the three new v4.5 Tier-A cell-type ontology IDs in `configs/cell_types.yaml` were wrong (verified via
|
|
35
|
+
EBI-OLS): `EFO:0002322` resolved to the **RPMI8226 myeloma line** (not a T cell) and `EFO:0004146` to an
|
|
36
|
+
**obsolete myopathy term** (not hepatocyte). Corrected to the canonical, non-obsolete Cell Ontology terms:
|
|
37
|
+
**primary_T_cell → `CL:0000084`** (T cell), **hepatocyte → `CL:0000182`** (hepatocyte). iPSC (`EFO:0004905`),
|
|
38
|
+
K562 (`EFO:0002067`), HepG2 (`EFO:0001187`) verified correct, as was the ISPpu10 back-test record (Europe PMC
|
|
39
|
+
**PPR1218813** — "ISPpu10 is a structure-gated bridge RNA recombinase…"). No result/test change (the IDs are
|
|
40
|
+
coverage-card metadata; `cell_types.py` reads coverage, not the ontology id).
|
|
41
|
+
|
|
6
42
|
## [4.5.0] - 2026-06-09 - v4.5 release: the Living World-Model (knowledge graph + gated living loop)
|
|
7
43
|
|
|
8
44
|
v4.5 promotes the flat tables into a queryable knowledge graph that keeps itself current. Workstreams
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pen-stack
|
|
3
|
-
Version:
|
|
3
|
+
Version: 5.0.0
|
|
4
4
|
Summary: Open infrastructure for genome writing: the Writable Genome atlas, the Writer Atlas, and the Write Planner.
|
|
5
5
|
Author-email: Anees Ahmed Mahaboob Ali <ahmedaneesm@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -89,12 +89,12 @@ and durably write new DNA, **which enzyme** can write it there, and **how** to d
|
|
|
89
89
|
[](https://codecov.io/gh/ahmedanees-m/pen-stack)
|
|
90
90
|
[](LICENSE)
|
|
91
91
|
[](https://www.python.org/)
|
|
92
|
-
[](CHANGELOG.md)
|
|
93
|
+
[](tests/)
|
|
94
94
|
[](https://github.com/astral-sh/ruff)
|
|
95
95
|
[](docker/)
|
|
96
96
|
[](prereg/)
|
|
97
|
-
[](benchmarks/genome_writing_bench/)
|
|
98
98
|
|
|
99
99
|
**Built on five prior, separately published repositories:**
|
|
100
100
|
|
|
@@ -133,6 +133,23 @@ Two questions gate every genome-writing project, and before PEN-STACK no resourc
|
|
|
133
133
|
Everything is built on bulk-downloadable public data, runs on a single GPU, and is validated **blind** against
|
|
134
134
|
a pre-registered, honest baseline before release.
|
|
135
135
|
|
|
136
|
+
## What is new in v5.0 — the Co-Scientist (smart because it is grounded)
|
|
137
|
+
|
|
138
|
+
v5.0 matures the reasoning layer on top of everything beneath it. Given a goal and an intent, PEN-STACK
|
|
139
|
+
returns a small set of **materially distinct, ranked, fully-traceable strategies** — each verified,
|
|
140
|
+
calibrated, cited, and scope-ledgered — while the **no-fabrication guarantee holds by construction**: the
|
|
141
|
+
reasoning layer proposes and critiques, but every number still comes from a validated tool or oracle.
|
|
142
|
+
Intelligence rises while groundedness never falls.
|
|
143
|
+
|
|
144
|
+
| Workstream | What it adds | Result |
|
|
145
|
+
|---|---|---|
|
|
146
|
+
| **PLAN + MULTI** | `agent/co_scientist.py` — `propose_strategies()` / `deliberate()` | 2–3 **materially-distinct** strategies (≥2 design axes differ — *measured*, not reworded), each independently **legal** + **confidence-tagged**; deliberative planner benchmarked vs the deterministic baseline |
|
|
147
|
+
| **CRIT + SCOPE2** | self-critique/revise loop + scope ledger | the critic only flags + swaps (never invents a number); revisions are **re-verified** and **falsifiable** (improve flawed plans illegal→legal, never touch clean ones); every recommendation carries a **complete scope ledger** itemising the known-unknowns |
|
|
148
|
+
| **CITE + GEN** | `agent/cite.py` — cited rationale + scoped generalisation | citations are **drawn from the curated world-model** (resolve by construction); a guard **rejects any hallucinated DOI**; adjacent tasks are **grounded-or-refused** |
|
|
149
|
+
| **central gate** | `co_scientist_grounded` bench (v0.3.2) | grounded rate **1.0** vs ungrounded **0.0**; **no-fabrication holds across the full reasoning stack** (asserted) |
|
|
150
|
+
|
|
151
|
+
See `docs/co_scientist.md` and `prereg/ws_{plan,crit,cite}.yaml`.
|
|
152
|
+
|
|
136
153
|
## What is new in v4.5 — the Living World-Model (a knowledge graph that keeps itself current)
|
|
137
154
|
|
|
138
155
|
v4.5 promotes the flat atlas/WT-KB/crosslink tables into a queryable **knowledge graph**: writers, loci,
|
|
@@ -412,7 +429,7 @@ pen-stack/
|
|
|
412
429
|
│ │ + v3.3 router (write-type dispatch) / delivery_vehicles (8-vehicle palette)
|
|
413
430
|
│ ├── bridge/ bridge off-target engine (Paper 4): offtarget / fold_qc / guide_qc / pipeline / cli
|
|
414
431
|
│ │ + v3.2 offtarget_energetics (position x substitution; held-out 0.88, ships)
|
|
415
|
-
│ ├── agent/ agentic platform: tools / orchestrator / pen_agent / mcp_server / guardrails
|
|
432
|
+
│ ├── agent/ agentic platform: tools / orchestrator / pen_agent / mcp_server / guardrails; v5.0 co_scientist + cite (multi-strategy, self-critique, cited rationale, scope ledger)
|
|
416
433
|
│ │ + v3.2 epistemic (3-tier status) / scope (known-unknowns matcher)
|
|
417
434
|
│ ├── graph/ v4.5 living world-model knowledge graph (schema/build/query/ingest/cell_types); typed provenanced edges; gated living loop (propose-only)
|
|
418
435
|
│ ├── oracles/ v4.0 L1 oracle mesh: OracleResult contract + adapters (genome/structure/protein_design/rna/energetics) over the foundation models; version-pinned cache
|
|
@@ -14,12 +14,12 @@ and durably write new DNA, **which enzyme** can write it there, and **how** to d
|
|
|
14
14
|
[](https://codecov.io/gh/ahmedanees-m/pen-stack)
|
|
15
15
|
[](LICENSE)
|
|
16
16
|
[](https://www.python.org/)
|
|
17
|
-
[](CHANGELOG.md)
|
|
18
|
+
[](tests/)
|
|
19
19
|
[](https://github.com/astral-sh/ruff)
|
|
20
20
|
[](docker/)
|
|
21
21
|
[](prereg/)
|
|
22
|
-
[](benchmarks/genome_writing_bench/)
|
|
23
23
|
|
|
24
24
|
**Built on five prior, separately published repositories:**
|
|
25
25
|
|
|
@@ -58,6 +58,23 @@ Two questions gate every genome-writing project, and before PEN-STACK no resourc
|
|
|
58
58
|
Everything is built on bulk-downloadable public data, runs on a single GPU, and is validated **blind** against
|
|
59
59
|
a pre-registered, honest baseline before release.
|
|
60
60
|
|
|
61
|
+
## What is new in v5.0 — the Co-Scientist (smart because it is grounded)
|
|
62
|
+
|
|
63
|
+
v5.0 matures the reasoning layer on top of everything beneath it. Given a goal and an intent, PEN-STACK
|
|
64
|
+
returns a small set of **materially distinct, ranked, fully-traceable strategies** — each verified,
|
|
65
|
+
calibrated, cited, and scope-ledgered — while the **no-fabrication guarantee holds by construction**: the
|
|
66
|
+
reasoning layer proposes and critiques, but every number still comes from a validated tool or oracle.
|
|
67
|
+
Intelligence rises while groundedness never falls.
|
|
68
|
+
|
|
69
|
+
| Workstream | What it adds | Result |
|
|
70
|
+
|---|---|---|
|
|
71
|
+
| **PLAN + MULTI** | `agent/co_scientist.py` — `propose_strategies()` / `deliberate()` | 2–3 **materially-distinct** strategies (≥2 design axes differ — *measured*, not reworded), each independently **legal** + **confidence-tagged**; deliberative planner benchmarked vs the deterministic baseline |
|
|
72
|
+
| **CRIT + SCOPE2** | self-critique/revise loop + scope ledger | the critic only flags + swaps (never invents a number); revisions are **re-verified** and **falsifiable** (improve flawed plans illegal→legal, never touch clean ones); every recommendation carries a **complete scope ledger** itemising the known-unknowns |
|
|
73
|
+
| **CITE + GEN** | `agent/cite.py` — cited rationale + scoped generalisation | citations are **drawn from the curated world-model** (resolve by construction); a guard **rejects any hallucinated DOI**; adjacent tasks are **grounded-or-refused** |
|
|
74
|
+
| **central gate** | `co_scientist_grounded` bench (v0.3.2) | grounded rate **1.0** vs ungrounded **0.0**; **no-fabrication holds across the full reasoning stack** (asserted) |
|
|
75
|
+
|
|
76
|
+
See `docs/co_scientist.md` and `prereg/ws_{plan,crit,cite}.yaml`.
|
|
77
|
+
|
|
61
78
|
## What is new in v4.5 — the Living World-Model (a knowledge graph that keeps itself current)
|
|
62
79
|
|
|
63
80
|
v4.5 promotes the flat atlas/WT-KB/crosslink tables into a queryable **knowledge graph**: writers, loci,
|
|
@@ -337,7 +354,7 @@ pen-stack/
|
|
|
337
354
|
│ │ + v3.3 router (write-type dispatch) / delivery_vehicles (8-vehicle palette)
|
|
338
355
|
│ ├── bridge/ bridge off-target engine (Paper 4): offtarget / fold_qc / guide_qc / pipeline / cli
|
|
339
356
|
│ │ + v3.2 offtarget_energetics (position x substitution; held-out 0.88, ships)
|
|
340
|
-
│ ├── agent/ agentic platform: tools / orchestrator / pen_agent / mcp_server / guardrails
|
|
357
|
+
│ ├── agent/ agentic platform: tools / orchestrator / pen_agent / mcp_server / guardrails; v5.0 co_scientist + cite (multi-strategy, self-critique, cited rationale, scope ledger)
|
|
341
358
|
│ │ + v3.2 epistemic (3-tier status) / scope (known-unknowns matcher)
|
|
342
359
|
│ ├── graph/ v4.5 living world-model knowledge graph (schema/build/query/ingest/cell_types); typed provenanced edges; gated living loop (propose-only)
|
|
343
360
|
│ ├── oracles/ v4.0 L1 oracle mesh: OracleResult contract + adapters (genome/structure/protein_design/rna/energetics) over the foundation models; version-pinned cache
|
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
# Genome-Writing Bench v0.3.
|
|
1
|
+
# Genome-Writing Bench v0.3.2 - Leaderboard
|
|
2
2
|
|
|
3
|
-
Tasks: **
|
|
4
|
-
Deterministic planner beats the naive baseline on **
|
|
3
|
+
Tasks: **16/16 available** in this run (unavailable = needs the Phase-1 atlas / Perry tables / an LLM, which run on the VM/local).
|
|
4
|
+
Deterministic planner beats the naive baseline on **12/12** grounded tasks with a baseline.
|
|
5
5
|
|
|
6
6
|
| Solver | Tasks scored | Beats naive | No-fabrication | Note |
|
|
7
7
|
|---|---|---|---|---|
|
|
8
|
-
| deterministic_planner |
|
|
9
|
-
| naive_baseline |
|
|
8
|
+
| deterministic_planner | 16 | 12/12 | n/a (deterministic) | validated planning tools - the reference |
|
|
9
|
+
| naive_baseline | 12 | - | n/a (deterministic) | safety-only / prevalence / Hamming baselines |
|
|
10
10
|
|
|
11
11
|
## Per-task results
|
|
12
12
|
| Task | Family | Available | Planner | Naive baseline | Gate |
|
|
@@ -26,6 +26,7 @@ Deterministic planner beats the naive baseline on **11/11** grounded tasks with
|
|
|
26
26
|
| multi_write_type_legality | MW_multi_write_type | True | 1.0 | 0.0 | - |
|
|
27
27
|
| adversarial_robustness | T13_scope_disguise | True | 1.0 | 0.0 | - |
|
|
28
28
|
| graph_multihop_reasoning | GR_graph_reasoning | True | 1.0 | 0.0 | - |
|
|
29
|
+
| co_scientist_grounded | CS_co_scientist | True | 1.0 | 0.0 | - |
|
|
29
30
|
|
|
30
31
|
## Trust tasks (T8-T11) - calibration + scope-awareness separate *trustworthy* agents
|
|
31
32
|
Each contrasts the **uncertainty-aware** agent (conformal coverage, selective prediction, OOD flagging, out-of-scope deferral) with an **over-confident** baseline (an uncalibrated interval, no abstention, never flags OOD, no scope layer). The over-confident agent is the realistic failure mode a calibrated co-scientist must beat.
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
# A task names a `scorer` (module.function in pen_stack.validate / pen_stack.bridge) and a `metric` key to
|
|
9
9
|
# read from its report. Solvers (deterministic planner, naive baseline, LLM agent) are compared on the same
|
|
10
10
|
# tasks; a solver that cannot ground a number must refuse, not invent (no-fabrication is a hard gate).
|
|
11
|
-
version: "0.3.
|
|
11
|
+
version: "0.3.2"
|
|
12
12
|
prepared: "2026-06-09"
|
|
13
13
|
|
|
14
14
|
taxonomy:
|
|
@@ -35,6 +35,8 @@ taxonomy:
|
|
|
35
35
|
T16_distribution_shift: "an OOD context -> confidence is deflated (extrapolating), not reported at the in-distribution level"
|
|
36
36
|
# v0.3.1 (v4.5): multi-hop reasoning over the living world-model graph.
|
|
37
37
|
GR_graph_reasoning: "answer a multi-hop design question (writers reaching a locus AND deliverable carrying a cargo form) as a PROVENANCED graph traversal (vs an ungrounded agent that cannot cite a path)"
|
|
38
|
+
# v0.3.2 (v5.0): the matured co-scientist as reference solver.
|
|
39
|
+
CS_co_scientist: "end-to-end grounded design: multiple materially-distinct legal confidence-tagged strategies, each citation-grounded + scope-ledgered, no-fabrication across the full reasoning stack (vs an ungrounded agent producing none of these)"
|
|
38
40
|
|
|
39
41
|
tasks:
|
|
40
42
|
- id: site_selection_blind_gsh
|
|
@@ -207,3 +209,18 @@ tasks:
|
|
|
207
209
|
circular: false
|
|
208
210
|
note: "v4.5 world-model graph: a design question answered as one grounded traversal; an ungrounded agent
|
|
209
211
|
has no graph and cannot produce a provenanced path (0 by construction). no-fabrication holds."
|
|
212
|
+
|
|
213
|
+
# ---- v0.3.2 (v5.0): the matured co-scientist as the reference solver.
|
|
214
|
+
- id: co_scientist_grounded
|
|
215
|
+
family: CS_co_scientist
|
|
216
|
+
scorer: "pen_stack.validate.bench_coscientist_tasks:run"
|
|
217
|
+
metric: "co_scientist_grounded_rate"
|
|
218
|
+
baseline_metric: "ungrounded_baseline_rate"
|
|
219
|
+
higher_is_better: true
|
|
220
|
+
ground_truth: "frozen panel of write goals; a recommendation set is 'fully grounded' iff it is multiple
|
|
221
|
+
materially-distinct (>=2 design axes) + each legal (verifier) + confidence-tagged (calibrated) + the
|
|
222
|
+
rationale's citations are in the curated DOI set + the scope ledger is complete + no-fabrication - all
|
|
223
|
+
mechanistic/verifier facts, not the agent's own claim (non-circular)"
|
|
224
|
+
circular: false
|
|
225
|
+
note: "v5.0 capstone: the matured co-scientist; the central gate is no-fabrication under the FULL reasoning
|
|
226
|
+
stack. An ungrounded agent produces none of these grounded properties (0 by construction)."
|
|
@@ -36,13 +36,13 @@ cell_types:
|
|
|
36
36
|
note: "iPSC/ESC; broad chromatin but TRIP durability not measured here -> durability OOD-labelled, degraded."
|
|
37
37
|
primary_T_cell:
|
|
38
38
|
tier: A
|
|
39
|
-
|
|
39
|
+
ontology: "CL:0000084" # T cell (Cell Ontology); CAR-T relevant
|
|
40
40
|
coverage: partial
|
|
41
41
|
tracks: [atac, expression]
|
|
42
42
|
note: "primary T cells (CAR-T context); accessibility + expression only -> histone-dependent safety degraded."
|
|
43
43
|
hepatocyte:
|
|
44
44
|
tier: A
|
|
45
|
-
|
|
45
|
+
ontology: "CL:0000182" # hepatocyte (Cell Ontology); in-vivo liver target
|
|
46
46
|
coverage: partial
|
|
47
47
|
tracks: [atac, h3k27ac, expression]
|
|
48
48
|
note: "primary hepatocytes (in-vivo liver target); partial panel -> graceful degradation, scope-flagged."
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# The co-scientist (v5.0)
|
|
2
|
+
|
|
3
|
+
v5.0 matures the reasoning layer on top of the verifier (v3.3), the environment (v3.4), the oracle mesh
|
|
4
|
+
(v4.0), and the living world-model (v4.5). Give it a goal and an intent and it returns a small set of
|
|
5
|
+
**materially distinct, ranked, fully-traceable strategies** — each verified, calibrated, cited, and
|
|
6
|
+
scope-ledgered — while the **no-fabrication guarantee holds by construction**: the reasoning layer proposes
|
|
7
|
+
and critiques, but every number still comes from a validated tool or oracle.
|
|
8
|
+
|
|
9
|
+
> **The central invariant.** Intelligence rises while groundedness never falls. A test asserts no-fabrication
|
|
10
|
+
> across the *full* reasoning stack (`pen_stack/validate/bench_coscientist_tasks.py`).
|
|
11
|
+
|
|
12
|
+
## What it does (`pen_stack/agent/co_scientist.py`, `pen_stack/agent/cite.py`)
|
|
13
|
+
|
|
14
|
+
| Capability | Function | Guarantee |
|
|
15
|
+
|---|---|---|
|
|
16
|
+
| **Multiple distinct strategies** | `propose_strategies(goal)` | 2–3 strategies differing on ≥2 design axes (write-type / writer / delivery / intent) — *materially* distinct, not reworded (`distinctness()` measures it); each independently **legal** + **confidence-tagged** |
|
|
17
|
+
| **Deliberative planning** | `deliberate(goal)` | the deliberative planner vs the deterministic `pen_agent` baseline, head-to-head; both grounded |
|
|
18
|
+
| **Self-critique / revise** | `critique_and_revise(design)` | the critic only flags + suggests a design-level swap (never invents a number); the revision is **re-verified**; falsifiable — it improves flawed plans (illegal→legal) and never spuriously touches clean ones (`critique_falsifiability()`) |
|
|
19
|
+
| **Cited rationale** | `cited_rationale(design)` | the "why" cites DOIs **drawn from the curated world-model** (so they resolve by construction); a hallucinated-citation guard rejects any DOI not in the curated set |
|
|
20
|
+
| **Scope ledger** | `scope_ledger(design)` | per recommendation, an itemised list of what **was** assessed (legality / reachability / delivery / payload / calibrated confidence) and what was **not** (the standing known-unknowns) — never silently omitted |
|
|
21
|
+
| **Scoped generalisation** | `generalise(task)` | adjacent genetic-engineering tasks are **grounded-or-refused**: answered only if they map to an existing grounded capability, otherwise refused with a scope statement |
|
|
22
|
+
|
|
23
|
+
## Honest scope
|
|
24
|
+
|
|
25
|
+
A better reasoner is **not a complete model of the cell**. structure→phenotype, in-vivo behaviour,
|
|
26
|
+
immunogenicity magnitude, long-term durability and higher-order epistasis remain out of scope — the
|
|
27
|
+
co-scientist makes that boundary *legible* (the scope ledger), it does not close it. Self-critique and
|
|
28
|
+
multi-strategy ship only because they help on held-out checks, or are reported as not-yet-useful.
|
|
29
|
+
Generalisation is approached only as far as the grounding allows; the rest is refused, not faked.
|
|
30
|
+
|
|
31
|
+
See `prereg/ws_{plan,crit,cite}.yaml` and the `co_scientist_grounded` bench task (Genome-Writing Bench v0.3.2).
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
"""PEN-STACK v3.0 - open infrastructure for genome writing."""
|
|
2
|
-
__version__ = "
|
|
2
|
+
__version__ = "5.0.0"
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Cited mechanistic rationale + scoped generalisation (v5.0, WS-CITE + WS-GEN).
|
|
2
|
+
|
|
3
|
+
Every recommendation carries a short, literature-cited "why". Crucially the citations are **drawn from the
|
|
4
|
+
curated world-model** (the verifier rule provenance, the writer/delivery/locus DOIs) — not generated by a
|
|
5
|
+
language model — so they **resolve by construction**. A hallucinated-citation guard rejects any DOI that is
|
|
6
|
+
not in the curated, already-verified set (WS-CITE). Numbers in the rationale remain tool-sourced (the prose is
|
|
7
|
+
a presentation layer over verified facts; no quantity is invented).
|
|
8
|
+
|
|
9
|
+
WS-GEN: generalisation toward adjacent genetic-engineering tasks is **grounded-or-refused** — a task is
|
|
10
|
+
answered only if it maps to an existing grounded capability; anything else is refused with a scope statement,
|
|
11
|
+
never faked.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from functools import lru_cache
|
|
16
|
+
|
|
17
|
+
import yaml
|
|
18
|
+
|
|
19
|
+
from pen_stack._resources import resource
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@lru_cache(maxsize=1)
|
|
23
|
+
def curated_dois() -> frozenset[str]:
|
|
24
|
+
"""The set of DOIs that appear in the CURATED, already-verified world-model (delivery palette, GSH loci,
|
|
25
|
+
writer panel, rule provenance). A citation is 'grounded' iff its DOI is in this set — so a citation can
|
|
26
|
+
only ever point at a source the substrate has actually curated (no hallucinated references)."""
|
|
27
|
+
dois: set[str] = set()
|
|
28
|
+
veh = yaml.safe_load(resource("configs/delivery_vehicles.yaml").read_text(encoding="utf-8"))["vehicles"]
|
|
29
|
+
for v in veh.values():
|
|
30
|
+
dois.update(v.get("dois", []) or [])
|
|
31
|
+
gsh = yaml.safe_load(resource("configs/gsh_validated_heldout.yaml").read_text(encoding="utf-8"))["gsh"]
|
|
32
|
+
for g in gsh:
|
|
33
|
+
if g.get("doi"):
|
|
34
|
+
dois.add(g["doi"])
|
|
35
|
+
import csv
|
|
36
|
+
with open(resource("data/writer_panel.csv"), encoding="utf-8") as f:
|
|
37
|
+
for row in csv.DictReader(f):
|
|
38
|
+
if row.get("doi"):
|
|
39
|
+
dois.add(row["doi"])
|
|
40
|
+
# rule provenance DOIs
|
|
41
|
+
from pen_stack.rules import load_ruleset
|
|
42
|
+
for r in load_ruleset().rules:
|
|
43
|
+
dois.update(r.provenance.get("doi", []) or [])
|
|
44
|
+
return frozenset(dois)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def citations_grounded(dois: list[str]) -> dict:
|
|
48
|
+
"""Hallucinated-citation guard: every cited DOI must be in the curated set. Returns the verdict + any
|
|
49
|
+
ungrounded DOIs (which would be a hallucination and are rejected)."""
|
|
50
|
+
curated = curated_dois()
|
|
51
|
+
ungrounded = [d for d in dois if d not in curated]
|
|
52
|
+
return {"all_grounded": not ungrounded, "ungrounded": ungrounded, "n_checked": len(dois)}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def cited_rationale(design: dict) -> dict:
|
|
56
|
+
"""A short, literature-cited mechanistic 'why' for a design, with citations DRAWN FROM the curated
|
|
57
|
+
world-model (so they resolve by construction). Numbers are tool-sourced; the prose is presentation only."""
|
|
58
|
+
from pen_stack.graph import build_graph
|
|
59
|
+
from pen_stack.verify import verify
|
|
60
|
+
g = build_graph()
|
|
61
|
+
v = verify(design)
|
|
62
|
+
fam = design.get("writer_family")
|
|
63
|
+
veh = design.get("delivery_vehicle")
|
|
64
|
+
cites: list[dict] = []
|
|
65
|
+
wnode = g.nodes.get(f"writer:{fam}")
|
|
66
|
+
if wnode:
|
|
67
|
+
for d in wnode.props.get("dois", [])[:1]:
|
|
68
|
+
cites.append({"doi": d, "claim": f"{fam} mechanism / characterisation", "source": "world-model writer"})
|
|
69
|
+
vnode = g.nodes.get(f"vehicle:{veh}")
|
|
70
|
+
if vnode:
|
|
71
|
+
for d in vnode.props.get("dois", [])[:1]:
|
|
72
|
+
cites.append({"doi": d, "claim": f"{veh} delivery properties", "source": "world-model vehicle"})
|
|
73
|
+
form = wnode.props.get("output_form") if wnode else None
|
|
74
|
+
verdict = "legal" if v.legal else "illegal"
|
|
75
|
+
rationale = (f"{fam} ({form}-form) installs a {design.get('cargo_bp')} bp cargo via {veh}; the rule-grounded "
|
|
76
|
+
f"verifier judges this {verdict}"
|
|
77
|
+
+ (f" (violated: {', '.join(x['rule_id'] for x in v.violations)})" if not v.legal else "")
|
|
78
|
+
+ (f"; calibrated confidence {v.confidence}" if v.confidence is not None else
|
|
79
|
+
"; confidence abstained (unscored)") + ".")
|
|
80
|
+
guard = citations_grounded([c["doi"] for c in cites])
|
|
81
|
+
return {"design": {k: val for k, val in design.items() if not str(k).startswith("_")},
|
|
82
|
+
"rationale": rationale, "citations": cites, "n_citations": len(cites),
|
|
83
|
+
"citations_grounded": guard["all_grounded"], "ungrounded_citations": guard["ungrounded"],
|
|
84
|
+
"no_fabrication": v.no_fabrication and guard["all_grounded"],
|
|
85
|
+
"note": "citations are drawn from the curated world-model (resolve by construction); numbers are "
|
|
86
|
+
"verifier-sourced; the prose is a presentation layer (no fabricated quantity)."}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# --------------------------------------------------------------------------------------------------
|
|
90
|
+
# WS-GEN - scoped generalisation: grounded-or-refused.
|
|
91
|
+
# --------------------------------------------------------------------------------------------------
|
|
92
|
+
# adjacent genetic-engineering tasks that MAP to an existing grounded capability (answerable);
|
|
93
|
+
# anything else is refused with a scope statement.
|
|
94
|
+
_GROUNDED_TASKS = {
|
|
95
|
+
"delivery_selection": "maps to the v3.3 delivery rules (cargo-form + capacity + integration)",
|
|
96
|
+
"write_type_legality": "maps to the v3.3 write-type router + verifier",
|
|
97
|
+
"off_target_screen": "maps to the bridge off-target engine (a screen, not a per-site calculator)",
|
|
98
|
+
"writer_variant_critique": "maps to the v4.0 writer-verification branch (score/critique, never invent)",
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def generalise(task: str, payload: dict | None = None) -> dict:
|
|
103
|
+
"""Answer an adjacent genetic-engineering task ONLY if it maps to a grounded capability; otherwise REFUSE
|
|
104
|
+
with a scope statement (Principle 4: grounded-or-refused, never faked)."""
|
|
105
|
+
payload = payload or {}
|
|
106
|
+
if task not in _GROUNDED_TASKS:
|
|
107
|
+
return {"task": task, "grounded": False, "refused": True,
|
|
108
|
+
"scope_statement": f"'{task}' has no grounding in PEN-STACK; refused rather than faked. "
|
|
109
|
+
f"Grounded tasks: {sorted(_GROUNDED_TASKS)}.",
|
|
110
|
+
"no_fabrication": True}
|
|
111
|
+
result: dict = {"task": task, "grounded": True, "refused": False,
|
|
112
|
+
"grounding": _GROUNDED_TASKS[task], "no_fabrication": True}
|
|
113
|
+
if task in ("delivery_selection", "write_type_legality") and payload:
|
|
114
|
+
from pen_stack.verify import verify
|
|
115
|
+
v = verify(payload)
|
|
116
|
+
result["verdict"] = {"legal": v.legal, "confidence": v.confidence,
|
|
117
|
+
"violations": [x["rule_id"] for x in v.violations]}
|
|
118
|
+
return result
|