nanoscope-lab 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nanoscope_lab-0.3.0/.gitignore +16 -0
- nanoscope_lab-0.3.0/LICENSE +21 -0
- nanoscope_lab-0.3.0/Makefile +30 -0
- nanoscope_lab-0.3.0/PKG-INFO +144 -0
- nanoscope_lab-0.3.0/README.md +90 -0
- nanoscope_lab-0.3.0/docs/blocks.md +144 -0
- nanoscope_lab-0.3.0/docs/checklist.md +744 -0
- nanoscope_lab-0.3.0/docs/dataset-card.md +33 -0
- nanoscope_lab-0.3.0/docs/learn.md +113 -0
- nanoscope_lab-0.3.0/docs/openapi.json +5204 -0
- nanoscope_lab-0.3.0/docs/plan-tool-landscape.md +136 -0
- nanoscope_lab-0.3.0/docs/plan-tool.md +1103 -0
- nanoscope_lab-0.3.0/docs/project-nanoscope.md +279 -0
- nanoscope_lab-0.3.0/docs/research.md +241 -0
- nanoscope_lab-0.3.0/docs/server.md +77 -0
- nanoscope_lab-0.3.0/nanoscope/__init__.py +14 -0
- nanoscope_lab-0.3.0/nanoscope/__main__.py +5 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/bigram/seed-0/config.json +44 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/bigram/seed-0/metrics.jsonl +50 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/bigram/seed-0/samples.txt +26 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/bigram/seed-1/config.json +44 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/bigram/seed-1/metrics.jsonl +50 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/bigram/seed-1/samples.txt +26 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/bigram/seed-2/config.json +44 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/bigram/seed-2/metrics.jsonl +50 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/bigram/seed-2/samples.txt +57 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/gpt2/seed-0/config.json +47 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/gpt2/seed-0/metrics.jsonl +50 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/gpt2/seed-0/samples.txt +28 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/gpt2/seed-1/config.json +47 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/gpt2/seed-1/metrics.jsonl +50 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/gpt2/seed-1/samples.txt +71 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/gpt2/seed-2/config.json +47 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/gpt2/seed-2/metrics.jsonl +50 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/gpt2/seed-2/samples.txt +48 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/modern/seed-0/config.json +54 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/modern/seed-0/metrics.jsonl +50 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/modern/seed-0/samples.txt +49 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/modern/seed-1/config.json +54 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/modern/seed-1/metrics.jsonl +50 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/modern/seed-1/samples.txt +37 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/modern/seed-2/config.json +54 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/modern/seed-2/metrics.jsonl +50 -0
- nanoscope_lab-0.3.0/nanoscope/baselines/tinystories-5min/modern/seed-2/samples.txt +55 -0
- nanoscope_lab-0.3.0/nanoscope/bench.py +136 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/__init__.py +73 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/attention.py +96 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/catalog.py +91 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/composite.py +69 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/discover.py +176 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/embedding.py +36 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/gate.py +59 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/graph.py +692 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/head.py +20 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/mlp.py +48 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/norm.py +39 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/positional.py +55 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/primitives.py +149 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/registry.py +120 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/spec.py +88 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/structure.py +113 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/templates/__init__.py +6 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/templates/attention.py +40 -0
- nanoscope_lab-0.3.0/nanoscope/blocks/templates/block.py +31 -0
- nanoscope_lab-0.3.0/nanoscope/blockstats.py +138 -0
- nanoscope_lab-0.3.0/nanoscope/cli.py +427 -0
- nanoscope_lab-0.3.0/nanoscope/compare.py +361 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/__init__.py +0 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/01-bigram/lesson.md +40 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/01-bigram/lesson.toml +37 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/01-bigram/starter.py +19 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/02-mlp/lesson.md +35 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/02-mlp/lesson.toml +27 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/02-mlp/starter.py +31 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/03-attention-head/lesson.md +40 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/03-attention-head/lesson.toml +33 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/03-attention-head/starter.py +28 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/04-multi-head/lesson.md +32 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/04-multi-head/lesson.toml +34 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/04-multi-head/starter.py +30 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/05-block/lesson.md +37 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/05-block/lesson.toml +39 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/05-block/starter.py +51 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/06-gpt2/lesson.md +40 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/06-gpt2/lesson.toml +36 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/06-gpt2/starter.py +84 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/foundations/path.toml +7 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/01-rmsnorm/lesson.md +29 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/01-rmsnorm/lesson.toml +26 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/01-rmsnorm/starter.py +15 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/02-rope/lesson.md +34 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/02-rope/lesson.toml +25 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/02-rope/starter.py +21 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/03-swiglu/lesson.md +27 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/03-swiglu/lesson.toml +25 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/03-swiglu/starter.py +18 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/04-gqa/lesson.md +31 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/04-gqa/lesson.toml +32 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/04-gqa/starter.py +39 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/05-qk-norm/lesson.md +32 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/05-qk-norm/lesson.toml +35 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/05-qk-norm/starter.py +44 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/06-z-loss/lesson.md +31 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/06-z-loss/lesson.toml +26 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/06-z-loss/starter.py +15 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/07-assemble/lesson.md +43 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/07-assemble/lesson.toml +49 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/07-assemble/starter.py +22 -0
- nanoscope_lab-0.3.0/nanoscope/curricula/modern-block/path.toml +12 -0
- nanoscope_lab-0.3.0/nanoscope/dataset.py +299 -0
- nanoscope_lab-0.3.0/nanoscope/estimate.py +74 -0
- nanoscope_lab-0.3.0/nanoscope/fsutil.py +16 -0
- nanoscope_lab-0.3.0/nanoscope/hardware.py +54 -0
- nanoscope_lab-0.3.0/nanoscope/inspect.py +280 -0
- nanoscope_lab-0.3.0/nanoscope/integrations.py +100 -0
- nanoscope_lab-0.3.0/nanoscope/jobs/__init__.py +1 -0
- nanoscope_lab-0.3.0/nanoscope/jobs/execute.py +143 -0
- nanoscope_lab-0.3.0/nanoscope/jobs/inference.py +143 -0
- nanoscope_lab-0.3.0/nanoscope/jobs/payload.py +28 -0
- nanoscope_lab-0.3.0/nanoscope/jobs/runner.py +95 -0
- nanoscope_lab-0.3.0/nanoscope/jobs/worker.py +273 -0
- nanoscope_lab-0.3.0/nanoscope/learn/__init__.py +4 -0
- nanoscope_lab-0.3.0/nanoscope/learn/checks.py +725 -0
- nanoscope_lab-0.3.0/nanoscope/learn/cli.py +300 -0
- nanoscope_lab-0.3.0/nanoscope/learn/gating.py +177 -0
- nanoscope_lab-0.3.0/nanoscope/learn/loader.py +381 -0
- nanoscope_lab-0.3.0/nanoscope/learn/progress.py +96 -0
- nanoscope_lab-0.3.0/nanoscope/learn/unlocks.py +105 -0
- nanoscope_lab-0.3.0/nanoscope/log.py +40 -0
- nanoscope_lab-0.3.0/nanoscope/modelref.py +96 -0
- nanoscope_lab-0.3.0/nanoscope/models/__init__.py +5 -0
- nanoscope_lab-0.3.0/nanoscope/models/bigram.py +20 -0
- nanoscope_lab-0.3.0/nanoscope/models/gpt2.py +33 -0
- nanoscope_lab-0.3.0/nanoscope/models/modern.py +54 -0
- nanoscope_lab-0.3.0/nanoscope/paths.py +94 -0
- nanoscope_lab-0.3.0/nanoscope/prepare.py +78 -0
- nanoscope_lab-0.3.0/nanoscope/presets.py +146 -0
- nanoscope_lab-0.3.0/nanoscope/progress.py +242 -0
- nanoscope_lab-0.3.0/nanoscope/queue.py +367 -0
- nanoscope_lab-0.3.0/nanoscope/reference/__init__.py +5 -0
- nanoscope_lab-0.3.0/nanoscope/reference/functional.py +200 -0
- nanoscope_lab-0.3.0/nanoscope/reference/gpt2_ref.py +102 -0
- nanoscope_lab-0.3.0/nanoscope/reference/modern_ref.py +199 -0
- nanoscope_lab-0.3.0/nanoscope/run.py +485 -0
- nanoscope_lab-0.3.0/nanoscope/runref.py +51 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/__init__.py +34 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/bench.v1.json +70 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/blocks.v1.json +149 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/blockstats.v1.json +71 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/check.v1.json +85 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/comparison.v1.json +182 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/config.v1.json +97 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/describe.v1.json +195 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/graph.v1.json +416 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/job.v1.json +303 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/lesson.v1.json +151 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/path.v1.json +80 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/plan.v1.json +41 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/prepare.v1.json +76 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/problem.v1.json +31 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/progress.v1.json +103 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/results.v1.json +50 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/status.v1.json +105 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/study.v1.json +48 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/studyspec.v1.json +30 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/unlocks.v1.json +74 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/upgrade.py +51 -0
- nanoscope_lab-0.3.0/nanoscope/schemas/worker.v1.json +20 -0
- nanoscope_lab-0.3.0/nanoscope/server/__init__.py +6 -0
- nanoscope_lab-0.3.0/nanoscope/server/app.py +109 -0
- nanoscope_lab-0.3.0/nanoscope/server/auth.py +47 -0
- nanoscope_lab-0.3.0/nanoscope/server/errors.py +75 -0
- nanoscope_lab-0.3.0/nanoscope/server/jobs.py +30 -0
- nanoscope_lab-0.3.0/nanoscope/server/models.py +244 -0
- nanoscope_lab-0.3.0/nanoscope/server/openapi.py +36 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/__init__.py +0 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/blocks.py +35 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/compare.py +28 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/events.py +66 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/files.py +172 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/graph.py +79 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/hardware.py +112 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/hub.py +28 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/jobs.py +44 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/learn.py +254 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/meta.py +21 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/models.py +93 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/presets.py +21 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/runs.py +344 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/studies.py +194 -0
- nanoscope_lab-0.3.0/nanoscope/server/routes/validate.py +134 -0
- nanoscope_lab-0.3.0/nanoscope/server/settings.py +80 -0
- nanoscope_lab-0.3.0/nanoscope/server/sse.py +26 -0
- nanoscope_lab-0.3.0/nanoscope/server/tail.py +225 -0
- nanoscope_lab-0.3.0/nanoscope/server/worker.py +36 -0
- nanoscope_lab-0.3.0/nanoscope/server/workspace.py +68 -0
- nanoscope_lab-0.3.0/nanoscope/sizing.py +61 -0
- nanoscope_lab-0.3.0/nanoscope/specs.py +218 -0
- nanoscope_lab-0.3.0/nanoscope/statistics.py +140 -0
- nanoscope_lab-0.3.0/nanoscope/status.py +93 -0
- nanoscope_lab-0.3.0/nanoscope/store.py +168 -0
- nanoscope_lab-0.3.0/nanoscope/study.py +588 -0
- nanoscope_lab-0.3.0/nanoscope/studyspec.py +163 -0
- nanoscope_lab-0.3.0/nanoscope/tokenizer.py +91 -0
- nanoscope_lab-0.3.0/nanoscope/train_loop.py +392 -0
- nanoscope_lab-0.3.0/pyproject.toml +103 -0
- nanoscope_lab-0.3.0/tests/conftest.py +38 -0
- nanoscope_lab-0.3.0/tests/fakes.py +13 -0
- nanoscope_lab-0.3.0/tests/fixtures/graphs/code_only.py +23 -0
- nanoscope_lab-0.3.0/tests/fixtures/graphs/composite_template.py +18 -0
- nanoscope_lab-0.3.0/tests/fixtures/graphs/gpt2_like.py +11 -0
- nanoscope_lab-0.3.0/tests/fixtures/graphs/layer_pattern.py +19 -0
- nanoscope_lab-0.3.0/tests/fixtures/graphs/modern_like.py +12 -0
- nanoscope_lab-0.3.0/tests/fixtures/graphs/mylm.py +12 -0
- nanoscope_lab-0.3.0/tests/fixtures/graphs/odd_formatting.py +31 -0
- nanoscope_lab-0.3.0/tests/fixtures/graphs/opaque_block.py +14 -0
- nanoscope_lab-0.3.0/tests/fixtures/graphs/template_fill.py +19 -0
- nanoscope_lab-0.3.0/tests/helpers.py +12 -0
- nanoscope_lab-0.3.0/tests/learn_helpers.py +41 -0
- nanoscope_lab-0.3.0/tests/server/conftest.py +33 -0
- nanoscope_lab-0.3.0/tests/server/test_api_learn.py +169 -0
- nanoscope_lab-0.3.0/tests/server/test_api_models.py +73 -0
- nanoscope_lab-0.3.0/tests/server/test_app.py +209 -0
- nanoscope_lab-0.3.0/tests/server/test_auth.py +134 -0
- nanoscope_lab-0.3.0/tests/server/test_e2e.py +157 -0
- nanoscope_lab-0.3.0/tests/server/test_files.py +266 -0
- nanoscope_lab-0.3.0/tests/server/test_import_guard.py +193 -0
- nanoscope_lab-0.3.0/tests/server/test_openapi.py +31 -0
- nanoscope_lab-0.3.0/tests/server/test_routes.py +395 -0
- nanoscope_lab-0.3.0/tests/server/test_runs.py +355 -0
- nanoscope_lab-0.3.0/tests/server/test_studies.py +127 -0
- nanoscope_lab-0.3.0/tests/server/test_tail.py +245 -0
- nanoscope_lab-0.3.0/tests/solutions/foundations/01-bigram.py +11 -0
- nanoscope_lab-0.3.0/tests/solutions/foundations/02-mlp.py +23 -0
- nanoscope_lab-0.3.0/tests/solutions/foundations/03-attention-head.py +21 -0
- nanoscope_lab-0.3.0/tests/solutions/foundations/04-multi-head.py +29 -0
- nanoscope_lab-0.3.0/tests/solutions/foundations/05-block.py +42 -0
- nanoscope_lab-0.3.0/tests/solutions/foundations/06-gpt2.py +75 -0
- nanoscope_lab-0.3.0/tests/solutions/modern-block/01-rmsnorm.py +13 -0
- nanoscope_lab-0.3.0/tests/solutions/modern-block/02-rope.py +17 -0
- nanoscope_lab-0.3.0/tests/solutions/modern-block/03-swiglu.py +14 -0
- nanoscope_lab-0.3.0/tests/solutions/modern-block/04-gqa.py +33 -0
- nanoscope_lab-0.3.0/tests/solutions/modern-block/05-qk-norm.py +41 -0
- nanoscope_lab-0.3.0/tests/solutions/modern-block/06-z-loss.py +11 -0
- nanoscope_lab-0.3.0/tests/solutions/modern-block/07-assemble.py +16 -0
- nanoscope_lab-0.3.0/tests/test_bench.py +44 -0
- nanoscope_lab-0.3.0/tests/test_blocks.py +750 -0
- nanoscope_lab-0.3.0/tests/test_blockstats.py +97 -0
- nanoscope_lab-0.3.0/tests/test_checks.py +457 -0
- nanoscope_lab-0.3.0/tests/test_compare.py +172 -0
- nanoscope_lab-0.3.0/tests/test_compile.py +38 -0
- nanoscope_lab-0.3.0/tests/test_curricula.py +361 -0
- nanoscope_lab-0.3.0/tests/test_data.py +207 -0
- nanoscope_lab-0.3.0/tests/test_describe.py +177 -0
- nanoscope_lab-0.3.0/tests/test_estimate.py +50 -0
- nanoscope_lab-0.3.0/tests/test_first_model_notebook.py +72 -0
- nanoscope_lab-0.3.0/tests/test_gating.py +526 -0
- nanoscope_lab-0.3.0/tests/test_graph.py +442 -0
- nanoscope_lab-0.3.0/tests/test_jobs.py +373 -0
- nanoscope_lab-0.3.0/tests/test_learn.py +403 -0
- nanoscope_lab-0.3.0/tests/test_models.py +253 -0
- nanoscope_lab-0.3.0/tests/test_packaging.py +42 -0
- nanoscope_lab-0.3.0/tests/test_paths.py +33 -0
- nanoscope_lab-0.3.0/tests/test_progress.py +128 -0
- nanoscope_lab-0.3.0/tests/test_queue.py +213 -0
- nanoscope_lab-0.3.0/tests/test_reference.py +135 -0
- nanoscope_lab-0.3.0/tests/test_run.py +329 -0
- nanoscope_lab-0.3.0/tests/test_schemas.py +104 -0
- nanoscope_lab-0.3.0/tests/test_specs.py +74 -0
- nanoscope_lab-0.3.0/tests/test_statistics.py +26 -0
- nanoscope_lab-0.3.0/tests/test_status.py +163 -0
- nanoscope_lab-0.3.0/tests/test_store.py +96 -0
- nanoscope_lab-0.3.0/tests/test_study.py +547 -0
- nanoscope_lab-0.3.0/tests/test_studyspec.py +145 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 almajd3713
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
.PHONY: install test test-all test-blocks test-curricula test-curricula-full openapi lint typecheck check
|
|
2
|
+
|
|
3
|
+
install: ## everything, including dev tools
|
|
4
|
+
uv sync --all-extras
|
|
5
|
+
|
|
6
|
+
test: ## fast offline tests
|
|
7
|
+
uv run pytest -m "not network and not gpu"
|
|
8
|
+
|
|
9
|
+
test-all: ## also the first-notebook timing test (downloads TinyStories once) and GPU tests
|
|
10
|
+
uv run pytest
|
|
11
|
+
|
|
12
|
+
test-blocks: ## the block library: blocks, references, graph, describe, block stats
|
|
13
|
+
uv run pytest tests/test_blocks.py tests/test_reference.py tests/test_graph.py tests/test_describe.py tests/test_blockstats.py
|
|
14
|
+
|
|
15
|
+
test-curricula: ## every lesson's starter fails and its solution passes (offline, CPU)
|
|
16
|
+
uv run pytest tests/test_curricula.py tests/test_checks.py tests/test_learn.py tests/test_gating.py -m "not network and not gpu"
|
|
17
|
+
|
|
18
|
+
test-curricula-full: ## also train every lesson's solution on real TinyStories (minutes; needs network once)
|
|
19
|
+
uv run pytest tests/test_curricula.py
|
|
20
|
+
|
|
21
|
+
openapi: ## regenerate docs/openapi.json from the API (commit the result)
|
|
22
|
+
uv run python -m nanoscope.server.openapi docs/openapi.json
|
|
23
|
+
|
|
24
|
+
lint:
|
|
25
|
+
uv run ruff check nanoscope tests
|
|
26
|
+
|
|
27
|
+
typecheck:
|
|
28
|
+
uv run pyright nanoscope
|
|
29
|
+
|
|
30
|
+
check: lint typecheck test ## what CI runs
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: nanoscope-lab
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: See what your language model learns
|
|
5
|
+
Project-URL: Homepage, https://github.com/almajd3713/nanoscope
|
|
6
|
+
Project-URL: Source, https://github.com/almajd3713/nanoscope
|
|
7
|
+
Project-URL: Issues, https://github.com/almajd3713/nanoscope/issues
|
|
8
|
+
Author: almajd3713
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: education,interpretability,language-models,pytorch,research,transformers
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Education
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Requires-Python: <3.13,>=3.10
|
|
22
|
+
Requires-Dist: datasets<5,>=3.0
|
|
23
|
+
Requires-Dist: huggingface-hub<2,>=0.27
|
|
24
|
+
Requires-Dist: jsonschema<5,>=4.20
|
|
25
|
+
Requires-Dist: matplotlib<4,>=3.8
|
|
26
|
+
Requires-Dist: numpy<3,>=1.26
|
|
27
|
+
Requires-Dist: scipy<2,>=1.11
|
|
28
|
+
Requires-Dist: tiktoken<1,>=0.8
|
|
29
|
+
Requires-Dist: tokenizers<1,>=0.20
|
|
30
|
+
Requires-Dist: tomli-w>=1.0
|
|
31
|
+
Requires-Dist: tomli>=2; python_full_version < '3.11'
|
|
32
|
+
Requires-Dist: torch<3,>=2.2
|
|
33
|
+
Provides-Extra: dev
|
|
34
|
+
Requires-Dist: httpx<1,>=0.27; extra == 'dev'
|
|
35
|
+
Requires-Dist: hypothesis<7,>=6.100; extra == 'dev'
|
|
36
|
+
Requires-Dist: jsonschema<5,>=4.20; extra == 'dev'
|
|
37
|
+
Requires-Dist: libcst<2,>=1.1; extra == 'dev'
|
|
38
|
+
Requires-Dist: nbformat<6,>=5.10; extra == 'dev'
|
|
39
|
+
Requires-Dist: pyright>=1.1.390; extra == 'dev'
|
|
40
|
+
Requires-Dist: pytest-cov<8,>=6; extra == 'dev'
|
|
41
|
+
Requires-Dist: pytest<9,>=8.3; extra == 'dev'
|
|
42
|
+
Requires-Dist: ruff<1,>=0.9; extra == 'dev'
|
|
43
|
+
Provides-Extra: graph
|
|
44
|
+
Requires-Dist: libcst<2,>=1.1; extra == 'graph'
|
|
45
|
+
Provides-Extra: server
|
|
46
|
+
Requires-Dist: fastapi<1,>=0.115; extra == 'server'
|
|
47
|
+
Requires-Dist: libcst<2,>=1.1; extra == 'server'
|
|
48
|
+
Requires-Dist: ruff<1,>=0.9; extra == 'server'
|
|
49
|
+
Requires-Dist: uvicorn[standard]<1,>=0.30; extra == 'server'
|
|
50
|
+
Requires-Dist: watchfiles<2,>=0.24; extra == 'server'
|
|
51
|
+
Provides-Extra: wandb
|
|
52
|
+
Requires-Dist: wandb<1,>=0.19; extra == 'wandb'
|
|
53
|
+
Description-Content-Type: text/markdown
|
|
54
|
+
|
|
55
|
+
# nanoscope
|
|
56
|
+
|
|
57
|
+
See what your language model learns. You write an `nn.Module`; nanoscope handles the
|
|
58
|
+
data, the training loop, evaluation, checkpoints and comparison against baselines.
|
|
59
|
+
|
|
60
|
+
It has two levels that share one core. Learners call `run()`. Researchers write a `Study`
|
|
61
|
+
with seeds, budgets, parameter matching and preregistration. The level changes what you
|
|
62
|
+
see, never which code runs.
|
|
63
|
+
|
|
64
|
+
## Learn
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
pip install git+https://github.com/almajd3713/nanoscope
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
import torch.nn as nn
|
|
72
|
+
from nanoscope import compare, run
|
|
73
|
+
|
|
74
|
+
class Bigram(nn.Module):
|
|
75
|
+
def __init__(self, vocab_size: int, d_model: int = 32):
|
|
76
|
+
super().__init__()
|
|
77
|
+
self.token_embedding = nn.Embedding(vocab_size, d_model)
|
|
78
|
+
self.head = nn.Linear(d_model, vocab_size, bias=False)
|
|
79
|
+
|
|
80
|
+
def forward(self, idx):
|
|
81
|
+
return self.head(self.token_embedding(idx))
|
|
82
|
+
|
|
83
|
+
result = run(Bigram, preset="tinystories-5min") # about a minute on a laptop CPU
|
|
84
|
+
result.plot()
|
|
85
|
+
compare(result, "gpt2") # against a shipped 3-seed baseline
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Work through the notebooks in order:
|
|
89
|
+
|
|
90
|
+
1. [`01-first-model`](notebooks/01-first-model.ipynb): write a bigram model and train it.
|
|
91
|
+
2. [`02-gpt2`](notebooks/02-gpt2.ipynb): a real transformer.
|
|
92
|
+
3. [`03-modern-block`](notebooks/03-modern-block.ipynb): RoPE, RMSNorm, SwiGLU, GQA, QK-norm, z-loss.
|
|
93
|
+
4. [`04-ablations`](notebooks/04-ablations.ipynb): which part matters, with seeds and confidence intervals.
|
|
94
|
+
|
|
95
|
+
Token data downloads from the Hub (`RedhouaneLazib/nanoscope-tokens`) when a preset has
|
|
96
|
+
it, and is tokenized locally otherwise. Everything is cached under `~/.nanoscope/data`.
|
|
97
|
+
|
|
98
|
+
## Research
|
|
99
|
+
|
|
100
|
+
See the [research guide](docs/research.md). The research program itself is in
|
|
101
|
+
[`docs/project-nanoscope.md`](docs/project-nanoscope.md).
|
|
102
|
+
|
|
103
|
+
## Build models from blocks
|
|
104
|
+
|
|
105
|
+
`GPT2` and `Modern` are short compositions of the blocks in `nanoscope.blocks`, and so can your
|
|
106
|
+
own models. See [docs/blocks.md](docs/blocks.md).
|
|
107
|
+
|
|
108
|
+
## Learn
|
|
109
|
+
|
|
110
|
+
Guided paths build the models step by step, with checks that say why. See
|
|
111
|
+
[docs/learn.md](docs/learn.md): `nanoscope learn list`, `learn start`, `learn check`.
|
|
112
|
+
|
|
113
|
+
## Command line
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
nanoscope presets # available presets
|
|
117
|
+
nanoscope run nanoscope/models/gpt2.py:GPT2 --seeds 3
|
|
118
|
+
nanoscope compare modern gpt2 --preset tinystories-5min
|
|
119
|
+
nanoscope study studies/m1_ablation.py --devices cuda:0
|
|
120
|
+
nanoscope report studies/m1_ablation.py # writes experiments/<name>/
|
|
121
|
+
nanoscope bench modern --compile reduce-overhead # speed, and whether CPU or GPU is the limit
|
|
122
|
+
nanoscope status runs # what is running and how far along
|
|
123
|
+
nanoscope describe nanoscope/models/modern.py:Modern # shapes, params, FLOPs, memory per module
|
|
124
|
+
nanoscope graph my_model.py # a model file's architecture, without running it
|
|
125
|
+
nanoscope blocks # the blocks models are composed from
|
|
126
|
+
nanoscope prepare-data tinystories-5min # download or tokenize now
|
|
127
|
+
nanoscope publish-data tinystories-5min user/repo # upload tokens to a Hub dataset
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
## Develop
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
uv sync --all-extras
|
|
134
|
+
make test # offline tests
|
|
135
|
+
make lint
|
|
136
|
+
make typecheck
|
|
137
|
+
make check # lint, typecheck, then test: what CI runs
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
## Data credits
|
|
141
|
+
|
|
142
|
+
Tokens are derived from [TinyStories](https://huggingface.co/datasets/roneneldan/TinyStories)
|
|
143
|
+
(CDLA-Sharing-1.0) and [FineWeb-Edu](https://huggingface.co/datasets/HuggingFaceFW/fineweb-edu)
|
|
144
|
+
(ODC-By 1.0).
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
# nanoscope
|
|
2
|
+
|
|
3
|
+
See what your language model learns. You write an `nn.Module`; nanoscope handles the
|
|
4
|
+
data, the training loop, evaluation, checkpoints and comparison against baselines.
|
|
5
|
+
|
|
6
|
+
It has two levels that share one core. Learners call `run()`. Researchers write a `Study`
|
|
7
|
+
with seeds, budgets, parameter matching and preregistration. The level changes what you
|
|
8
|
+
see, never which code runs.
|
|
9
|
+
|
|
10
|
+
## Learn
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
pip install git+https://github.com/almajd3713/nanoscope
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
```python
|
|
17
|
+
import torch.nn as nn
|
|
18
|
+
from nanoscope import compare, run
|
|
19
|
+
|
|
20
|
+
class Bigram(nn.Module):
|
|
21
|
+
def __init__(self, vocab_size: int, d_model: int = 32):
|
|
22
|
+
super().__init__()
|
|
23
|
+
self.token_embedding = nn.Embedding(vocab_size, d_model)
|
|
24
|
+
self.head = nn.Linear(d_model, vocab_size, bias=False)
|
|
25
|
+
|
|
26
|
+
def forward(self, idx):
|
|
27
|
+
return self.head(self.token_embedding(idx))
|
|
28
|
+
|
|
29
|
+
result = run(Bigram, preset="tinystories-5min") # about a minute on a laptop CPU
|
|
30
|
+
result.plot()
|
|
31
|
+
compare(result, "gpt2") # against a shipped 3-seed baseline
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Work through the notebooks in order:
|
|
35
|
+
|
|
36
|
+
1. [`01-first-model`](notebooks/01-first-model.ipynb): write a bigram model and train it.
|
|
37
|
+
2. [`02-gpt2`](notebooks/02-gpt2.ipynb): a real transformer.
|
|
38
|
+
3. [`03-modern-block`](notebooks/03-modern-block.ipynb): RoPE, RMSNorm, SwiGLU, GQA, QK-norm, z-loss.
|
|
39
|
+
4. [`04-ablations`](notebooks/04-ablations.ipynb): which part matters, with seeds and confidence intervals.
|
|
40
|
+
|
|
41
|
+
Token data downloads from the Hub (`RedhouaneLazib/nanoscope-tokens`) when a preset has
|
|
42
|
+
it, and is tokenized locally otherwise. Everything is cached under `~/.nanoscope/data`.
|
|
43
|
+
|
|
44
|
+
## Research
|
|
45
|
+
|
|
46
|
+
See the [research guide](docs/research.md). The research program itself is in
|
|
47
|
+
[`docs/project-nanoscope.md`](docs/project-nanoscope.md).
|
|
48
|
+
|
|
49
|
+
## Build models from blocks
|
|
50
|
+
|
|
51
|
+
`GPT2` and `Modern` are short compositions of the blocks in `nanoscope.blocks`, and so can your
|
|
52
|
+
own models. See [docs/blocks.md](docs/blocks.md).
|
|
53
|
+
|
|
54
|
+
## Learn
|
|
55
|
+
|
|
56
|
+
Guided paths build the models step by step, with checks that say why. See
|
|
57
|
+
[docs/learn.md](docs/learn.md): `nanoscope learn list`, `learn start`, `learn check`.
|
|
58
|
+
|
|
59
|
+
## Command line
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
nanoscope presets # available presets
|
|
63
|
+
nanoscope run nanoscope/models/gpt2.py:GPT2 --seeds 3
|
|
64
|
+
nanoscope compare modern gpt2 --preset tinystories-5min
|
|
65
|
+
nanoscope study studies/m1_ablation.py --devices cuda:0
|
|
66
|
+
nanoscope report studies/m1_ablation.py # writes experiments/<name>/
|
|
67
|
+
nanoscope bench modern --compile reduce-overhead # speed, and whether CPU or GPU is the limit
|
|
68
|
+
nanoscope status runs # what is running and how far along
|
|
69
|
+
nanoscope describe nanoscope/models/modern.py:Modern # shapes, params, FLOPs, memory per module
|
|
70
|
+
nanoscope graph my_model.py # a model file's architecture, without running it
|
|
71
|
+
nanoscope blocks # the blocks models are composed from
|
|
72
|
+
nanoscope prepare-data tinystories-5min # download or tokenize now
|
|
73
|
+
nanoscope publish-data tinystories-5min user/repo # upload tokens to a Hub dataset
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## Develop
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
uv sync --all-extras
|
|
80
|
+
make test # offline tests
|
|
81
|
+
make lint
|
|
82
|
+
make typecheck
|
|
83
|
+
make check # lint, typecheck, then test: what CI runs
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## Data credits
|
|
87
|
+
|
|
88
|
+
Tokens are derived from [TinyStories](https://huggingface.co/datasets/roneneldan/TinyStories)
|
|
89
|
+
(CDLA-Sharing-1.0) and [FineWeb-Edu](https://huggingface.co/datasets/HuggingFaceFW/fineweb-edu)
|
|
90
|
+
(ODC-By 1.0).
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
# Blocks
|
|
2
|
+
|
|
3
|
+
`nanoscope.blocks` is a library of short modules, one idea each, that models are composed
|
|
4
|
+
from. `GPT2` and `Modern` are themselves compositions of these blocks, so everything you read
|
|
5
|
+
here is what the shipped models run.
|
|
6
|
+
|
|
7
|
+
## Compose a model
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from nanoscope.blocks import Attention, Block, Decoder, RMSNorm, RoPE, SwiGLU
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class MyLM(Decoder):
|
|
14
|
+
def __init__(self, vocab_size: int, context_length: int = 256):
|
|
15
|
+
super().__init__(
|
|
16
|
+
vocab_size, context_length, d_model=128, n_layers=4,
|
|
17
|
+
block=Block(norm=RMSNorm(), mlp=SwiGLU(hidden=344),
|
|
18
|
+
attn=Attention(n_heads=4, n_kv_heads=2, pos=RoPE(), qk_norm=True)),
|
|
19
|
+
final_norm=RMSNorm(),
|
|
20
|
+
tie_weights=True, z_loss=1e-4,
|
|
21
|
+
)
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
`run(MyLM)` trains it like any model. Calling a block with only its options (`SwiGLU(hidden=344)`)
|
|
25
|
+
gives a *spec*, a description with no weights. `Decoder` builds the spec once per layer, so
|
|
26
|
+
layers never share parameters. A spec knows its arguments, so `spec.to_dict()` is what the graph
|
|
27
|
+
and `describe` show.
|
|
28
|
+
|
|
29
|
+
Useful `Decoder` options:
|
|
30
|
+
|
|
31
|
+
| Option | Meaning |
|
|
32
|
+
|---|---|
|
|
33
|
+
| `block` | One block spec used for every layer. |
|
|
34
|
+
| `pattern=[a, b]` | Instead of `block`: repeat a list of specs through the layers (`a, b, a, b, ...`), e.g. sliding-window and global attention. |
|
|
35
|
+
| `final_norm` | A norm after the last layer. |
|
|
36
|
+
| `pos_emb` | A position embedding added after the token embedding (`LearnedPosition()`); leave it out when attention carries positions (`RoPE`). |
|
|
37
|
+
| `tie_weights` | Share the output matrix with the token embedding. |
|
|
38
|
+
| `z_loss` | Add a penalty that keeps the softmax normaliser near 1. The model then returns `(logits, aux_loss)`. |
|
|
39
|
+
|
|
40
|
+
Parameters start as N(0, 0.02), biases at zero, and each layer's output projections at
|
|
41
|
+
`0.02 / sqrt(2 * n_layers)`.
|
|
42
|
+
|
|
43
|
+
## The representable subset
|
|
44
|
+
|
|
45
|
+
The architecture graph and `nanoscope graph` read your file with `ast`, without importing or
|
|
46
|
+
running it. A class is shown as a graph when it is a `Decoder` (or `Composite`) subclass whose
|
|
47
|
+
`__init__` is one `super().__init__(...)` call. Each argument is one of:
|
|
48
|
+
|
|
49
|
+
- a literal (`128`, `True`, `1e-4`),
|
|
50
|
+
- one of the class's own `__init__` parameters (`vocab_size`),
|
|
51
|
+
- a call to a registered block, with its options by keyword,
|
|
52
|
+
- a list of those (`pattern=[...]`).
|
|
53
|
+
|
|
54
|
+
A call to anything else becomes an *opaque* node: it is drawn, and `describe` reports its
|
|
55
|
+
shapes, but its inside is not editable. Any other expression is kept as text. Any other
|
|
56
|
+
*statement* in `__init__` (an assignment, a loop, an `if`) makes the class **code-only**: it
|
|
57
|
+
still appears, with the reason and the line, and you edit it as source.
|
|
58
|
+
|
|
59
|
+
```
|
|
60
|
+
$ nanoscope graph my_model.py:MyLM
|
|
61
|
+
MyLM(Decoder) my_model.py:5
|
|
62
|
+
vocab_size = <vocab_size>
|
|
63
|
+
context_length = <context_length>
|
|
64
|
+
d_model = 128
|
|
65
|
+
block = Block
|
|
66
|
+
norm = RMSNorm
|
|
67
|
+
...
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Edits go back into the file with `blocks.graph.emit(graph, source)`, which changes only the
|
|
71
|
+
arguments that changed (keeping comments and formatting) and adds a missing import when a new
|
|
72
|
+
block is used. An unedited round trip is byte-identical.
|
|
73
|
+
|
|
74
|
+
## Blocks and their references
|
|
75
|
+
|
|
76
|
+
Every block has a typed constructor, a docstring, `flops_per_token(context_length)` (6 FLOPs per
|
|
77
|
+
weight used, plus the two attention matmuls) and a naive reference in `nanoscope/reference/`
|
|
78
|
+
that its test compares it with. The references import only `torch` and `math`.
|
|
79
|
+
|
|
80
|
+
| Family | Blocks | Reference |
|
|
81
|
+
|---|---|---|
|
|
82
|
+
| embedding | `TokenEmbedding`, `LearnedPosition`, `Head` (tied or untied) | `embed_one_hot`, `add_learned_position`, `tied_head` |
|
|
83
|
+
| positional | `RoPE`, `NoPE` | `naive_rope` (complex rotation; scores depend only on distance) |
|
|
84
|
+
| norm | `LayerNorm`, `RMSNorm` | `layer_norm`, `rms_norm` |
|
|
85
|
+
| attention | `Attention(n_heads, n_kv_heads, pos, qk_norm, window, bias)`: MHA, GQA, MQA and sliding window | `naive_causal_attention` (loops over heads and positions) |
|
|
86
|
+
| mlp | `GELUMLP(hidden, bias)`, `SwiGLU(hidden)` | `gelu`, `swiglu` |
|
|
87
|
+
| structure | `Block(norm, attn, mlp, order)` (pre or post norm), `Decoder` | causality and an untrained loss near `ln(vocab)` |
|
|
88
|
+
| primitive | `Linear`, `Activation`, `CausalMask`, `ScaledDotScores`, `Softmax`, `WeightedSum`, `SplitHeads`, `MergeHeads`, `Residual` | one formula each |
|
|
89
|
+
|
|
90
|
+
`nanoscope blocks` prints the same table with every option, and `nanoscope blocks --json` prints
|
|
91
|
+
it as `blocks.v1`. *Primitives* are the smallest pieces: build your own attention from them to
|
|
92
|
+
see how the bigger blocks work. `Attention.attention_weights(x)` shows the softmax the fused
|
|
93
|
+
kernel never exposes.
|
|
94
|
+
|
|
95
|
+
## Templates
|
|
96
|
+
|
|
97
|
+
A `Composite` is a block made of other blocks in named slots. A lesson template is a
|
|
98
|
+
`Composite` with empty slots for you to fill:
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
from nanoscope.blocks import Composite
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class PreNormAttention(Composite):
|
|
105
|
+
SLOTS = ("norm", "attn")
|
|
106
|
+
|
|
107
|
+
def forward(self, x):
|
|
108
|
+
return x + self.attn(self.norm(x))
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
block = PreNormAttention(norm=RMSNorm(), attn=Attention(n_heads=4))
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
Each slot is a spec, built once per instance. A `Composite` has slots and a `forward`, not an
|
|
115
|
+
`__init__`.
|
|
116
|
+
|
|
117
|
+
## Register your own block
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
from nanoscope.blocks import register_block
|
|
121
|
+
|
|
122
|
+
def naive_gate(x, weight):
|
|
123
|
+
return x * torch.sigmoid(weight)
|
|
124
|
+
|
|
125
|
+
@register_block(reference=naive_gate, family="mlp")
|
|
126
|
+
class Gate(nn.Module):
|
|
127
|
+
"""x times sigmoid of a learned vector."""
|
|
128
|
+
def __init__(self, d_model, context_length, scale=1.0): ...
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
The constructor takes `d_model` and `context_length` first, like every block; the rest are the
|
|
132
|
+
options a model file writes as `Gate(scale=2.0)`. `reference` is a plain function that computes
|
|
133
|
+
the same thing slowly and obviously. A block without one is listed as uncertified.
|
|
134
|
+
`nanoscope blocks --workspace my_folder` finds registered blocks by reading the files.
|
|
135
|
+
|
|
136
|
+
## Inspect
|
|
137
|
+
|
|
138
|
+
| Command | Shows |
|
|
139
|
+
|---|---|
|
|
140
|
+
| `nanoscope graph file.py[:Class] [--json]` | the architecture graph, without running the file |
|
|
141
|
+
| `nanoscope describe file.py:Class [--preset p] [--json] [--set k=v ...]` | shapes, parameters, FLOPs per token and memory per module, traced on the `meta` device (it imports your file). A shape error names the module and the line. |
|
|
142
|
+
| `nanoscope blocks [--json] [--workspace dir]` | the palette |
|
|
143
|
+
| `run(Model, block_stats=True)` | at each eval step, per block: activation size, gradient norm, update-to-weight ratio and attention entropy, appended to `blockstats.jsonl` |
|
|
144
|
+
| `nanoscope status --blocks <run ref>` | the latest block statistics of a run |
|