@genai-fi/nanogpt 0.20.0 → 0.20.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/BaseTokeniser-DSg9zcYq.js +221 -0
- package/dist/DatasetBuilder-DgURD85T.js +712 -0
- package/dist/Generator.d.ts +82 -0
- package/dist/Generator.js +2 -0
- package/dist/RealDiv-DBu0FQqT.js +362 -0
- package/dist/Reshape-CABOPB9d.js +94 -0
- package/dist/Reshape-DqO3r8BC.js +17 -0
- package/dist/TeachableLLM.d.ts +70 -0
- package/dist/TeachableLLM.js +2 -0
- package/dist/Trainer.d.ts +43 -0
- package/dist/Trainer.js +2 -0
- package/dist/backend.d.ts +2 -0
- package/dist/backend.js +13 -0
- package/dist/backend_util-Cg-roD1p.js +399 -0
- package/dist/binary_op_util-CrYk9LXL.js +103 -0
- package/dist/checks/appendCache.d.ts +1 -0
- package/dist/checks/appendCache.js +55 -0
- package/dist/checks/attentionMask.d.ts +1 -0
- package/dist/checks/attentionMask.js +56 -0
- package/dist/checks/check.d.ts +9 -0
- package/dist/checks/check.js +32 -0
- package/dist/checks/gelu.d.ts +1 -0
- package/dist/checks/gelu.js +46 -0
- package/dist/checks/index.d.ts +26 -0
- package/dist/checks/index.js +28 -0
- package/dist/checks/matMulGelu.d.ts +1 -0
- package/dist/checks/matMulGelu.js +84 -0
- package/dist/checks/normRMS.d.ts +1 -0
- package/dist/checks/normRMS.js +28 -0
- package/dist/checks/normRMSGrad.d.ts +1 -0
- package/dist/checks/normRMSGrad.js +22 -0
- package/dist/checks/packUnpack.d.ts +1 -0
- package/dist/checks/packUnpack.js +46 -0
- package/dist/checks/qkv.d.ts +1 -0
- package/dist/checks/qkv.js +34 -0
- package/dist/checks/rope.d.ts +1 -0
- package/dist/checks/rope.js +30 -0
- package/dist/checks/weights.d.ts +14 -0
- package/dist/checks/weights.js +27 -0
- package/dist/chunk-BPntVaq0.js +23 -0
- package/dist/complex_util-CkazZsaH.js +60 -0
- package/dist/concat_util-CWDZCBlA.js +19 -0
- package/dist/data/docx.d.ts +2 -0
- package/dist/data/docx.js +3046 -0
- package/dist/data/pdf.d.ts +2 -0
- package/dist/data/pdf.js +17 -0
- package/dist/data/textLoader.d.ts +7 -0
- package/dist/data/textLoader.js +613 -0
- package/dist/dist-BewPQWjc.js +7572 -0
- package/dist/dist-DVmq73nz.js +8775 -0
- package/dist/dist-DXwIvKxl.js +896 -0
- package/dist/dist-VEU5mfO0.js +7545 -0
- package/dist/gelu-Bf1HW1RY.js +27 -0
- package/dist/gpgpu_math-DvLcCH6u.js +1612 -0
- package/dist/inference/types.d.ts +16 -0
- package/dist/inference/types.js +0 -0
- package/dist/kernel_funcs_utils-HiXOOx3f.js +229 -0
- package/dist/layers/BaseLayer.d.ts +44 -0
- package/dist/layers/BaseLayer.js +76 -0
- package/dist/layers/CausalSelfAttention.d.ts +39 -0
- package/dist/layers/CausalSelfAttention.js +99 -0
- package/dist/layers/LoRA.d.ts +14 -0
- package/dist/layers/LoRA.js +48 -0
- package/dist/layers/MLP.d.ts +17 -0
- package/dist/layers/MLP.js +34 -0
- package/dist/layers/PositionEmbedding.d.ts +8 -0
- package/dist/layers/PositionEmbedding.js +27 -0
- package/dist/layers/RMSNorm.d.ts +12 -0
- package/dist/layers/RMSNorm.js +20 -0
- package/dist/layers/RoPECache.d.ts +18 -0
- package/dist/layers/RoPECache.js +337 -0
- package/dist/layers/TiedEmbedding.d.ts +13 -0
- package/dist/layers/TiedEmbedding.js +32 -0
- package/dist/layers/TransformerBlock.d.ts +27 -0
- package/dist/layers/TransformerBlock.js +51 -0
- package/dist/layers/WeightStore.d.ts +20 -0
- package/dist/layers/WeightStore.js +69 -0
- package/dist/loader/load.d.ts +6 -0
- package/dist/loader/load.js +2 -0
- package/dist/loader/loadHF.d.ts +8 -0
- package/dist/loader/loadHF.js +2 -0
- package/dist/loader/loadTransformers.d.ts +4 -0
- package/dist/loader/loadTransformers.js +2 -0
- package/dist/loader/loadZipMeta.d.ts +3 -0
- package/dist/loader/loadZipMeta.js +16 -0
- package/dist/loader/newZipLoad.d.ts +3 -0
- package/dist/loader/newZipLoad.js +2 -0
- package/dist/loader/oldZipLoad.d.ts +9 -0
- package/dist/loader/oldZipLoad.js +2 -0
- package/dist/loader/save.d.ts +16 -0
- package/dist/loader/save.js +2 -0
- package/dist/loader/types.d.ts +68 -0
- package/dist/loader/types.js +0 -0
- package/dist/main-D5CbfCiV.js +13500 -0
- package/dist/main.d.ts +50 -0
- package/dist/main.js +16 -0
- package/dist/matMul16-BNfZSnNM.js +81 -0
- package/dist/matMulGelu-CPTntosE.js +162 -0
- package/dist/models/NanoGPTV1.d.ts +16 -0
- package/dist/models/NanoGPTV1.js +2 -0
- package/dist/models/NanoGPTV2.d.ts +16 -0
- package/dist/models/NanoGPTV2.js +2 -0
- package/dist/models/config.d.ts +27 -0
- package/dist/models/config.js +37 -0
- package/dist/models/factory.d.ts +3 -0
- package/dist/models/factory.js +2 -0
- package/dist/models/model.d.ts +44 -0
- package/dist/models/model.js +2 -0
- package/dist/ops/adamAdjust.d.ts +2 -0
- package/dist/ops/adamAdjust.js +18 -0
- package/dist/ops/adamMoments.d.ts +2 -0
- package/dist/ops/adamMoments.js +16 -0
- package/dist/ops/add16.d.ts +2 -0
- package/dist/ops/add16.js +12 -0
- package/dist/ops/appendCache.d.ts +2 -0
- package/dist/ops/appendCache.js +25 -0
- package/dist/ops/attentionMask.d.ts +2 -0
- package/dist/ops/attentionMask.js +16 -0
- package/dist/ops/concat16.d.ts +2 -0
- package/dist/ops/concat16.js +8 -0
- package/dist/ops/cpu/adamAdjust.d.ts +1 -0
- package/dist/ops/cpu/adamAdjust.js +16 -0
- package/dist/ops/cpu/adamMoments.d.ts +1 -0
- package/dist/ops/cpu/adamMoments.js +16 -0
- package/dist/ops/cpu/appendCache.d.ts +1 -0
- package/dist/ops/cpu/appendCache.js +65 -0
- package/dist/ops/cpu/attentionMask.d.ts +1 -0
- package/dist/ops/cpu/attentionMask.js +16 -0
- package/dist/ops/cpu/fusedSoftmax.d.ts +9 -0
- package/dist/ops/cpu/fusedSoftmax.js +22 -0
- package/dist/ops/cpu/gatherSub.d.ts +1 -0
- package/dist/ops/cpu/gatherSub.js +12 -0
- package/dist/ops/cpu/gelu.d.ts +1 -0
- package/dist/ops/cpu/gelu.js +36 -0
- package/dist/ops/cpu/matMul16.d.ts +1 -0
- package/dist/ops/cpu/matMul16.js +14 -0
- package/dist/ops/cpu/matMulGelu.d.ts +1 -0
- package/dist/ops/cpu/matMulGelu.js +41 -0
- package/dist/ops/cpu/matMulMul.d.ts +1 -0
- package/dist/ops/cpu/matMulMul.js +20 -0
- package/dist/ops/cpu/mulDropout.d.ts +1 -0
- package/dist/ops/cpu/mulDropout.js +20 -0
- package/dist/ops/cpu/normRMS.d.ts +1 -0
- package/dist/ops/cpu/normRMS.js +35 -0
- package/dist/ops/cpu/qkv.d.ts +5 -0
- package/dist/ops/cpu/qkv.js +73 -0
- package/dist/ops/cpu/rope.d.ts +6 -0
- package/dist/ops/cpu/rope.js +81 -0
- package/dist/ops/cpu/scatterSub.d.ts +1 -0
- package/dist/ops/cpu/scatterSub.js +12 -0
- package/dist/ops/dot16.d.ts +2 -0
- package/dist/ops/dot16.js +29 -0
- package/dist/ops/dropout.d.ts +2 -0
- package/dist/ops/dropout.js +11 -0
- package/dist/ops/dropout16.d.ts +2 -0
- package/dist/ops/dropout16.js +22 -0
- package/dist/ops/gatherSub.d.ts +2 -0
- package/dist/ops/gatherSub.js +13 -0
- package/dist/ops/gelu.d.ts +3 -0
- package/dist/ops/gelu.js +2 -0
- package/dist/ops/globalNorm.d.ts +2 -0
- package/dist/ops/globalNorm.js +19 -0
- package/dist/ops/grads/add16.d.ts +1 -0
- package/dist/ops/grads/add16.js +27 -0
- package/dist/ops/grads/attentionMask.d.ts +1 -0
- package/dist/ops/grads/attentionMask.js +26 -0
- package/dist/ops/grads/dropout16.d.ts +1 -0
- package/dist/ops/grads/dropout16.js +1 -0
- package/dist/ops/grads/gelu.d.ts +2 -0
- package/dist/ops/grads/gelu.js +2 -0
- package/dist/ops/grads/matMul16.d.ts +2 -0
- package/dist/ops/grads/matMul16.js +2 -0
- package/dist/ops/grads/matMulGelu.d.ts +1 -0
- package/dist/ops/grads/matMulGelu.js +22 -0
- package/dist/ops/grads/mul16.d.ts +1 -0
- package/dist/ops/grads/mul16.js +1 -0
- package/dist/ops/grads/normRMS.d.ts +3 -0
- package/dist/ops/grads/normRMS.js +37 -0
- package/dist/ops/grads/pack16.d.ts +2 -0
- package/dist/ops/grads/pack16.js +2 -0
- package/dist/ops/grads/qkv.d.ts +3 -0
- package/dist/ops/grads/qkv.js +46 -0
- package/dist/ops/grads/rope.d.ts +2 -0
- package/dist/ops/grads/rope.js +2 -0
- package/dist/ops/grads/softmax16.d.ts +2 -0
- package/dist/ops/grads/softmax16.js +23 -0
- package/dist/ops/grads/unpack16.d.ts +2 -0
- package/dist/ops/grads/unpack16.js +2 -0
- package/dist/ops/grads/utils.d.ts +4 -0
- package/dist/ops/grads/utils.js +12 -0
- package/dist/ops/log.d.ts +0 -0
- package/dist/ops/log.js +1 -0
- package/dist/ops/matMul16.d.ts +15 -0
- package/dist/ops/matMul16.js +2 -0
- package/dist/ops/matMulGelu.d.ts +3 -0
- package/dist/ops/matMulGelu.js +20 -0
- package/dist/ops/matMulMul.d.ts +2 -0
- package/dist/ops/matMulMul.js +16 -0
- package/dist/ops/mul16.d.ts +2 -0
- package/dist/ops/mul16.js +43 -0
- package/dist/ops/mulDrop.d.ts +2 -0
- package/dist/ops/mulDrop.js +15 -0
- package/dist/ops/normRMS.d.ts +2 -0
- package/dist/ops/normRMS.js +22 -0
- package/dist/ops/pack16.d.ts +2 -0
- package/dist/ops/pack16.js +2 -0
- package/dist/ops/qkv.d.ts +2 -0
- package/dist/ops/qkv.js +16 -0
- package/dist/ops/reshape16.d.ts +2 -0
- package/dist/ops/reshape16.js +33 -0
- package/dist/ops/rope.d.ts +3 -0
- package/dist/ops/rope.js +2 -0
- package/dist/ops/scatterSub.d.ts +2 -0
- package/dist/ops/scatterSub.js +13 -0
- package/dist/ops/slice16.d.ts +2 -0
- package/dist/ops/slice16.js +11 -0
- package/dist/ops/softmax16.d.ts +2 -0
- package/dist/ops/softmax16.js +9 -0
- package/dist/ops/sub16.d.ts +2 -0
- package/dist/ops/sub16.js +11 -0
- package/dist/ops/sum16.d.ts +2 -0
- package/dist/ops/sum16.js +13 -0
- package/dist/ops/transpose16.d.ts +3 -0
- package/dist/ops/transpose16.js +32 -0
- package/dist/ops/unpack16.d.ts +2 -0
- package/dist/ops/unpack16.js +2 -0
- package/dist/ops/webgl/adamAdjust.d.ts +1 -0
- package/dist/ops/webgl/adamAdjust.js +82 -0
- package/dist/ops/webgl/adamMoments.d.ts +1 -0
- package/dist/ops/webgl/adamMoments.js +44 -0
- package/dist/ops/webgl/appendCache.d.ts +1 -0
- package/dist/ops/webgl/appendCache.js +53 -0
- package/dist/ops/webgl/attentionMask.d.ts +1 -0
- package/dist/ops/webgl/attentionMask.js +64 -0
- package/dist/ops/webgl/dropout16.d.ts +1 -0
- package/dist/ops/webgl/dropout16.js +12 -0
- package/dist/ops/webgl/fusedSoftmax.d.ts +11 -0
- package/dist/ops/webgl/fusedSoftmax.js +70 -0
- package/dist/ops/webgl/gatherSub.d.ts +1 -0
- package/dist/ops/webgl/gatherSub.js +28 -0
- package/dist/ops/webgl/gelu.d.ts +2 -0
- package/dist/ops/webgl/gelu.js +48 -0
- package/dist/ops/webgl/log.d.ts +17 -0
- package/dist/ops/webgl/log.js +14 -0
- package/dist/ops/webgl/matMul16.d.ts +1 -0
- package/dist/ops/webgl/matMul16.js +37 -0
- package/dist/ops/webgl/matMulGelu.d.ts +21 -0
- package/dist/ops/webgl/matMulGelu.js +2 -0
- package/dist/ops/webgl/matMulMul.d.ts +14 -0
- package/dist/ops/webgl/matMulMul.js +24 -0
- package/dist/ops/webgl/mulDropout.d.ts +1 -0
- package/dist/ops/webgl/mulDropout.js +32 -0
- package/dist/ops/webgl/normRMS.d.ts +1 -0
- package/dist/ops/webgl/normRMS.js +114 -0
- package/dist/ops/webgl/qkv.d.ts +1 -0
- package/dist/ops/webgl/qkv.js +54 -0
- package/dist/ops/webgl/rope.d.ts +1 -0
- package/dist/ops/webgl/rope.js +72 -0
- package/dist/ops/webgl/scatterSub.d.ts +1 -0
- package/dist/ops/webgl/scatterSub.js +28 -0
- package/dist/ops/webgpu/adamAdjust.d.ts +1 -0
- package/dist/ops/webgpu/adamAdjust.js +77 -0
- package/dist/ops/webgpu/adamMoments.d.ts +1 -0
- package/dist/ops/webgpu/adamMoments.js +76 -0
- package/dist/ops/webgpu/add16.d.ts +1 -0
- package/dist/ops/webgpu/add16.js +14 -0
- package/dist/ops/webgpu/appendCache.d.ts +1 -0
- package/dist/ops/webgpu/appendCache.js +130 -0
- package/dist/ops/webgpu/attentionMask.d.ts +1 -0
- package/dist/ops/webgpu/attentionMask.js +42 -0
- package/dist/ops/webgpu/attentionMask32_program.d.ts +19 -0
- package/dist/ops/webgpu/attentionMask32_program.js +62 -0
- package/dist/ops/webgpu/clipScale.d.ts +1 -0
- package/dist/ops/webgpu/clipScale.js +45 -0
- package/dist/ops/webgpu/concat16.d.ts +19 -0
- package/dist/ops/webgpu/concat16.js +111 -0
- package/dist/ops/webgpu/dropout16.d.ts +1 -0
- package/dist/ops/webgpu/dropout16.js +59 -0
- package/dist/ops/webgpu/gatherSub.d.ts +1 -0
- package/dist/ops/webgpu/gatherSub.js +52 -0
- package/dist/ops/webgpu/gelu.d.ts +14 -0
- package/dist/ops/webgpu/gelu.js +147 -0
- package/dist/ops/webgpu/index.d.ts +0 -0
- package/dist/ops/webgpu/index.js +26 -0
- package/dist/ops/webgpu/matMul16.d.ts +1 -0
- package/dist/ops/webgpu/matMul16.js +70 -0
- package/dist/ops/webgpu/matMul16_program.d.ts +42 -0
- package/dist/ops/webgpu/matMul16_program.js +303 -0
- package/dist/ops/webgpu/mul16.d.ts +1 -0
- package/dist/ops/webgpu/mul16.js +14 -0
- package/dist/ops/webgpu/norm2.d.ts +1 -0
- package/dist/ops/webgpu/norm2.js +46 -0
- package/dist/ops/webgpu/normRMS.d.ts +1 -0
- package/dist/ops/webgpu/normRMS.js +26 -0
- package/dist/ops/webgpu/normRMS16_program.d.ts +10 -0
- package/dist/ops/webgpu/normRMS16_program.js +28 -0
- package/dist/ops/webgpu/normRMS32_program.d.ts +10 -0
- package/dist/ops/webgpu/normRMS32_program.js +28 -0
- package/dist/ops/webgpu/normRMSGrad.d.ts +1 -0
- package/dist/ops/webgpu/normRMSGrad.js +225 -0
- package/dist/ops/webgpu/pack16.d.ts +1 -0
- package/dist/ops/webgpu/pack16.js +21 -0
- package/dist/ops/webgpu/pack16_program.d.ts +19 -0
- package/dist/ops/webgpu/pack16_program.js +93 -0
- package/dist/ops/webgpu/qkv.d.ts +1 -0
- package/dist/ops/webgpu/qkv.js +64 -0
- package/dist/ops/webgpu/rope.d.ts +1 -0
- package/dist/ops/webgpu/rope.js +163 -0
- package/dist/ops/webgpu/scatterSub.d.ts +1 -0
- package/dist/ops/webgpu/scatterSub.js +53 -0
- package/dist/ops/webgpu/slice16.d.ts +7 -0
- package/dist/ops/webgpu/slice16.js +74 -0
- package/dist/ops/webgpu/softmax16.d.ts +17 -0
- package/dist/ops/webgpu/softmax16.js +18 -0
- package/dist/ops/webgpu/softmax16_program.d.ts +13 -0
- package/dist/ops/webgpu/softmax16_program.js +89 -0
- package/dist/ops/webgpu/softmax16_subgroup_program.d.ts +17 -0
- package/dist/ops/webgpu/softmax16_subgroup_program.js +70 -0
- package/dist/ops/webgpu/softmax16grad.d.ts +1 -0
- package/dist/ops/webgpu/softmax16grad.js +31 -0
- package/dist/ops/webgpu/sub16.d.ts +1 -0
- package/dist/ops/webgpu/sub16.js +14 -0
- package/dist/ops/webgpu/sum16.d.ts +1 -0
- package/dist/ops/webgpu/sum16.js +29 -0
- package/dist/ops/webgpu/transpose16.d.ts +1 -0
- package/dist/ops/webgpu/transpose16.js +37 -0
- package/dist/ops/webgpu/transpose16_program.d.ts +16 -0
- package/dist/ops/webgpu/transpose16_program.js +51 -0
- package/dist/ops/webgpu/transpose16_shared_program.d.ts +15 -0
- package/dist/ops/webgpu/transpose16_shared_program.js +79 -0
- package/dist/ops/webgpu/unpack16.d.ts +1 -0
- package/dist/ops/webgpu/unpack16.js +60 -0
- package/dist/ops/webgpu/utils/binary_op.d.ts +35 -0
- package/dist/ops/webgpu/utils/binary_op.js +141 -0
- package/dist/ops/webgpu/utils/deviceInfo.d.ts +7 -0
- package/dist/ops/webgpu/utils/deviceInfo.js +11 -0
- package/dist/ops/webgpu/utils/reductions.d.ts +43 -0
- package/dist/ops/webgpu/utils/reductions.js +263 -0
- package/dist/pack16-Ck-spx_F.js +39 -0
- package/dist/patches/webgpu_backend.d.ts +18 -0
- package/dist/patches/webgpu_backend.js +43 -0
- package/dist/patches/webgpu_base.d.ts +21 -0
- package/dist/patches/webgpu_base.js +22 -0
- package/dist/patches/webgpu_program.d.ts +36 -0
- package/dist/patches/webgpu_program.js +293 -0
- package/dist/pdf-UoDqCYzz.js +16726 -0
- package/dist/picomatch-3tUnMMbd.js +1063 -0
- package/dist/rope-CbeGlsV8.js +25 -0
- package/dist/selu_util-zkAx5doH.js +24 -0
- package/dist/shared-D1coEFea.js +1314 -0
- package/dist/shared-DOgWaqvL.js +5 -0
- package/dist/slice_util-Dgb3ANWI.js +208 -0
- package/dist/tfjs_backend-BjuQ5FqB.js +614 -0
- package/dist/tokeniser/BaseTokeniser.d.ts +33 -0
- package/dist/tokeniser/BaseTokeniser.js +2 -0
- package/dist/tokeniser/CharTokeniser.d.ts +24 -0
- package/dist/tokeniser/CharTokeniser.js +92 -0
- package/dist/tokeniser/bpe.d.ts +28 -0
- package/dist/tokeniser/bpe.js +170 -0
- package/dist/tokeniser/messages.d.ts +61 -0
- package/dist/tokeniser/messages.js +0 -0
- package/dist/tokeniser/type.d.ts +34 -0
- package/dist/tokeniser/type.js +0 -0
- package/dist/training/AdamW.d.ts +36 -0
- package/dist/training/AdamW.js +128 -0
- package/dist/training/BasicTrainer.d.ts +63 -0
- package/dist/training/BasicTrainer.js +265 -0
- package/dist/training/DatasetBuilder.d.ts +26 -0
- package/dist/training/DatasetBuilder.js +2 -0
- package/dist/training/Evaluator.d.ts +19 -0
- package/dist/training/Evaluator.js +48 -0
- package/dist/training/LRScheduler.d.ts +12 -0
- package/dist/training/LRScheduler.js +38 -0
- package/dist/training/PreTrainer.d.ts +11 -0
- package/dist/training/PreTrainer.js +22 -0
- package/dist/training/SFTTrainer.d.ts +12 -0
- package/dist/training/SFTTrainer.js +24 -0
- package/dist/training/loss.d.ts +3 -0
- package/dist/training/loss.js +19 -0
- package/dist/training/orthoGrad.d.ts +2 -0
- package/dist/training/orthoGrad.js +10 -0
- package/dist/training/sparseCrossEntropy.d.ts +7 -0
- package/dist/training/sparseCrossEntropy.js +47 -0
- package/dist/training/tasks/ConversationTask.d.ts +18 -0
- package/dist/training/tasks/ConversationTask.js +38 -0
- package/dist/training/tasks/PretrainingTask.d.ts +17 -0
- package/dist/training/tasks/PretrainingTask.js +42 -0
- package/dist/training/tasks/StartSentenceTask.d.ts +18 -0
- package/dist/training/tasks/StartSentenceTask.js +45 -0
- package/dist/training/tasks/Task.d.ts +22 -0
- package/dist/training/tasks/Task.js +55 -0
- package/dist/training/tasks/splitter.d.ts +5 -0
- package/dist/training/tasks/splitter.js +18 -0
- package/dist/training/types.d.ts +78 -0
- package/dist/training/types.js +0 -0
- package/dist/training/validation.d.ts +17 -0
- package/dist/training/validation.js +2 -0
- package/dist/utilities/arrayClose.d.ts +1 -0
- package/dist/utilities/arrayClose.js +16 -0
- package/dist/utilities/datasetID.d.ts +2 -0
- package/dist/utilities/datasetID.js +18 -0
- package/dist/utilities/dummy.d.ts +9 -0
- package/dist/utilities/dummy.js +36 -0
- package/dist/utilities/multinomialCPU.d.ts +2 -0
- package/dist/utilities/multinomialCPU.js +9 -0
- package/dist/utilities/naming.d.ts +4 -0
- package/dist/utilities/naming.js +0 -0
- package/dist/utilities/packed.d.ts +4 -0
- package/dist/utilities/packed.js +13 -0
- package/dist/utilities/parameters.d.ts +11 -0
- package/dist/utilities/parameters.js +38 -0
- package/dist/utilities/performance.d.ts +2 -0
- package/dist/utilities/performance.js +16 -0
- package/dist/utilities/profile.d.ts +17 -0
- package/dist/utilities/profile.js +33 -0
- package/dist/utilities/safetensors.d.ts +3 -0
- package/dist/utilities/safetensors.js +53 -0
- package/dist/utilities/sentences.d.ts +5 -0
- package/dist/utilities/sentences.js +32 -0
- package/dist/utilities/tokenParse.d.ts +1 -0
- package/dist/utilities/tokenParse.js +17 -0
- package/dist/utilities/topP.d.ts +1 -0
- package/dist/utilities/topP.js +12 -0
- package/dist/utilities/waitForModel.d.ts +2 -0
- package/dist/utilities/waitForModel.js +12 -0
- package/dist/utilities/weights.d.ts +12 -0
- package/dist/utilities/weights.js +40 -0
- package/dist/utilities/yielder.d.ts +1 -0
- package/dist/utilities/yielder.js +7 -0
- package/dist/webgpu-Dt7BMzWz.js +525 -0
- package/dist/webgpu_program-WOyIVMlZ.js +392 -0
- package/dist/webgpu_util-B_F3SShA.js +106 -0
- package/package.json +1 -1
package/dist/main.d.ts
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
import { default as PretrainingTask } from './training/tasks/PretrainingTask';
|
|
2
|
+
import { default as StartSentenceTask } from './training/tasks/StartSentenceTask';
|
|
3
|
+
import { default as ConversationTask } from './training/tasks/ConversationTask';
|
|
4
|
+
import { pack16 } from './ops/pack16';
|
|
5
|
+
import { unpack16 } from './ops/unpack16';
|
|
6
|
+
import { default as CausalSelfAttention } from './layers/CausalSelfAttention';
|
|
7
|
+
import { default as MLP } from './layers/MLP';
|
|
8
|
+
import { default as TransformerBlock } from './layers/TransformerBlock';
|
|
9
|
+
import { default as RoPECache } from './layers/RoPECache';
|
|
10
|
+
export { default as NanoGPT } from './models/NanoGPTV1';
|
|
11
|
+
export { default as TeachableLLM } from './TeachableLLM';
|
|
12
|
+
export { default as CharTokeniser } from './tokeniser/CharTokeniser';
|
|
13
|
+
export { default as BPETokeniser } from './tokeniser/bpe';
|
|
14
|
+
export { default as waitForModel } from './utilities/waitForModel';
|
|
15
|
+
export { default as generateDatasetID } from './utilities/datasetID';
|
|
16
|
+
export { default as loadTextData } from './data/textLoader';
|
|
17
|
+
export type { DatasetMetadata, ModelMode } from './loader/types';
|
|
18
|
+
export { default as Generator, type IGenerator } from './Generator';
|
|
19
|
+
export { default as Evaluator } from './training/Evaluator';
|
|
20
|
+
export { default as Trainer } from './Trainer';
|
|
21
|
+
export type { IGenerateOptions } from './Generator';
|
|
22
|
+
export { type ModelForwardAttributes, default as Model } from './models/model';
|
|
23
|
+
export type { ITokeniser, Conversation, Roles } from './tokeniser/type';
|
|
24
|
+
export type { TrainingOptions, TrainingLogEntry } from './training/types';
|
|
25
|
+
export type { GPTConfig } from './models/config';
|
|
26
|
+
export { estimateParameterCount, estimateMemoryUsage, estimateTrainingMemoryUsage, estimateResources, validateConfig, } from './utilities/parameters';
|
|
27
|
+
export { default as topP } from './utilities/topP';
|
|
28
|
+
export { Task, tokensFromTasks } from './training/tasks/Task';
|
|
29
|
+
export declare const tasks: {
|
|
30
|
+
PretrainingTask: typeof PretrainingTask;
|
|
31
|
+
StartSentenceTask: typeof StartSentenceTask;
|
|
32
|
+
ConversationTask: typeof ConversationTask;
|
|
33
|
+
};
|
|
34
|
+
declare const ops: {
|
|
35
|
+
pack16: typeof pack16;
|
|
36
|
+
unpack16: typeof unpack16;
|
|
37
|
+
};
|
|
38
|
+
export { ops };
|
|
39
|
+
export { selectBackend } from './backend';
|
|
40
|
+
export { default as performanceTest } from './utilities/performance';
|
|
41
|
+
export declare const layers: {
|
|
42
|
+
CausalSelfAttention: typeof CausalSelfAttention;
|
|
43
|
+
MLP: typeof MLP;
|
|
44
|
+
TransformerBlock: typeof TransformerBlock;
|
|
45
|
+
RoPECache: typeof RoPECache;
|
|
46
|
+
};
|
|
47
|
+
export { AdamWOptimizer } from './training/AdamW';
|
|
48
|
+
export { default as checks } from './checks';
|
|
49
|
+
export type { TensorStatistics } from './checks/weights';
|
|
50
|
+
export { sentenceEmbeddings, sentenceEmbeddingsTensor } from './utilities/sentences';
|
package/dist/main.js
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { a as e, b as t, i as n, n as r, r as i, s as a, t as o, y as s } from "./main-D5CbfCiV.js";
|
|
2
|
+
import c from "./tokeniser/CharTokeniser.js";
|
|
3
|
+
import l from "./tokeniser/bpe.js";
|
|
4
|
+
import { AdamWOptimizer as u } from "./training/AdamW.js";
|
|
5
|
+
import d from "./utilities/topP.js";
|
|
6
|
+
import f from "./training/Evaluator.js";
|
|
7
|
+
import p from "./utilities/waitForModel.js";
|
|
8
|
+
import m from "./utilities/datasetID.js";
|
|
9
|
+
import h from "./data/textLoader.js";
|
|
10
|
+
import { estimateMemoryUsage as g, estimateParameterCount as _, estimateResources as v, estimateTrainingMemoryUsage as y, validateConfig as b } from "./utilities/parameters.js";
|
|
11
|
+
import { Task as x, tokensFromTasks as S } from "./training/tasks/Task.js";
|
|
12
|
+
import { selectBackend as C } from "./backend.js";
|
|
13
|
+
import w from "./utilities/performance.js";
|
|
14
|
+
import T from "./checks/index.js";
|
|
15
|
+
import { sentenceEmbeddings as E, sentenceEmbeddingsTensor as D } from "./utilities/sentences.js";
|
|
16
|
+
export { u as AdamWOptimizer, l as BPETokeniser, c as CharTokeniser, f as Evaluator, a as Generator, t as Model, s as NanoGPT, x as Task, n as TeachableLLM, e as Trainer, T as checks, g as estimateMemoryUsage, _ as estimateParameterCount, v as estimateResources, y as estimateTrainingMemoryUsage, m as generateDatasetID, o as layers, h as loadTextData, r as ops, w as performanceTest, C as selectBackend, E as sentenceEmbeddings, D as sentenceEmbeddingsTensor, i as tasks, S as tokensFromTasks, d as topP, b as validateConfig, p as waitForModel };
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import { Fi as e, ii as t } from "./dist-BewPQWjc.js";
|
|
2
|
+
import { t as n } from "./gelu-Bf1HW1RY.js";
|
|
3
|
+
import { isPackedTensor as r } from "./utilities/packed.js";
|
|
4
|
+
import { transpose16 as i } from "./ops/transpose16.js";
|
|
5
|
+
import { reshape16 as a } from "./ops/reshape16.js";
|
|
6
|
+
import { mul16 as o } from "./ops/mul16.js";
|
|
7
|
+
import "./ops/webgl/matMul16.js";
|
|
8
|
+
import "./ops/cpu/matMul16.js";
|
|
9
|
+
import { t as s } from "./pack16-Ck-spx_F.js";
|
|
10
|
+
//#region lib/ops/grads/matMul16.ts
|
|
11
|
+
var c = {
|
|
12
|
+
kernelName: "MatMul16",
|
|
13
|
+
inputsToSave: ["A", "B"],
|
|
14
|
+
outputsToSave: [],
|
|
15
|
+
gradFunc: (e, t, r) => {
|
|
16
|
+
let [s, c] = t;
|
|
17
|
+
if (Array.isArray(e)) throw Error("Expected dy to be a single Tensor");
|
|
18
|
+
let u = e, { transposeA: p, transposeB: m, scale: h, activation: g, originalShape: _, perm: v } = r;
|
|
19
|
+
if (v && _) {
|
|
20
|
+
let e = Array(v.length);
|
|
21
|
+
for (let t = 0; t < v.length; ++t) e[v[t]] = t;
|
|
22
|
+
let t = u;
|
|
23
|
+
u = i(u, e), t.dispose();
|
|
24
|
+
}
|
|
25
|
+
if (_) {
|
|
26
|
+
let e = u;
|
|
27
|
+
u = a(u, _), e.dispose();
|
|
28
|
+
}
|
|
29
|
+
if (g === "gelu") {
|
|
30
|
+
let e = u, t = l(s, c, p, m);
|
|
31
|
+
u = n(e, t), e.dispose(), t.dispose();
|
|
32
|
+
} else if (g === "relu2") {
|
|
33
|
+
let e = u, t = l(s, c, p, m, {
|
|
34
|
+
activation: "relu",
|
|
35
|
+
scale: 2
|
|
36
|
+
});
|
|
37
|
+
u = o(e, t), e.dispose(), t.dispose();
|
|
38
|
+
}
|
|
39
|
+
if (!p && !m) return {
|
|
40
|
+
A: () => h === void 0 ? l(u, c, !1, !0) : d(u, c, h, !1, !0),
|
|
41
|
+
B: () => h === void 0 ? l(s, u, !0, !1) : f(s, u, h, !0, !1)
|
|
42
|
+
};
|
|
43
|
+
if (!p && m) return {
|
|
44
|
+
A: () => h === void 0 ? l(u, c, !1, !1) : d(u, c, h, !1, !1),
|
|
45
|
+
B: () => h === void 0 ? l(s, u, !0, !1) : f(s, u, h, !0, !1)
|
|
46
|
+
};
|
|
47
|
+
if (p && !m) return {
|
|
48
|
+
A: () => h === void 0 ? l(c, u, !1, !0) : f(c, u, h, !1, !0),
|
|
49
|
+
B: () => h === void 0 ? l(s, u, !1, !1) : f(s, u, h, !1, !1)
|
|
50
|
+
};
|
|
51
|
+
throw Error("Gradient for transposeA=true and transposeB=true is not supported yet.");
|
|
52
|
+
}
|
|
53
|
+
};
|
|
54
|
+
e(c);
|
|
55
|
+
//#endregion
|
|
56
|
+
//#region lib/ops/matMul16.ts
|
|
57
|
+
function l(e, n, i = !1, a = !1, o = {}) {
|
|
58
|
+
let c = r(e), l = r(n), u = c || l, d = !u || c ? e : s(e), f = !u || l ? n : s(n), p = t().runKernel("MatMul16", {
|
|
59
|
+
A: d,
|
|
60
|
+
B: f
|
|
61
|
+
}, {
|
|
62
|
+
transposeA: i,
|
|
63
|
+
transposeB: a,
|
|
64
|
+
...o
|
|
65
|
+
});
|
|
66
|
+
return u && !c && d.dispose(), u && !l && f.dispose(), p;
|
|
67
|
+
}
|
|
68
|
+
function u(e, t, n, r = !1, i = !1) {
|
|
69
|
+
return l(e, t, r, i, { scale: n });
|
|
70
|
+
}
|
|
71
|
+
function d(e, t, n, r = !1, i = !1) {
|
|
72
|
+
return l(e, t, r, i, { scaleA: n });
|
|
73
|
+
}
|
|
74
|
+
function f(e, t, n, r = !1, i = !1) {
|
|
75
|
+
return l(e, t, r, i, { scaleB: n });
|
|
76
|
+
}
|
|
77
|
+
function p(e, t, n = !1, r = !1) {
|
|
78
|
+
return l(e, t, n, r, { activation: "gelu" });
|
|
79
|
+
}
|
|
80
|
+
//#endregion
|
|
81
|
+
export { u as a, f as i, p as n, c as o, d as r, l as t };
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
import { Ci as e, Ii as t, In as n, Ps as r, di as i, gr as a, ii as o, oc as s } from "./dist-BewPQWjc.js";
|
|
2
|
+
import { a as c } from "./gpgpu_math-DvLcCH6u.js";
|
|
3
|
+
import { t as l } from "./Reshape-CABOPB9d.js";
|
|
4
|
+
//#region node_modules/@tensorflow/tfjs-backend-webgl/dist/mulmat_packed_gpu.js
|
|
5
|
+
var u = class {
|
|
6
|
+
constructor(e, t, n, r = !1, i = !1, a = !1, o = null, s = !1, l = !1) {
|
|
7
|
+
this.variableNames = ["matrixA", "matrixB"], this.packedInputs = !0, this.packedOutput = !0, this.outputShape = n, this.enableShapeUniforms = c(this.outputShape.length);
|
|
8
|
+
let u = r ? e[1] : e[2], d = Math.ceil(u / 2), f = r ? "i * 2, rc.y" : "rc.y, i * 2", p = i ? "rc.z, i * 2" : "i * 2, rc.z", m = r ? ["a.xxyy", "a.zzww"] : ["a.xxzz", "a.yyww"], h = i ? ["b.xzxz", "b.ywyw"] : ["b.xyxy", "b.zwzw"], g = "", _ = "";
|
|
9
|
+
o && (g = s ? `vec4 activation(vec4 a) {
|
|
10
|
+
vec4 b = getPreluActivationWeightsAtOutCoords();
|
|
11
|
+
${o}
|
|
12
|
+
}` : l ? `vec4 activation(vec4 a) {
|
|
13
|
+
vec4 b = getLeakyreluAlphaAtOutCoords();
|
|
14
|
+
${o}
|
|
15
|
+
}` : `vec4 activation(vec4 x) {
|
|
16
|
+
${o}
|
|
17
|
+
}`, _ = "result = activation(result);");
|
|
18
|
+
let v = a ? "result += getBiasAtOutCoords();" : "";
|
|
19
|
+
a && this.variableNames.push("bias"), s && this.variableNames.push("preluActivationWeights"), l && this.variableNames.push("leakyreluAlpha");
|
|
20
|
+
let y = "rc.x", b = "rc.x";
|
|
21
|
+
e[0] < t[0] ? y = `imod(rc.x, ${e[0]})` : t[0] < e[0] && (b = `imod(rc.x, ${t[0]})`), this.userCode = `
|
|
22
|
+
${g}
|
|
23
|
+
// Don't use uniform for sharedDimensionPacked for performance.
|
|
24
|
+
const float sharedDimension = ${d}.0;
|
|
25
|
+
|
|
26
|
+
vec4 dot2x2ARowBCol(ivec3 rc) {
|
|
27
|
+
vec4 result = vec4(0);
|
|
28
|
+
int batchA = ${y};
|
|
29
|
+
int batchB = ${b};
|
|
30
|
+
for (int i = 0; i < ${d}; i++) {
|
|
31
|
+
vec4 a = getMatrixA(batchA, ${f});
|
|
32
|
+
vec4 b = getMatrixB(batchB, ${p});
|
|
33
|
+
|
|
34
|
+
// These swizzled products need to be separately added.
|
|
35
|
+
// See: https://github.com/tensorflow/tfjs/issues/1735
|
|
36
|
+
result += (${m[0]} * ${h[0]});
|
|
37
|
+
result += (${m[1]} * ${h[1]});
|
|
38
|
+
}
|
|
39
|
+
return result;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
void main() {
|
|
43
|
+
ivec3 rc = getOutputCoords();
|
|
44
|
+
vec4 result = dot2x2ARowBCol(rc);
|
|
45
|
+
|
|
46
|
+
${v}
|
|
47
|
+
|
|
48
|
+
${_}
|
|
49
|
+
|
|
50
|
+
setOutput(result);
|
|
51
|
+
}
|
|
52
|
+
`;
|
|
53
|
+
}
|
|
54
|
+
}, d = .7978845608028654, f = .044715, p = `
|
|
55
|
+
vec4 x3 = x * x * x;
|
|
56
|
+
vec4 inner = x + ${f} * x3;
|
|
57
|
+
inner = ${d} * inner;
|
|
58
|
+
inner = vec4(
|
|
59
|
+
abs(inner[0]) > 15.0 ? sign(inner[0]) : tanh(inner[0]),
|
|
60
|
+
abs(inner[1]) > 15.0 ? sign(inner[1]) : tanh(inner[1]),
|
|
61
|
+
abs(inner[2]) > 15.0 ? sign(inner[2]) : tanh(inner[2]),
|
|
62
|
+
abs(inner[3]) > 15.0 ? sign(inner[3]) : tanh(inner[3])
|
|
63
|
+
);
|
|
64
|
+
inner = 0.5 * (1.0 + inner);
|
|
65
|
+
vec4 result = x * inner;
|
|
66
|
+
return result;
|
|
67
|
+
`, m = `
|
|
68
|
+
vec4 a2 = a * a;
|
|
69
|
+
vec4 a3 = a2 * a;
|
|
70
|
+
vec4 u = ${d} * (a + ${f} * a3);
|
|
71
|
+
vec4 t = vec4(
|
|
72
|
+
abs(u[0]) > 15.0 ? sign(u[0]) : tanh(u[0]),
|
|
73
|
+
abs(u[1]) > 15.0 ? sign(u[1]) : tanh(u[1]),
|
|
74
|
+
abs(u[2]) > 15.0 ? sign(u[2]) : tanh(u[2]),
|
|
75
|
+
abs(u[3]) > 15.0 ? sign(u[3]) : tanh(u[3])
|
|
76
|
+
);
|
|
77
|
+
vec4 sech2 = 1.0 - t * t;
|
|
78
|
+
vec4 du_dx = ${d} * (1.0 + 3.0 * ${f} * a2);
|
|
79
|
+
vec4 dgelu = 0.5 * (1.0 + t) + 0.5 * a * sech2 * du_dx;
|
|
80
|
+
return dgelu * b;
|
|
81
|
+
`, h = 1e3;
|
|
82
|
+
function g({ a: t, b: i, transposeA: a, transposeB: o, backend: c, activationSnippet: d, multiplier: f }) {
|
|
83
|
+
let p = t.shape.length, m = i.shape.length, h = a ? t.shape[p - 2] : t.shape[p - 1], g = o ? i.shape[m - 1] : i.shape[m - 2], _ = a ? t.shape[p - 1] : t.shape[p - 2], v = o ? i.shape[m - 2] : i.shape[m - 1], y = t.shape.slice(0, -2), b = i.shape.slice(0, -2), x = s(y), S = s(b), C = n(t.shape.slice(0, -2), i.shape.slice(0, -2)).concat([_, v]);
|
|
84
|
+
r(h === g, () => `Error in matMul: inner shapes (${h}) and (${g}) of Tensors with shapes ${t.shape} and ${i.shape} and transposeA=${a} and transposeB=${o} must match.`);
|
|
85
|
+
let w = a ? [
|
|
86
|
+
x,
|
|
87
|
+
h,
|
|
88
|
+
_
|
|
89
|
+
] : [
|
|
90
|
+
x,
|
|
91
|
+
_,
|
|
92
|
+
h
|
|
93
|
+
], T = o ? [
|
|
94
|
+
S,
|
|
95
|
+
v,
|
|
96
|
+
g
|
|
97
|
+
] : [
|
|
98
|
+
S,
|
|
99
|
+
g,
|
|
100
|
+
v
|
|
101
|
+
], E = l({
|
|
102
|
+
inputs: { x: t },
|
|
103
|
+
backend: c,
|
|
104
|
+
attrs: { shape: w }
|
|
105
|
+
}), D = l({
|
|
106
|
+
inputs: { x: i },
|
|
107
|
+
backend: c,
|
|
108
|
+
attrs: { shape: T }
|
|
109
|
+
}), O = [E, D], k = Math.max(x, S), A = d, j = e(t.dtype, i.dtype), M = new u(w, T, [
|
|
110
|
+
k,
|
|
111
|
+
_,
|
|
112
|
+
v
|
|
113
|
+
], a, o, !1, A, !!f, !1), N = [E, D];
|
|
114
|
+
f && N.push(f);
|
|
115
|
+
let P = c.runWebGLProgram(M, N, j), F = l({
|
|
116
|
+
inputs: { x: P },
|
|
117
|
+
backend: c,
|
|
118
|
+
attrs: { shape: C }
|
|
119
|
+
});
|
|
120
|
+
O.push(P);
|
|
121
|
+
for (let e of O) c.disposeIntermediateTensorInfo(e);
|
|
122
|
+
return F;
|
|
123
|
+
}
|
|
124
|
+
function _(e) {
|
|
125
|
+
let { inputs: t, backend: n } = e, { x: r, kernel: i } = t;
|
|
126
|
+
if (r === void 0 || i === void 0) throw Error("BatchMatMul requires two input tensors.");
|
|
127
|
+
return g({
|
|
128
|
+
a: r,
|
|
129
|
+
b: i,
|
|
130
|
+
transposeA: !1,
|
|
131
|
+
transposeB: !1,
|
|
132
|
+
backend: n,
|
|
133
|
+
activationSnippet: p
|
|
134
|
+
});
|
|
135
|
+
}
|
|
136
|
+
t({
|
|
137
|
+
kernelName: "MatMulGelu",
|
|
138
|
+
backendName: "webgl",
|
|
139
|
+
kernelFunc: _
|
|
140
|
+
});
|
|
141
|
+
function v(e) {
|
|
142
|
+
let { dy: t, x: n, kernel: r } = e.inputs, s = e.backend;
|
|
143
|
+
return i(() => {
|
|
144
|
+
let e = o().makeTensorFromTensorInfo(g({
|
|
145
|
+
a: n,
|
|
146
|
+
b: r,
|
|
147
|
+
transposeA: !1,
|
|
148
|
+
transposeB: !1,
|
|
149
|
+
backend: s,
|
|
150
|
+
activationSnippet: m,
|
|
151
|
+
multiplier: t
|
|
152
|
+
}));
|
|
153
|
+
return [a(e, r, !1, !0), a(n, e, !0, !1)];
|
|
154
|
+
});
|
|
155
|
+
}
|
|
156
|
+
t({
|
|
157
|
+
kernelName: "MatMulGeluGrad",
|
|
158
|
+
backendName: "webgl",
|
|
159
|
+
kernelFunc: v
|
|
160
|
+
});
|
|
161
|
+
//#endregion
|
|
162
|
+
export { u as i, g as n, _ as r, h as t };
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { GPTConfigV1 } from './config';
|
|
2
|
+
import { Tensor } from '@tensorflow/tfjs-core';
|
|
3
|
+
import { default as Model, ModelForwardAttributes } from './model';
|
|
4
|
+
export default class NanoGPTV1 extends Model<ModelForwardAttributes, GPTConfigV1> {
|
|
5
|
+
private wte;
|
|
6
|
+
private wpe?;
|
|
7
|
+
private blocks;
|
|
8
|
+
private lnF;
|
|
9
|
+
private ropeCache?;
|
|
10
|
+
constructor(config?: Partial<GPTConfigV1>);
|
|
11
|
+
getClassName(): string;
|
|
12
|
+
private inputPhase;
|
|
13
|
+
forward(attrs: ModelForwardAttributes, idx: Tensor): Tensor;
|
|
14
|
+
project(embeddings: Tensor): Tensor;
|
|
15
|
+
dispose(): void;
|
|
16
|
+
}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { GPTConfigV2 } from './config';
|
|
2
|
+
import { Tensor } from '@tensorflow/tfjs-core';
|
|
3
|
+
import { default as Model, ModelForwardAttributes } from './model';
|
|
4
|
+
export default class NanoGPTV2 extends Model<ModelForwardAttributes, GPTConfigV2> {
|
|
5
|
+
private wte;
|
|
6
|
+
private wpe?;
|
|
7
|
+
private blocks;
|
|
8
|
+
private lnF;
|
|
9
|
+
private ropeCache?;
|
|
10
|
+
constructor(config?: Partial<GPTConfigV2>);
|
|
11
|
+
getClassName(): string;
|
|
12
|
+
private inputPhase;
|
|
13
|
+
forward(attrs: ModelForwardAttributes, idx: Tensor): Tensor;
|
|
14
|
+
project(embeddings: Tensor): Tensor;
|
|
15
|
+
dispose(): void;
|
|
16
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
export interface LoRAConfig {
|
|
2
|
+
rank: number;
|
|
3
|
+
alpha: number;
|
|
4
|
+
variables: string[];
|
|
5
|
+
}
|
|
6
|
+
export interface GPTConfigBase {
|
|
7
|
+
modelType?: string;
|
|
8
|
+
vocabSize: number;
|
|
9
|
+
blockSize: number;
|
|
10
|
+
nLayer: number;
|
|
11
|
+
nHead: number;
|
|
12
|
+
nEmbed: number;
|
|
13
|
+
mlpFactor: number;
|
|
14
|
+
loraConfig?: Map<string, LoRAConfig>;
|
|
15
|
+
loraName?: string;
|
|
16
|
+
}
|
|
17
|
+
export interface GPTConfigV1 extends GPTConfigBase {
|
|
18
|
+
modelType: 'GenAI_NanoGPT_v1';
|
|
19
|
+
useRope: boolean;
|
|
20
|
+
}
|
|
21
|
+
export interface GPTConfigV2 extends GPTConfigBase {
|
|
22
|
+
modelType: 'GenAI_NanoGPT_v2';
|
|
23
|
+
windowSize?: string;
|
|
24
|
+
}
|
|
25
|
+
export type GPTConfig = GPTConfigV1 | GPTConfigV2;
|
|
26
|
+
export declare const defaultConfig: GPTConfig;
|
|
27
|
+
export declare function validateConfig(config: unknown): asserts config is GPTConfig;
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
//#region lib/models/config.ts
|
|
2
|
+
var e = {
|
|
3
|
+
modelType: "GenAI_NanoGPT_v2",
|
|
4
|
+
vocabSize: 2e3,
|
|
5
|
+
blockSize: 128,
|
|
6
|
+
nLayer: 6,
|
|
7
|
+
nHead: 4,
|
|
8
|
+
nEmbed: 256,
|
|
9
|
+
mlpFactor: 4
|
|
10
|
+
};
|
|
11
|
+
function t(e, t) {
|
|
12
|
+
if (typeof e[t] != "number" || Number.isNaN(e[t])) throw Error(`Invalid config: "${t}" must be a number.`);
|
|
13
|
+
}
|
|
14
|
+
function n(e) {
|
|
15
|
+
let n = (e) => typeof e == "object" && !!e && !Array.isArray(e);
|
|
16
|
+
if (!n(e)) throw Error("Invalid config: expected an object.");
|
|
17
|
+
if (t(e, "vocabSize"), t(e, "blockSize"), t(e, "nLayer"), t(e, "nHead"), t(e, "nEmbed"), t(e, "mlpFactor"), e.loraConfig !== void 0) {
|
|
18
|
+
if (!n(e.loraConfig)) throw Error("Invalid config: \"loraConfig\" must be an object.");
|
|
19
|
+
let r = Object.values(e.loraConfig);
|
|
20
|
+
if (!r.every((e) => n(e))) throw Error("Invalid config: each entry in \"loraConfig\" must be an object.");
|
|
21
|
+
if (!r.every((e) => "rank" in e && "alpha" in e && "variables" in e)) throw Error("Invalid config: each LoRA config must have \"rank\", \"alpha\", and \"variables\" fields.");
|
|
22
|
+
r.forEach((e) => {
|
|
23
|
+
if (t(e, "rank"), t(e, "alpha"), !Array.isArray(e.variables) || !e.variables.every((e) => typeof e == "string")) throw Error("Invalid config: \"variables\" must be a string array.");
|
|
24
|
+
});
|
|
25
|
+
}
|
|
26
|
+
if (e.modelType === "GenAI_NanoGPT_v1") {
|
|
27
|
+
if (typeof e.useRope != "boolean") throw Error("Invalid config: \"useRope\" must be a boolean for GenAI_NanoGPT_v1.");
|
|
28
|
+
return;
|
|
29
|
+
}
|
|
30
|
+
if (e.modelType === "GenAI_NanoGPT_v2") {
|
|
31
|
+
if (e.windowSize !== void 0 && typeof e.windowSize != "string") throw Error("Invalid config: \"windowSize\" must be a string for GenAI_NanoGPT_v2.");
|
|
32
|
+
return;
|
|
33
|
+
}
|
|
34
|
+
throw Error("Invalid config: \"modelType\" must be \"GenAI_NanoGPT_v1\" or \"GenAI_NanoGPT_v2\".");
|
|
35
|
+
}
|
|
36
|
+
//#endregion
|
|
37
|
+
export { e as defaultConfig, n as validateConfig };
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
import { Tensor } from '@tensorflow/tfjs-core';
|
|
2
|
+
import { ForwardAttributes, default as BaseLayer } from '../layers/BaseLayer';
|
|
3
|
+
import { AttentionScores, KVCache } from '../layers/CausalSelfAttention';
|
|
4
|
+
import { TransformersMetadata } from '../../loader/types';
|
|
5
|
+
import { GPTConfig, LoRAConfig } from './config';
|
|
6
|
+
import { default as LoRA } from '../../layers/LoRA';
|
|
7
|
+
export interface ModelForwardAttributes extends ForwardAttributes {
|
|
8
|
+
cache?: KVCache[];
|
|
9
|
+
attentionScores?: AttentionScores;
|
|
10
|
+
seed?: number;
|
|
11
|
+
skipLogits?: boolean;
|
|
12
|
+
ropePositionOffset?: number;
|
|
13
|
+
}
|
|
14
|
+
export interface TrainingState {
|
|
15
|
+
steps: number;
|
|
16
|
+
learningRate: number;
|
|
17
|
+
batchSize: number;
|
|
18
|
+
loss: number;
|
|
19
|
+
tokensProcessed: number;
|
|
20
|
+
duration: number;
|
|
21
|
+
}
|
|
22
|
+
export default abstract class Model<T extends ModelForwardAttributes, C extends GPTConfig = GPTConfig> extends BaseLayer<T, C> {
|
|
23
|
+
lossScaling: number;
|
|
24
|
+
trainingState: TrainingState | null;
|
|
25
|
+
metaData: TransformersMetadata;
|
|
26
|
+
private loraLayer?;
|
|
27
|
+
private loraMap;
|
|
28
|
+
constructor(config: C);
|
|
29
|
+
createLoRA(name: string, loraConfig: LoRAConfig): void;
|
|
30
|
+
deleteLoRA(name: string): void;
|
|
31
|
+
renameLoRA(oldName: string, newName: string): void;
|
|
32
|
+
mergeLoRA(name: string): void;
|
|
33
|
+
attachLoRA(name: string): void;
|
|
34
|
+
detachLoRA(): void;
|
|
35
|
+
hasLoRA(name?: string): boolean;
|
|
36
|
+
listLoRAs(): string[];
|
|
37
|
+
get lora(): LoRA | null;
|
|
38
|
+
abstract getClassName(): string;
|
|
39
|
+
abstract forward(attrs: T, idx: Tensor): Tensor;
|
|
40
|
+
abstract project(embeddings: Tensor): Tensor;
|
|
41
|
+
abstract dispose(): void;
|
|
42
|
+
getNumParams(): number;
|
|
43
|
+
protected validateInput(idx: Tensor): void;
|
|
44
|
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import { ii as e } from "../dist-BewPQWjc.js";
|
|
2
|
+
import "./cpu/adamAdjust.js";
|
|
3
|
+
import "./webgl/adamAdjust.js";
|
|
4
|
+
//#region lib/ops/adamAdjust.ts
|
|
5
|
+
function t(t, n, r, i, a, o, s = 0) {
|
|
6
|
+
return e().runKernel("AdamAdjust", {
|
|
7
|
+
moments: t,
|
|
8
|
+
value: n
|
|
9
|
+
}, {
|
|
10
|
+
beta1: r,
|
|
11
|
+
beta2: i,
|
|
12
|
+
epsilon: a,
|
|
13
|
+
learningRate: o,
|
|
14
|
+
weightDecay: s
|
|
15
|
+
});
|
|
16
|
+
}
|
|
17
|
+
//#endregion
|
|
18
|
+
export { t as adamAdjust };
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { ii as e } from "../dist-BewPQWjc.js";
|
|
2
|
+
import "./cpu/adamMoments.js";
|
|
3
|
+
import "./webgl/adamMoments.js";
|
|
4
|
+
//#region lib/ops/adamMoments.ts
|
|
5
|
+
function t(t, n, r, i, a) {
|
|
6
|
+
return e().runKernel("AdamMoments", {
|
|
7
|
+
moments: t,
|
|
8
|
+
gradient: n,
|
|
9
|
+
scaling: a
|
|
10
|
+
}, {
|
|
11
|
+
beta1: r,
|
|
12
|
+
beta2: i
|
|
13
|
+
});
|
|
14
|
+
}
|
|
15
|
+
//#endregion
|
|
16
|
+
export { t as adamMoments };
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import { ii as e, qr as t } from "../dist-BewPQWjc.js";
|
|
2
|
+
import { isPackedTensor as n } from "../utilities/packed.js";
|
|
3
|
+
import "./grads/add16.js";
|
|
4
|
+
//#region lib/ops/add16.ts
|
|
5
|
+
function r(r, i) {
|
|
6
|
+
return !n(r) && !n(i) ? t(r, i) : e().runKernel("Add16", {
|
|
7
|
+
a: r,
|
|
8
|
+
b: i
|
|
9
|
+
});
|
|
10
|
+
}
|
|
11
|
+
//#endregion
|
|
12
|
+
export { r as add16 };
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { _r as e, ii as t, kt as n } from "../dist-BewPQWjc.js";
|
|
2
|
+
import { isPackedTensor as r } from "../utilities/packed.js";
|
|
3
|
+
import "./cpu/appendCache.js";
|
|
4
|
+
import "./webgl/appendCache.js";
|
|
5
|
+
//#region lib/ops/appendCache.ts
|
|
6
|
+
function i(i, a, o, s) {
|
|
7
|
+
if (!s) {
|
|
8
|
+
let t = i.shape[2], o = r(i);
|
|
9
|
+
return e([i, n([
|
|
10
|
+
i.shape[0],
|
|
11
|
+
i.shape[1],
|
|
12
|
+
a - t,
|
|
13
|
+
i.shape[3]
|
|
14
|
+
], o ? "int32" : i.dtype)], 2);
|
|
15
|
+
}
|
|
16
|
+
return t().runKernel("AppendCache", {
|
|
17
|
+
cache: s,
|
|
18
|
+
item: i
|
|
19
|
+
}, {
|
|
20
|
+
maxSize: a,
|
|
21
|
+
pastLen: o
|
|
22
|
+
});
|
|
23
|
+
}
|
|
24
|
+
//#endregion
|
|
25
|
+
export { i as appendCache };
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { ii as e } from "../dist-BewPQWjc.js";
|
|
2
|
+
import "./cpu/attentionMask.js";
|
|
3
|
+
import "./webgl/attentionMask.js";
|
|
4
|
+
import "./grads/attentionMask.js";
|
|
5
|
+
//#region lib/ops/attentionMask.ts
|
|
6
|
+
function t(t, n, r, i) {
|
|
7
|
+
return e().runKernel("AttentionMask", {
|
|
8
|
+
q: t,
|
|
9
|
+
k: n
|
|
10
|
+
}, {
|
|
11
|
+
divisor: r,
|
|
12
|
+
pastLen: i || 0
|
|
13
|
+
});
|
|
14
|
+
}
|
|
15
|
+
//#endregion
|
|
16
|
+
export { t as attentionMask };
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
import { _r as e, ii as t } from "../dist-BewPQWjc.js";
|
|
2
|
+
import { isPackedTensor as n } from "../utilities/packed.js";
|
|
3
|
+
//#region lib/ops/concat16.ts
|
|
4
|
+
function r(r, i) {
|
|
5
|
+
return n(r[0]) ? t().runKernel("Concat16", r, { axis: i ?? -1 }) : e(r, i);
|
|
6
|
+
}
|
|
7
|
+
//#endregion
|
|
8
|
+
export { r as concat16 };
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { Gr as e, Ii as t, Wr as n, gn as r, qr as i } from "../../dist-BewPQWjc.js";
|
|
2
|
+
//#region lib/ops/cpu/adamAdjust.ts
|
|
3
|
+
function a(t) {
|
|
4
|
+
let { moments: a, value: o } = t.inputs, { beta1: s, beta2: c, epsilon: l, learningRate: u } = t.attrs, d = a.shape.length, f = Array(d).fill(0), p = a.shape.slice();
|
|
5
|
+
p[d - 1] = 1;
|
|
6
|
+
let m = f.slice();
|
|
7
|
+
m[d - 1] = 1;
|
|
8
|
+
let h = p.slice(), g = a.slice(f, p).squeeze([d - 1]), _ = a.slice(m, h).squeeze([d - 1]);
|
|
9
|
+
return i(n(e(e(g, s), i(r(e(_, c)), l ?? 1e-8)), -u), o);
|
|
10
|
+
}
|
|
11
|
+
t({
|
|
12
|
+
kernelName: "AdamAdjust",
|
|
13
|
+
backendName: "cpu",
|
|
14
|
+
kernelFunc: a
|
|
15
|
+
});
|
|
16
|
+
//#endregion
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { Ii as e, V as t } from "../../dist-BewPQWjc.js";
|
|
2
|
+
//#region lib/ops/cpu/adamMoments.ts
|
|
3
|
+
function n(e) {
|
|
4
|
+
let { moments: n, gradient: r } = e.inputs, { beta1: i, beta2: a } = e.attrs, o = n.shape.length, s = Array(o).fill(0), c = n.shape.slice();
|
|
5
|
+
c[o - 1] = 1;
|
|
6
|
+
let l = s.slice();
|
|
7
|
+
l[o - 1] = 1;
|
|
8
|
+
let u = c.slice(), d = n.slice(s, c).squeeze([o - 1]), f = n.slice(l, u).squeeze([o - 1]);
|
|
9
|
+
return t([d.mul(i).add(r.mul(1 - i)), f.mul(a).add(r.square().mul(1 - a))], -1);
|
|
10
|
+
}
|
|
11
|
+
e({
|
|
12
|
+
kernelName: "AdamMoments",
|
|
13
|
+
backendName: "cpu",
|
|
14
|
+
kernelFunc: n
|
|
15
|
+
});
|
|
16
|
+
//#endregion
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|