npm - @genai-fi/nanogpt - Versions diffs - 0.11.0 → 0.12.0 - Mend

@genai-fi/nanogpt 0.11.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (236) hide show

package/dist/Generator.js +29 -29
package/dist/{RealDiv-Ds-jvL09.js → RealDiv-C8neBwFi.js} +17 -17
package/dist/{Reshape-Cd6e-Otn.js → Reshape-Bd4V_4X7.js} +1 -1
package/dist/{Reshape-Ct266DEk.js → Reshape-Ck29jQSY.js} +7 -7
package/dist/TeachableLLM.d.ts +2 -1
package/dist/TeachableLLM.js +9 -9
package/dist/Trainer.d.ts +4 -2
package/dist/Trainer.js +11 -8
package/dist/{axis_util-DofAuy0p.js → axis_util-DGqbT-FX.js} +1 -1
package/dist/backend.js +2 -2
package/dist/{backend_util-C7NWHpv7.js → backend_util-DC3rBo_H.js} +18 -18
package/dist/{backend_webgpu-B0Vls736.js → backend_webgpu-mbhNnlx9.js} +10 -10
package/dist/{broadcast_to-DDaNMbX7.js → broadcast_to-D1Dmg2Oz.js} +2 -2
package/dist/checks/appendCache.js +2 -2
package/dist/checks/attentionMask.js +3 -3
package/dist/checks/gelu.js +2 -2
package/dist/checks/matMulGelu.js +2 -2
package/dist/checks/normRMS.js +4 -4
package/dist/checks/normRMSGrad.js +3 -3
package/dist/checks/packUnpack.js +2 -2
package/dist/checks/qkv.js +2 -2
package/dist/checks/rope.js +2 -2
package/dist/clip_by_value-fg2aKzUy.js +12 -0
package/dist/{complex-DClmWqJt.js → complex-Cyg-eQeZ.js} +1 -1
package/dist/concat-CSm2rMwe.js +17 -0
package/dist/{concat_util-CHsJFZJJ.js → concat_util-D0je5Ppu.js} +1 -1
package/dist/{dataset-DcjWqUVQ.js → dataset-CVIJu7Xa.js} +3 -3
package/dist/{dropout-OxuaJz6z.js → dropout-DLhSMNTZ.js} +14 -14
package/dist/expand_dims-ChkuOp6I.js +11 -0
package/dist/{exports_initializers-eS9QJ6ut.js → exports_initializers-1KWPiStI.js} +1 -1
package/dist/{floor-DIb-lN_u.js → floor-BRMPgeIs.js} +1 -1
package/dist/gather-BSULDalH.js +9 -0
package/dist/{gelu-DqTbCx5x.js → gelu-BK1k-n1i.js} +1 -1
package/dist/{gpgpu_math-CJcbnKPC.js → gpgpu_math-BJSTk_mW.js} +25 -25
package/dist/{index-Dj5TkmPY.js → index-BBVLAXZD.js} +14 -14
package/dist/{index-D0RBWjq8.js → index-Duu1Lvvv.js} +45 -45
package/dist/{kernel_funcs_utils-CSaumNDs.js → kernel_funcs_utils-BtYrPoJu.js} +8 -8
package/dist/layers/BaseLayer.js +2 -2
package/dist/layers/CausalSelfAttention.js +6 -6
package/dist/layers/MLP.js +4 -4
package/dist/layers/PositionEmbedding.js +5 -5
package/dist/layers/RMSNorm.js +3 -3
package/dist/layers/RoPECache.js +4 -4
package/dist/layers/TiedEmbedding.js +6 -6
package/dist/layers/TransformerBlock.js +1 -1
package/dist/loader/loadTransformers.js +1 -1
package/dist/loader/oldZipLoad.js +17 -17
package/dist/{log_sum_exp-VLZgbFAH.js → log_sum_exp-CVqLsVLl.js} +4 -4
package/dist/main.d.ts +9 -0
package/dist/main.js +68 -58
package/dist/{matMul16-cDxwemKj.js → matMul16-xswmhSuF.js} +7 -7
package/dist/{matMulGelu-B2s_80-H.js → matMulGelu-BpvgnYG8.js} +26 -26
package/dist/mat_mul-Bn2BDpT4.js +11 -0
package/dist/{mod-PrOKlFxH.js → mod-B4AUd1Np.js} +1 -1
package/dist/models/NanoGPTV1.js +2 -2
package/dist/models/model.js +9 -9
package/dist/{ones-BX_wEgzB.js → ones-CBI1AQjb.js} +3 -3
package/dist/ops/adamAdjust.js +1 -1
package/dist/ops/adamMoments.js +1 -1
package/dist/ops/add16.js +1 -1
package/dist/ops/appendCache.js +3 -3
package/dist/ops/attentionMask.js +1 -1
package/dist/ops/concat16.js +2 -2
package/dist/ops/cpu/adamAdjust.js +7 -7
package/dist/ops/cpu/adamMoments.js +5 -5
package/dist/ops/cpu/appendCache.js +6 -6
package/dist/ops/cpu/attentionMask.js +6 -6
package/dist/ops/cpu/fusedSoftmax.js +5 -5
package/dist/ops/cpu/gatherSub.js +7 -7
package/dist/ops/cpu/gelu.js +5 -5
package/dist/ops/cpu/matMul16.js +2 -2
package/dist/ops/cpu/matMulGelu.js +3 -3
package/dist/ops/cpu/matMulMul.js +5 -5
package/dist/ops/cpu/mulDropout.js +1 -1
package/dist/ops/cpu/normRMS.js +5 -5
package/dist/ops/cpu/qkv.js +3 -3
package/dist/ops/cpu/rope.js +9 -9
package/dist/ops/cpu/scatterSub.js +5 -5
package/dist/ops/dot16.js +2 -2
package/dist/ops/gatherSub.js +1 -1
package/dist/ops/gelu.js +2 -2
package/dist/ops/grads/add16.js +1 -1
package/dist/ops/grads/attentionMask.js +2 -2
package/dist/ops/grads/gelu.js +2 -2
package/dist/ops/grads/matMul16.js +3 -3
package/dist/ops/grads/matMulGelu.js +5 -5
package/dist/ops/grads/normRMS.js +6 -6
package/dist/ops/grads/pack16.js +3 -3
package/dist/ops/grads/qkv.js +9 -9
package/dist/ops/grads/rope.js +2 -2
package/dist/ops/grads/softmax16.js +1 -1
package/dist/ops/grads/unpack16.js +2 -2
package/dist/ops/matMul16.js +3 -3
package/dist/ops/matMulGelu.js +2 -2
package/dist/ops/matMulMul.js +1 -1
package/dist/ops/mul16.js +1 -1
package/dist/ops/mulDrop.js +1 -1
package/dist/ops/normRMS.js +1 -1
package/dist/ops/pack16.js +2 -2
package/dist/ops/qkv.js +1 -1
package/dist/ops/reshape16.js +6 -6
package/dist/ops/rope.js +2 -2
package/dist/ops/scatterSub.js +1 -1
package/dist/ops/slice16.js +2 -2
package/dist/ops/softmax16.js +1 -1
package/dist/ops/sub16.js +1 -1
package/dist/ops/sum16.js +2 -2
package/dist/ops/transpose16.js +6 -6
package/dist/ops/unpack16.js +2 -2
package/dist/ops/webgl/adamAdjust.js +2 -2
package/dist/ops/webgl/adamMoments.js +1 -1
package/dist/ops/webgl/appendCache.js +1 -1
package/dist/ops/webgl/attentionMask.js +4 -4
package/dist/ops/webgl/fusedSoftmax.js +6 -6
package/dist/ops/webgl/gatherSub.js +1 -1
package/dist/ops/webgl/gelu.js +2 -2
package/dist/ops/webgl/log.js +3 -3
package/dist/ops/webgl/matMul16.js +10 -10
package/dist/ops/webgl/matMulGelu.js +4 -4
package/dist/ops/webgl/matMulMul.js +2 -2
package/dist/ops/webgl/mulDropout.js +1 -1
package/dist/ops/webgl/normRMS.js +2 -2
package/dist/ops/webgl/qkv.js +1 -1
package/dist/ops/webgl/rope.js +4 -4
package/dist/ops/webgl/scatterSub.js +1 -1
package/dist/ops/webgpu/adamAdjust.js +3 -3
package/dist/ops/webgpu/adamMoments.js +5 -5
package/dist/ops/webgpu/add16.js +1 -1
package/dist/ops/webgpu/appendCache.js +3 -3
package/dist/ops/webgpu/attentionMask.js +5 -5
package/dist/ops/webgpu/attentionMask32_program.js +2 -2
package/dist/ops/webgpu/concat16.js +5 -5
package/dist/ops/webgpu/gatherSub.js +3 -3
package/dist/ops/webgpu/gelu.js +3 -3
package/dist/ops/webgpu/matMul16.js +19 -19
package/dist/ops/webgpu/matMul16_program.js +2 -2
package/dist/ops/webgpu/mul16.js +1 -1
package/dist/ops/webgpu/normRMS.js +2 -2
package/dist/ops/webgpu/normRMSGrad.js +4 -4
package/dist/ops/webgpu/pack16.js +3 -3
package/dist/ops/webgpu/pack16_program.js +2 -2
package/dist/ops/webgpu/qkv.js +4 -4
package/dist/ops/webgpu/rope.js +3 -3
package/dist/ops/webgpu/scatterSub.js +3 -3
package/dist/ops/webgpu/slice16.js +4 -4
package/dist/ops/webgpu/softmax16.js +4 -4
package/dist/ops/webgpu/softmax16_program.js +2 -2
package/dist/ops/webgpu/softmax16_subgroup_program.js +2 -2
package/dist/ops/webgpu/softmax16grad.js +1 -1
package/dist/ops/webgpu/sub16.js +1 -1
package/dist/ops/webgpu/sum16.js +5 -5
package/dist/ops/webgpu/transpose16.js +2 -2
package/dist/ops/webgpu/transpose16_program.js +2 -2
package/dist/ops/webgpu/transpose16_shared_program.js +3 -3
package/dist/ops/webgpu/unpack16.js +5 -5
package/dist/ops/webgpu/utils/binary_op.js +3 -3
package/dist/ops/webgpu/utils/reductions.js +4 -4
package/dist/{ops-FJapAPfm.js → ops-C2_OXuZ4.js} +35 -35
package/dist/{pack16-k4jq6aMX.js → pack16-atD0eYRm.js} +6 -6
package/dist/patches/webgpu_backend.js +8 -8
package/dist/patches/webgpu_base.js +1 -1
package/dist/patches/webgpu_program.js +2 -2
package/dist/{random_width-UGQn4OWb.js → random_width-BN4wGJaW.js} +33 -33
package/dist/{range-CuGvVN2c.js → range-DKmP1-OQ.js} +1 -1
package/dist/relu-BsXmGzzu.js +9 -0
package/dist/{reshape-CkjKPPqB.js → reshape-BI0yzp1T.js} +1 -1
package/dist/{resize_nearest_neighbor-DB8k9KN_.js → resize_nearest_neighbor-BA_BX-ub.js} +25 -25
package/dist/{rope-BmZmp9uP.js → rope-DJ7Y7c-u.js} +1 -1
package/dist/{scatter_nd_util-BY22Cc-C.js → scatter_nd_util-k9MUVUkn.js} +1 -1
package/dist/{selu_util-BuLbmbrl.js → selu_util-DyW0X1WG.js} +5 -5
package/dist/{shared-B7USJZgw.js → shared-Q3BS6T03.js} +1 -1
package/dist/{shared-BQboIImQ.js → shared-nnSWpC3u.js} +6 -6
package/dist/{slice-Aqy7KbJh.js → slice-wBNvzVyz.js} +3 -3
package/dist/{slice_util-D8CQRenR.js → slice_util-zN8KFC5I.js} +7 -7
package/dist/{softmax-faLoUZVT.js → softmax-DfuYyjMh.js} +1 -1
package/dist/split-BYrLboMq.js +9 -0
package/dist/squeeze-Bk8Brcct.js +10 -0
package/dist/{stack-WJK22CFn.js → stack-CDWShFHF.js} +1 -1
package/dist/{step-dXR33iOg.js → step-BS5JXRR6.js} +14 -14
package/dist/sum-BPUfDB2X.js +11 -0
package/dist/{tensor-BQqrDvpx.js → tensor-CEt9Nm2s.js} +1 -1
package/dist/{tensor1d-LxP9asMm.js → tensor1d-Cc_KCIDg.js} +1 -1
package/dist/{tensor2d-BN1sSfQO.js → tensor2d-BN97fF71.js} +1 -1
package/dist/{tensor4d-DVwr7pLF.js → tensor4d-vuDDgdUI.js} +1 -1
package/dist/{tfjs_backend-Vi4JfLzT.js → tfjs_backend-806hyYve.js} +36 -36
package/dist/tile-OWUvpIVt.js +11 -0
package/dist/tokeniser/BaseTokeniser.d.ts +6 -8
package/dist/tokeniser/BaseTokeniser.js +6 -6
package/dist/tokeniser/CharTokeniser.d.ts +6 -6
package/dist/tokeniser/CharTokeniser.js +26 -26
package/dist/tokeniser/bpe.d.ts +6 -6
package/dist/tokeniser/bpe.js +9 -9
package/dist/tokeniser/type.d.ts +6 -8
package/dist/training/Adam.js +2 -2
package/dist/training/AdamExt.js +1 -1
package/dist/training/DatasetBuilder.d.ts +1 -1
package/dist/training/DatasetBuilder.js +29 -29
package/dist/training/FullTrainer.js +1 -1
package/dist/training/Trainer.d.ts +5 -4
package/dist/training/Trainer.js +22 -25
package/dist/training/sparseCrossEntropy.js +3 -3
package/dist/training/tasks/ConversationTask.d.ts +11 -0
package/dist/training/tasks/ConversationTask.js +26 -0
package/dist/training/tasks/PretrainingTask.d.ts +11 -0
package/dist/training/tasks/PretrainingTask.js +34 -0
package/dist/training/tasks/StartSentenceTask.d.ts +12 -0
package/dist/training/tasks/StartSentenceTask.js +42 -0
package/dist/training/tasks/Task.d.ts +8 -0
package/dist/training/tasks/Task.js +41 -0
package/dist/{transpose-JawVKyZy.js → transpose-BUkQCJp9.js} +7 -7
package/dist/{unsorted_segment_sum-LAbmE9G4.js → unsorted_segment_sum-BljxHhCY.js} +78 -78
package/dist/utilities/dummy.js +3 -3
package/dist/utilities/multinomialCPU.js +2 -2
package/dist/utilities/packed.js +1 -1
package/dist/utilities/performance.js +1 -1
package/dist/utilities/profile.js +1 -1
package/dist/utilities/safetensors.js +2 -2
package/dist/utilities/sentences.d.ts +1 -1
package/dist/utilities/sentences.js +11 -11
package/dist/utilities/weights.js +2 -2
package/dist/{variable-DQ9yYgEU.js → variable-DPt_Iuog.js} +1 -1
package/dist/{webgpu_program-CAE4RICo.js → webgpu_program-BpWRlghH.js} +1 -1
package/dist/{webgpu_util-BdovYhXr.js → webgpu_util-DMiKzzQM.js} +7 -7
package/dist/{zeros-DeiE2zTa.js → zeros-5YROwwUH.js} +2 -2
package/dist/{zeros_like-BAz3iKru.js → zeros_like-De4n1C3m.js} +57 -57
package/package.json +1 -1
package/dist/clip_by_value-Dn5tzexi.js +0 -12
package/dist/concat-C6X3AAlQ.js +0 -17
package/dist/expand_dims-BzfJK2uc.js +0 -11
package/dist/gather-BcO5UQNJ.js +0 -9
package/dist/mat_mul-DxpNTCRz.js +0 -11
package/dist/relu-Cf80uA2p.js +0 -9
package/dist/split-BNz5jcGc.js +0 -9
package/dist/squeeze--YMgaAAf.js +0 -10
package/dist/sum-BdplSvq_.js +0 -11
package/dist/tile-CvN_LyVr.js +0 -11

package/dist/tokeniser/BaseTokeniser.js CHANGED Viewed

@@ -24,11 +24,11 @@ class k extends r {
   addSpecialToken(e, t) {
     this.specialTokens.set(e, t), this.specialTokenSet.add(t);
   }
-  async encodeSequence(e) {
-    const t = await this.encode(e);
+  encodeSequence(e) {
+    const t = this.encode(e);
     return [this.bosToken, ...t, this.eosToken];
   }
-  async encodeConversation(e, t) {
+  encodeConversation(e, t) {
     const s = [[this.bosToken]], a = [
       this.getSpecialTokenIndex("<|user_start|>"),
       this.getSpecialTokenIndex("<|assistant_start|>"),
@@ -39,7 +39,7 @@ class k extends r {
       this.getSpecialTokenIndex("<|system_end|>")
     ];
     for (const i of e) {
-      const c = await this.encode(i.content);
+      const c = this.encode(i.content);
       switch (i.role) {
         case "user":
           s.push([a[0]]);
@@ -66,7 +66,7 @@ class k extends r {
     const o = s.flat();
     return t ? o.push(a[1]) : o.push(this.eosToken), o;
   }
-  async decodeConversation(e) {
+  decodeConversation(e) {
     const t = [];
     let s = 0;
     for (; s < e.length; ) {
@@ -77,7 +77,7 @@ class k extends r {
         const o = [];
         for (; s < e.length && e[s] !== this.getSpecialTokenIndex(`<|${n}_end|>`); )
           o.push(e[s]), s++;
-        const i = await this.decode(o);
+        const i = this.decode(o);
         t.push({ role: n, content: i });
       }
       s++;

package/dist/tokeniser/CharTokeniser.d.ts CHANGED Viewed

@@ -13,12 +13,12 @@ export default class CharTokeniser extends BaseTokeniser {
     get trained(): boolean;
     destroy(): void;
     train(text: string[]): Promise<number>;
-    tokenise(text: string[], numeric: true): Promise<number[][]>;
-    tokenise(text: string[]): Promise<string[][]>;
-    detokenise(tokens: number[][]): Promise<string[]>;
-    encode(text: string): Promise<number[]>;
-    decode(tokens: number[]): Promise<string>;
+    tokenise(text: string[], numeric: true): number[][];
+    tokenise(text: string[]): string[][];
+    detokenise(tokens: (number[] | Uint16Array)[]): string[];
+    encode(text: string): number[];
+    decode(tokens: number[] | Uint16Array): string;
     getVocab(): string[];
-    getMerges(): Promise<[string, string][]>;
+    getMerges(): [string, string][];
     createTrainingData(text: string[], windowSize?: number): Promise<[number[], number[]]>;
 }

package/dist/tokeniser/CharTokeniser.js CHANGED Viewed

@@ -8,9 +8,9 @@ class b extends k {
   vocab = [];
   cache = /* @__PURE__ */ new Map();
   _trained = !1;
-  constructor(s) {
-    if (super(), Array.isArray(s)) {
-      if (this.vocab = s, this.vocab.length > 0)
+  constructor(i) {
+    if (super(), Array.isArray(i)) {
+      if (this.vocab = i, this.vocab.length > 0)
         this.vocabSize = this.vocab.length, d.forEach((t) => {
           const e = this.vocab.indexOf(t);
           e !== -1 && this.addSpecialToken(t, e);
@@ -21,37 +21,37 @@ class b extends k {
         throw new Error("Vocab cannot be empty");
       this._trained = !0;
     } else
-      this.vocabSize = s, this.vocab = new Array(this.vocabSize).fill(""), this.addSpecialTokens(), this.eosToken = this.getSpecialTokenIndex("<eos>"), this.bosToken = this.getSpecialTokenIndex("<bos>") ?? this.eosToken, this.unkToken = this.getSpecialTokenIndex(""), this.vocab.forEach((t, e) => {
+      this.vocabSize = i, this.vocab = new Array(this.vocabSize).fill(""), this.addSpecialTokens(), this.eosToken = this.getSpecialTokenIndex("<eos>"), this.bosToken = this.getSpecialTokenIndex("<bos>") ?? this.eosToken, this.unkToken = this.getSpecialTokenIndex(""), this.vocab.forEach((t, e) => {
         this.cache.set(t, e);
       }), this.cache.set("", this.unkToken);
   }
-  addToken(s, t) {
-    if (this.cache.has(s))
-      return this.cache.get(s);
+  addToken(i, t) {
+    if (this.cache.has(i))
+      return this.cache.get(i);
     let e;
     if (t !== void 0 ? e = t : (e = this.vocab.indexOf("", this.unkToken + 1), e === -1 && (e = this.vocabSize)), e >= this.vocabSize)
       throw new Error("Vocab size exceeded");
-    return this.vocab[e] = s, this.cache.set(s, e), e;
+    return this.vocab[e] = i, this.cache.set(i, e), e;
   }
   get trained() {
     return this.vocab.length === this.vocabSize && this._trained;
   }
   destroy() {
   }
-  async train(s) {
-    const t = s.map((n) => n.split("")).flat(), e = new Set(t), i = Array.from(e), h = this.vocab.indexOf("", this.unkToken + 1), o = this.vocabSize - u.length;
+  async train(i) {
+    const t = i.map((n) => n.split("")).flat(), e = new Set(t), s = Array.from(e), h = this.vocab.indexOf("", this.unkToken + 1), o = this.vocabSize - u.length;
     if (h === -1)
       return this.vocabSize;
-    if (this._trained = !0, i.length > o) {
+    if (this._trained = !0, s.length > o) {
       const n = /* @__PURE__ */ new Map();
       t.forEach((a) => {
         n.set(a, (n.get(a) || 0) + 1);
-      }), i.sort((a, r) => (n.get(a) || 0) - (n.get(r) || 0)), i.splice(0, i.length - o);
+      }), s.sort((a, r) => (n.get(a) || 0) - (n.get(r) || 0)), s.splice(0, s.length - o);
     }
     let c = h;
     if (c !== -1) {
       const n = new Set(this.vocab);
-      for (const a of i)
+      for (const a of s)
         if (!n.has(a) && (this.vocab[c] = a, n.add(a), c = this.vocab.indexOf("", c + 1), c === -1))
           break;
     }
@@ -59,34 +59,34 @@ class b extends k {
       this.cache.set(n, a);
     }), this.emit("trainStatus", "trained"), this.vocabSize;
   }
-  async tokenise(s, t) {
+  tokenise(i, t) {
     if (!this.trained)
       throw new Error("Tokeniser not trained");
-    return s.map((i) => t ? i.split("").map((h) => this.cache.get(h) ?? this.unkToken) : i.split("").map((h) => {
+    return i.map((s) => t ? s.split("").map((h) => this.cache.get(h) ?? this.unkToken) : s.split("").map((h) => {
       const o = this.cache.get(h);
       return o !== void 0 ? this.vocab[o] : "";
     }));
   }
-  async detokenise(s) {
-    return s.map((e) => e.map((i) => this.vocab[i]).join(""));
+  detokenise(i) {
+    return i.map((e) => Array.from(e).map((s) => this.vocab[s] || "").join(""));
   }
-  async encode(s) {
-    return (await this.tokenise([s], !0))[0];
+  encode(i) {
+    return this.tokenise([i], !0)[0];
   }
-  async decode(s) {
-    return (await this.detokenise([s]))[0];
+  decode(i) {
+    return this.detokenise([i])[0];
   }
   getVocab() {
     return this.vocab;
   }
-  async getMerges() {
+  getMerges() {
     return [];
   }
-  async createTrainingData(s, t = 5) {
-    const e = await this.tokenise(s, !0), i = [], h = [];
+  async createTrainingData(i, t = 5) {
+    const e = await this.tokenise(i, !0), s = [], h = [];
     for (let o = 0; o < e.length - t; o++)
-      i.push(...e[o].slice(0, t)), h.push(e[o + 1][0]);
-    return [i, h];
+      s.push(...e[o].slice(0, t)), h.push(e[o + 1][0]);
+    return [s, h];
   }
 }
 export {

package/dist/tokeniser/bpe.d.ts CHANGED Viewed

@@ -16,12 +16,12 @@ export default class BPETokeniser extends BaseTokeniser {
     get unkToken(): number;
     train(text: string[]): Promise<number>;
     getVocab(): string[];
-    getMerges(): Promise<[string, string][]>;
+    getMerges(): [string, string][];
     private tokeniseWord;
     private tokeniseStrings;
-    tokenise(text: string[], numeric: true): Promise<number[][]>;
-    tokenise(text: string[]): Promise<string[][]>;
-    detokenise(tokens: number[][]): Promise<string[]>;
-    encode(text: string): Promise<number[]>;
-    decode(tokens: number[]): Promise<string>;
+    tokenise(text: string[], numeric: true): number[][];
+    tokenise(text: string[]): string[][];
+    detokenise(tokens: number[][]): string[];
+    encode(text: string): number[];
+    decode(tokens: number[]): string;
 }

package/dist/tokeniser/bpe.js CHANGED Viewed

@@ -53,7 +53,7 @@ function v(o, e) {
     o.tokens[s] = n;
   }), o.pairs.delete(u(e.a, e.b));
 }
-class T extends d {
+class x extends d {
   targetSize;
   vocab = /* @__PURE__ */ new Set();
   vocabIndex = /* @__PURE__ */ new Map();
@@ -116,7 +116,7 @@ class T extends d {
   getVocab() {
     return Array.from(this.vocab);
   }
-  async getMerges() {
+  getMerges() {
     return this.merges;
   }
   tokeniseWord(e) {
@@ -128,21 +128,21 @@ class T extends d {
   tokeniseStrings(e) {
     return e.map((s) => l(s).map((r) => this.pretokenMap.has(r) ? this.pretokenMap.get(r) : this.tokeniseWord(r)).flat(1));
   }
-  async tokenise(e, s) {
+  tokenise(e, s) {
     const t = this.tokeniseStrings(e);
     return s ? t.map((n) => n.map((r) => this.vocabIndex.get(r) ?? this.unkToken)) : t.map((n) => n.map((r) => this.vocab.has(r) ? r : ""));
   }
-  async detokenise(e) {
+  detokenise(e) {
     const s = this.getVocab();
     return e.map((n) => n.map((r) => s[r]).join(""));
   }
-  async encode(e) {
-    return (await this.tokenise([e], !0))[0];
+  encode(e) {
+    return this.tokenise([e], !0)[0];
   }
-  async decode(e) {
-    return (await this.detokenise([e]))[0];
+  decode(e) {
+    return this.detokenise([e])[0];
   }
 }
 export {
-  T as default
+  x as default
 };

package/dist/tokeniser/type.d.ts CHANGED Viewed

@@ -6,16 +6,14 @@ export interface Conversation {
 }
 export interface ITokeniser extends EE<'trainStatus'> {
     train(text: string[]): Promise<number>;
-    tokenise(text: string[], numeric?: boolean): Promise<string[][] | number[][]>;
-    detokenise(tokens: string[][] | number[][]): Promise<string[]>;
     getVocab(): string[];
-    getMerges(): Promise<[string, string][]>;
+    getMerges(): [string, string][];
     destroy(): void;
-    encode(text: string): Promise<number[]>;
-    encodeConversation(conversation: Conversation[], completion?: boolean): Promise<number[]>;
-    encodeSequence(text: string): Promise<number[]>;
-    decode(tokens: number[]): Promise<string>;
-    decodeConversation(tokens: number[]): Promise<Conversation[]>;
+    encode(text: string): number[];
+    encodeConversation(conversation: Conversation[], completion?: boolean): number[];
+    encodeSequence(text: string): number[];
+    decode(tokens: number[] | Uint16Array): string;
+    decodeConversation(tokens: number[] | Uint16Array): Conversation[];
     vocabSize: number;
     eosToken: number;
     bosToken: number;

package/dist/training/Adam.js CHANGED Viewed

@@ -1,7 +1,7 @@
 import { adamAdjust as b } from "../ops/adamAdjust.js";
 import { adamMoments as d } from "../ops/adamMoments.js";
-import { O as g, e as h, t as o, d as B } from "../index-D0RBWjq8.js";
-import { z as M } from "../zeros-DeiE2zTa.js";
+import { O as g, e as h, t as o, d as B } from "../index-Duu1Lvvv.js";
+import { z as M } from "../zeros-5YROwwUH.js";
 class R extends g {
   constructor(t, a, e, s, i = null) {
     super(), this.learningRate = t, this.beta1 = a, this.beta2 = e, this.lossScaling = s, this.epsilon = i, this.accBeta1 = a, this.accBeta2 = e, i === null && (this.epsilon = h().backend.epsilon());

package/dist/training/AdamExt.js CHANGED Viewed

@@ -1,4 +1,4 @@
-import { m as r, b as c, c as h, e as o } from "../index-D0RBWjq8.js";
+import { m as r, b as c, c as h, e as o } from "../index-Duu1Lvvv.js";
 import { AdamOptimizer as g } from "./Adam.js";
 class y extends g {
   constructor(t, e, s, i, a) {

package/dist/training/DatasetBuilder.d.ts CHANGED Viewed

@@ -8,7 +8,7 @@ export declare class DatasetBuilder {
     blockSize: number;
     private pageSize;
     constructor(tokenizer: ITokeniser, blockSize?: number);
-    createTextDataset(flatTokens: number[], batchSize?: number, masked?: Set<number>, invertMask?: boolean): Promise<Dataset<{
+    createTextDataset(flatTokens: Uint16Array, batchSize?: number, masked?: Set<number>, invertMask?: boolean): Promise<Dataset<{
         xs: Tensor;
         ys: Tensor;
     }>>;

package/dist/training/DatasetBuilder.js CHANGED Viewed

@@ -1,63 +1,63 @@
-import { t as z } from "../index-D0RBWjq8.js";
-import { d as u, i as f } from "../dataset-DcjWqUVQ.js";
+import { t as y } from "../index-Duu1Lvvv.js";
+import { d as g, i as z } from "../dataset-CVIJu7Xa.js";
 import "../index-Cp39cXWe.js";
-function S(a) {
-  return u(async () => {
+function b(a) {
+  return g(async () => {
     const t = await a();
-    return f(() => t.next());
+    return z(() => t.next());
   });
 }
-const b = 8;
-async function y(a, t) {
-  return (await Promise.all(a.map((r) => t.encodeConversation(r)))).flat();
+const f = 8;
+async function w(a, t) {
+  return (await Promise.all(a.map((s) => t.encodeConversation(s)))).flat();
 }
-class x {
+class m {
   tokenizer;
   blockSize;
   pageSize;
-  constructor(t, s = 128) {
-    this.tokenizer = t, this.blockSize = s, this.pageSize = s * b;
+  constructor(t, r = 128) {
+    this.tokenizer = t, this.blockSize = r, this.pageSize = r * f;
   }
   // Create dataset from text files
-  async createTextDataset(t, s = 32, i, r) {
+  async createTextDataset(t, r = 32, i, s) {
     if (t.length < this.blockSize + 1)
       throw new Error(`Not enough tokens (${t.length}) for block size ${this.blockSize}`);
     if (i && i.size > t.length / this.pageSize / 2)
       throw new Error("Too many masked pages - would leave insufficient training data");
     const l = (function* () {
-      if (i && r) {
+      if (i && s) {
         const e = Array.from(i);
         for (; ; ) {
-          const n = Math.floor(Math.random() * e.length), h = Math.floor(Math.random() * this.pageSize), o = e[n] * this.pageSize + h;
-          if (o + this.blockSize + 1 > t.length)
+          const o = Math.floor(Math.random() * e.length), h = Math.floor(Math.random() * this.pageSize), n = e[o] * this.pageSize + h;
+          if (n + this.blockSize + 1 > t.length)
             continue;
-          const c = t.slice(o, o + this.blockSize), g = t.slice(o + 1, o + this.blockSize + 1);
-          yield { xs: c, ys: g };
+          const c = new Int32Array(t.subarray(n, n + this.blockSize)), u = new Int32Array(t.subarray(n + 1, n + this.blockSize + 1));
+          yield { xs: c, ys: u };
         }
       } else
         for (; ; ) {
           const e = Math.floor(Math.random() * (t.length - this.blockSize - 1));
           if (i) {
-            const o = Math.floor(e / this.pageSize), c = i.has(o);
-            if (c && !r || !c && r)
+            const n = Math.floor(e / this.pageSize), c = i.has(n);
+            if (c && !s || !c && s)
               continue;
           }
-          const n = t.slice(e, e + this.blockSize), h = t.slice(e + 1, e + this.blockSize + 1);
-          yield { xs: n, ys: h };
+          const o = new Int32Array(t.subarray(e, e + this.blockSize)), h = new Int32Array(t.subarray(e + 1, e + this.blockSize + 1));
+          yield { xs: o, ys: h };
         }
     }).bind(this);
-    return S(l).batch(s).map((e) => {
-      const n = e;
-      return z(() => ({
-        xs: n.xs.cast("int32"),
-        ys: n.ys.cast("int32")
+    return b(l).batch(r).map((e) => {
+      const o = e;
+      return y(() => ({
+        xs: o.xs.cast("int32"),
+        ys: o.ys.cast("int32")
         // this.tf.oneHot(batchData.ys.cast('int32'), this.tokenizer.vocabSize),
       }));
     }).prefetch(2);
   }
 }
 export {
-  x as DatasetBuilder,
-  b as PAGE_FACTOR,
-  y as flattenTokens
+  m as DatasetBuilder,
+  f as PAGE_FACTOR,
+  w as flattenTokens
 };

package/dist/training/FullTrainer.js CHANGED Viewed

@@ -1,6 +1,6 @@
 import b from "./Trainer.js";
 import L from "./Evaluator.js";
-import { d as w } from "../index-D0RBWjq8.js";
+import { d as w } from "../index-Duu1Lvvv.js";
 import y from "../utilities/profile.js";
 import { createTensorStatistics as D } from "../checks/weights.js";
 const T = {

package/dist/training/Trainer.d.ts CHANGED Viewed

@@ -1,11 +1,12 @@
-import { Conversation, ITokeniser } from '../tokeniser/type';
+import { ITokeniser } from '../tokeniser/type';
 import { DatasetBuilder } from './DatasetBuilder';
 import { default as AdamExt } from './AdamExt';
-import { NamedTensorMap, TensorContainer } from '@tensorflow/tfjs-core/dist/tensor_types';
+import { NamedTensorMap } from '@tensorflow/tfjs-core/dist/tensor_types';
 import { Scalar, Tensor } from '@tensorflow/tfjs-core';
 import { Dataset } from '@tensorflow/tfjs-data';
 import { default as Model, ModelForwardAttributes } from '../models/model';
 import { TensorStatistics } from '../checks/weights';
+import { Task } from './tasks/Task';
 export interface TrainingLogEntry {
     loss: number;
     valLoss?: number;
@@ -93,7 +94,7 @@ export default abstract class GPTTrainer {
         log: TrainingLogEntry;
         progress: TrainingProgress;
     }>;
-    createTrainValidationSplit(textData: Conversation[][], batchSize?: number, validationSplit?: number): Promise<{
+    createTrainValidationSplit(tasks: Task[], batchSize?: number, validationSplit?: number): Promise<{
         trainDataset: Dataset<{
             xs: Tensor;
             ys: Tensor;
@@ -102,7 +103,7 @@ export default abstract class GPTTrainer {
             xs: Tensor;
             ys: Tensor;
         }>;
+        size: number;
     }>;
-    createDataset(textData: Conversation[][], batchSize?: number): Promise<Dataset<TensorContainer>>;
     dispose(): void;
 }

package/dist/training/Trainer.js CHANGED Viewed

@@ -1,10 +1,11 @@
-import { DatasetBuilder as f, flattenTokens as h, PAGE_FACTOR as y } from "./DatasetBuilder.js";
+import { DatasetBuilder as u, PAGE_FACTOR as f } from "./DatasetBuilder.js";
 import z from "./AdamExt.js";
-import { t as S, v as k, k as x, d as p, b as m } from "../index-D0RBWjq8.js";
-import { z as g } from "../zeros-DeiE2zTa.js";
-class M {
+import { t as S, v as y, k, d as h, b as p } from "../index-Duu1Lvvv.js";
+import { tokensFromTasks as x } from "./tasks/Task.js";
+import { z as m } from "../zeros-5YROwwUH.js";
+class B {
   constructor(t, e, s = 1e-3) {
-    this.tokenizer = e, this.model = t, this.lossScaling = t.lossScaling, this.learningRate = s, this.resetOptimizer(), this.datasetBuilder = new f(e, t.config.blockSize);
+    this.tokenizer = e, this.model = t, this.lossScaling = t.lossScaling, this.learningRate = s, this.resetOptimizer(), this.datasetBuilder = new u(e, t.config.blockSize);
   }
   model;
   optimizer;
@@ -53,8 +54,8 @@ class M {
   trainStep(t, e, s = !1, i = !1) {
     return S(() => {
       this.model.getProfiler()?.startMemory();
-      const { xs: a, ys: l } = e, c = () => {
-        const [n, d] = this.model.forward(
+      const { xs: a, ys: l } = e, d = () => {
+        const [o, c] = this.model.forward(
           {
             training: !0,
             checkpointing: this._gradientCheckpointing,
@@ -63,15 +64,15 @@ class M {
           a,
           l
         );
-        n.dispose();
-        const u = d.mul(m(this.lossScaling));
-        return d.dispose(), u;
-      }, { value: o, grads: r } = k(c);
-      return s ? this.model.getProfiler()?.endMemory("Training") : (this.optimizer.applyGradients(r), this.model.getProfiler()?.endMemory("Training"), i ? (t.gradients = r, Object.values(r).forEach((n) => x(n))) : p(r)), o.mul(m(1 / this.lossScaling));
+        o.dispose();
+        const g = c.mul(p(this.lossScaling));
+        return c.dispose(), g;
+      }, { value: n, grads: r } = y(d);
+      return s ? this.model.getProfiler()?.endMemory("Training") : (this.optimizer.applyGradients(r), this.model.getProfiler()?.endMemory("Training"), i ? (t.gradients = r, Object.values(r).forEach((o) => k(o))) : h(r)), n.mul(p(1 / this.lossScaling));
     });
   }
   async dummyPass() {
-    const t = g([1, this.model.config.blockSize], "int32"), e = g([1, this.model.config.blockSize], "int32");
+    const t = m([1, this.model.config.blockSize], "int32"), e = m([1, this.model.config.blockSize], "int32");
     try {
       const s = this.trainStep({}, { xs: t, ys: e }, !0);
       await s.data(), s.dispose();
@@ -86,34 +87,30 @@ class M {
       const i = this.trainStep(t, e, !1, s);
       return e.xs.dispose(), e.ys.dispose(), t.step++, t.totalSteps++, i;
     } catch (i) {
-      throw console.error(`Error processing batch at step ${t.step}:`, i), p(), i;
+      throw console.error(`Error processing batch at step ${t.step}:`, i), h(), i;
     }
   }
   async createTrainValidationSplit(t, e = 32, s = 0.1) {
-    const i = await h(t, this.tokenizer), a = /* @__PURE__ */ new Set();
+    const i = await x(t, this.tokenizer), a = /* @__PURE__ */ new Set();
     if (s > 0) {
-      const o = Math.floor(i.length / (this.datasetBuilder.blockSize * y)), r = Math.max(1, Math.floor(o * s));
+      const n = Math.floor(i.length / (this.datasetBuilder.blockSize * f)), r = Math.max(1, Math.floor(n * s));
       for (; a.size < r; ) {
-        const n = Math.floor(Math.random() * o);
-        a.add(n);
+        const o = Math.floor(Math.random() * n);
+        a.add(o);
       }
     }
-    const l = await this.datasetBuilder.createTextDataset(i, e, a, !1), c = await this.datasetBuilder.createTextDataset(
+    const l = await this.datasetBuilder.createTextDataset(i, e, a, !1), d = await this.datasetBuilder.createTextDataset(
       i,
       e,
       a,
       !0
     );
-    return { trainDataset: l, validationDataset: c };
-  }
-  async createDataset(t, e = 32) {
-    const s = await h(t, this.tokenizer);
-    return await this.datasetBuilder.createTextDataset(s, e);
+    return { trainDataset: l, validationDataset: d, size: i.length };
   }
   dispose() {
     this.optimizer && this.optimizer.dispose();
   }
 }
 export {
-  M as default
+  B as default
 };

package/dist/training/sparseCrossEntropy.js CHANGED Viewed

@@ -1,8 +1,8 @@
 import { gatherSub as x } from "../ops/gatherSub.js";
 import { scatterSub as L } from "../ops/scatterSub.js";
-import { a2 as C, t as u, a3 as E, c as G } from "../index-D0RBWjq8.js";
-import { s as y } from "../softmax-faLoUZVT.js";
-import { m as z, l as v } from "../log_sum_exp-VLZgbFAH.js";
+import { a1 as C, t as u, a2 as E, c as G } from "../index-Duu1Lvvv.js";
+import { s as y } from "../softmax-DfuYyjMh.js";
+import { m as z, l as v } from "../log_sum_exp-CVqLsVLl.js";
 function k(t, s) {
   return u(() => {
     const n = t.shape[t.shape.length - 1], c = t.shape.slice(0, -1).reduce((o, e) => o * e, 1), h = t.shape.length > 2 ? t.reshape([c, n]) : t, p = s.shape.length > 1 ? s.reshape([c]).cast("int32") : s.cast("int32"), r = z(h, -1, !0), a = G(h, r), d = v(a, -1);

package/dist/training/tasks/ConversationTask.d.ts ADDED Viewed

@@ -0,0 +1,11 @@
+import { Conversation, ITokeniser } from '../../main';
+import { Task } from './Task';
+export default class ConversationTask extends Task {
+    private rawConvo;
+    private index;
+    get length(): number;
+    constructor(conversations: Conversation[][]);
+    hasMoreConversations(): boolean;
+    nextConversation(): Conversation[] | null;
+    estimateTokens(tokeniser: ITokeniser): Promise<number>;
+}

package/dist/training/tasks/ConversationTask.js ADDED Viewed

@@ -0,0 +1,26 @@
+import { Task as t } from "./Task.js";
+class s extends t {
+  rawConvo;
+  index = 0;
+  get length() {
+    return this.rawConvo.length;
+  }
+  constructor(n) {
+    super(), this.rawConvo = n;
+  }
+  hasMoreConversations() {
+    return this.index < this.rawConvo.length;
+  }
+  nextConversation() {
+    if (this.index >= this.rawConvo.length)
+      return null;
+    const n = this.rawConvo[this.index];
+    return this.index++, n;
+  }
+  async estimateTokens(n) {
+    return (await n.encodeConversation(this.rawConvo[0])).length * this.length;
+  }
+}
+export {
+  s as default
+};

package/dist/training/tasks/PretrainingTask.d.ts ADDED Viewed

@@ -0,0 +1,11 @@
+import { Conversation, ITokeniser } from '../../main';
+import { Task } from './Task';
+export default class PretrainingTask extends Task {
+    private rawText;
+    private index;
+    get length(): number;
+    constructor(texts: string[]);
+    hasMoreConversations(): boolean;
+    nextConversation(): Conversation[] | null;
+    estimateTokens(tokeniser: ITokeniser): Promise<number>;
+}

package/dist/training/tasks/PretrainingTask.js ADDED Viewed

@@ -0,0 +1,34 @@
+import { Task as e } from "./Task.js";
+class r extends e {
+  rawText;
+  index = 0;
+  get length() {
+    return this.rawText.length;
+  }
+  constructor(t) {
+    super(), this.rawText = t;
+  }
+  hasMoreConversations() {
+    return this.index < this.rawText.length;
+  }
+  nextConversation() {
+    if (this.index >= this.rawText.length)
+      return null;
+    const t = {
+      role: "assistant",
+      content: this.rawText[this.index]
+    };
+    return this.index++, [t];
+  }
+  async estimateTokens(t) {
+    return (await t.encodeConversation([
+      {
+        role: "assistant",
+        content: this.rawText[0]
+      }
+    ])).length * this.length;
+  }
+}
+export {
+  r as default
+};

package/dist/training/tasks/StartSentenceTask.d.ts ADDED Viewed

@@ -0,0 +1,12 @@
+import { Conversation, ITokeniser } from '../../main';
+import { Task } from './Task';
+export default class StartSentenceTask extends Task {
+    private rawText;
+    private index;
+    get length(): number;
+    constructor(texts: string[]);
+    hasMoreConversations(): boolean;
+    nextConversation(): Conversation[] | null;
+    private conversationFromString;
+    estimateTokens(tokeniser: ITokeniser): Promise<number>;
+}