npm - @genai-fi/nanogpt - Versions diffs - 0.10.3 → 0.11.0 - Mend

@genai-fi/nanogpt 0.10.3 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (225) hide show

package/dist/Generator.d.ts +10 -5
package/dist/Generator.js +1789 -1765
package/dist/{RealDiv-KAPDe8zB.js → RealDiv-Ds-jvL09.js} +22 -22
package/dist/{Reshape-BYkmUnAv.js → Reshape-Cd6e-Otn.js} +1 -1
package/dist/{Reshape-Zt6eb7yh.js → Reshape-Ct266DEk.js} +9 -9
package/dist/TeachableLLM.d.ts +4 -3
package/dist/TeachableLLM.js +14 -14
package/dist/Trainer.d.ts +2 -2
package/dist/Trainer.js +6 -6
package/dist/{axis_util-BaG7mf5A.js → axis_util-DofAuy0p.js} +3 -3
package/dist/backend.js +2 -2
package/dist/{backend_util-RCe-rHaj.js → backend_util-C7NWHpv7.js} +7 -7
package/dist/{backend_webgpu-DE3ACOLx.js → backend_webgpu-B0Vls736.js} +10 -10
package/dist/{broadcast_to-B3eYlZm7.js → broadcast_to-DDaNMbX7.js} +2 -2
package/dist/checks/appendCache.js +2 -2
package/dist/checks/attentionMask.js +3 -3
package/dist/checks/gelu.js +2 -2
package/dist/checks/matMulGelu.js +2 -2
package/dist/checks/normRMS.js +4 -4
package/dist/checks/normRMSGrad.js +3 -3
package/dist/checks/packUnpack.js +2 -2
package/dist/checks/qkv.js +4 -4
package/dist/checks/rope.js +2 -2
package/dist/{clip_by_value-BnO7-a88.js → clip_by_value-Dn5tzexi.js} +4 -4
package/dist/complex-DClmWqJt.js +11 -0
package/dist/{concat-BV8bt5H-.js → concat-C6X3AAlQ.js} +1 -1
package/dist/{concat_util-DpW8mL_l.js → concat_util-CHsJFZJJ.js} +1 -1
package/dist/{dataset-BcwmTGYc.js → dataset-DcjWqUVQ.js} +7 -7
package/dist/{dropout-BcvN9JYi.js → dropout-OxuaJz6z.js} +11 -11
package/dist/{expand_dims-DT4tEPwA.js → expand_dims-BzfJK2uc.js} +3 -3
package/dist/{exports_initializers-Hta_rEnm.js → exports_initializers-eS9QJ6ut.js} +1 -1
package/dist/{floor-D5QdR_le.js → floor-DIb-lN_u.js} +1 -1
package/dist/gather-BcO5UQNJ.js +9 -0
package/dist/{gelu-CjNPL4OH.js → gelu-DqTbCx5x.js} +1 -1
package/dist/{gpgpu_math-DAOmgtXR.js → gpgpu_math-CJcbnKPC.js} +2 -2
package/dist/{index-DOvlwCh-.js → index-D0RBWjq8.js} +52 -52
package/dist/{index-BwexR4lA.js → index-Dj5TkmPY.js} +89 -89
package/dist/{kernel_funcs_utils-CCzYdUZg.js → kernel_funcs_utils-CSaumNDs.js} +11 -11
package/dist/layers/BaseLayer.js +2 -2
package/dist/layers/CausalSelfAttention.js +6 -6
package/dist/layers/MLP.js +4 -4
package/dist/layers/PositionEmbedding.js +5 -5
package/dist/layers/RMSNorm.js +3 -3
package/dist/layers/RoPECache.js +4 -4
package/dist/layers/TiedEmbedding.js +6 -6
package/dist/layers/TransformerBlock.js +1 -1
package/dist/loader/loadTransformers.js +1 -1
package/dist/loader/oldZipLoad.js +17 -17
package/dist/log_sum_exp-VLZgbFAH.js +39 -0
package/dist/main.d.ts +1 -1
package/dist/main.js +9 -9
package/dist/{matMul16-BWRSOCWB.js → matMul16-cDxwemKj.js} +7 -7
package/dist/{matMulGelu-CzfgT6Wq.js → matMulGelu-B2s_80-H.js} +18 -18
package/dist/{mat_mul-SjpJRLyL.js → mat_mul-DxpNTCRz.js} +3 -3
package/dist/{mod-AnXEvvpo.js → mod-PrOKlFxH.js} +1 -1
package/dist/models/NanoGPTV1.js +2 -2
package/dist/models/model.js +9 -9
package/dist/{ones-D2rT0xk2.js → ones-BX_wEgzB.js} +3 -3
package/dist/ops/adamAdjust.js +1 -1
package/dist/ops/adamMoments.js +1 -1
package/dist/ops/add16.js +1 -1
package/dist/ops/appendCache.js +3 -3
package/dist/ops/attentionMask.js +1 -1
package/dist/ops/concat16.js +2 -2
package/dist/ops/cpu/adamAdjust.js +6 -6
package/dist/ops/cpu/adamMoments.js +2 -2
package/dist/ops/cpu/appendCache.js +5 -5
package/dist/ops/cpu/attentionMask.js +10 -10
package/dist/ops/cpu/fusedSoftmax.js +2 -2
package/dist/ops/cpu/gatherSub.js +6 -6
package/dist/ops/cpu/gelu.js +9 -9
package/dist/ops/cpu/matMul16.js +2 -2
package/dist/ops/cpu/matMulGelu.js +3 -3
package/dist/ops/cpu/matMulMul.js +1 -1
package/dist/ops/cpu/mulDropout.js +1 -1
package/dist/ops/cpu/normRMS.js +3 -3
package/dist/ops/cpu/qkv.js +3 -3
package/dist/ops/cpu/rope.js +9 -9
package/dist/ops/cpu/scatterSub.js +11 -11
package/dist/ops/dot16.js +2 -2
package/dist/ops/gatherSub.js +1 -1
package/dist/ops/gelu.js +2 -2
package/dist/ops/grads/add16.js +4 -4
package/dist/ops/grads/attentionMask.js +2 -2
package/dist/ops/grads/gelu.js +2 -2
package/dist/ops/grads/matMul16.js +3 -3
package/dist/ops/grads/matMulGelu.js +3 -3
package/dist/ops/grads/normRMS.js +7 -7
package/dist/ops/grads/pack16.js +3 -3
package/dist/ops/grads/qkv.js +6 -6
package/dist/ops/grads/rope.js +2 -2
package/dist/ops/grads/softmax16.js +1 -1
package/dist/ops/grads/unpack16.js +2 -2
package/dist/ops/matMul16.js +3 -3
package/dist/ops/matMulGelu.js +2 -2
package/dist/ops/matMulMul.js +1 -1
package/dist/ops/mul16.js +1 -1
package/dist/ops/mulDrop.js +1 -1
package/dist/ops/normRMS.js +1 -1
package/dist/ops/pack16.js +2 -2
package/dist/ops/qkv.js +1 -1
package/dist/ops/reshape16.js +6 -6
package/dist/ops/rope.js +2 -2
package/dist/ops/scatterSub.js +1 -1
package/dist/ops/slice16.js +2 -2
package/dist/ops/softmax16.js +1 -1
package/dist/ops/sub16.js +1 -1
package/dist/ops/sum16.js +2 -2
package/dist/ops/transpose16.js +3 -3
package/dist/ops/unpack16.js +2 -2
package/dist/ops/webgl/adamAdjust.js +2 -2
package/dist/ops/webgl/adamMoments.js +1 -1
package/dist/ops/webgl/appendCache.js +1 -1
package/dist/ops/webgl/attentionMask.js +4 -4
package/dist/ops/webgl/fusedSoftmax.js +6 -6
package/dist/ops/webgl/gatherSub.js +1 -1
package/dist/ops/webgl/gelu.js +2 -2
package/dist/ops/webgl/log.js +3 -3
package/dist/ops/webgl/matMul16.js +11 -11
package/dist/ops/webgl/matMulGelu.js +4 -4
package/dist/ops/webgl/matMulMul.js +7 -7
package/dist/ops/webgl/mulDropout.js +1 -1
package/dist/ops/webgl/normRMS.js +7 -7
package/dist/ops/webgl/qkv.js +1 -1
package/dist/ops/webgl/rope.js +4 -4
package/dist/ops/webgl/scatterSub.js +1 -1
package/dist/ops/webgpu/adamAdjust.js +3 -3
package/dist/ops/webgpu/adamMoments.js +3 -3
package/dist/ops/webgpu/add16.js +1 -1
package/dist/ops/webgpu/appendCache.js +3 -3
package/dist/ops/webgpu/attentionMask.js +5 -5
package/dist/ops/webgpu/attentionMask32_program.js +2 -2
package/dist/ops/webgpu/concat16.js +5 -5
package/dist/ops/webgpu/gatherSub.js +5 -5
package/dist/ops/webgpu/gelu.js +3 -3
package/dist/ops/webgpu/matMul16.js +18 -18
package/dist/ops/webgpu/matMul16_program.js +2 -2
package/dist/ops/webgpu/mul16.js +4 -4
package/dist/ops/webgpu/normRMS.js +6 -6
package/dist/ops/webgpu/normRMSGrad.js +4 -4
package/dist/ops/webgpu/pack16.js +1 -1
package/dist/ops/webgpu/pack16_program.js +2 -2
package/dist/ops/webgpu/qkv.js +6 -6
package/dist/ops/webgpu/rope.js +3 -3
package/dist/ops/webgpu/scatterSub.js +3 -3
package/dist/ops/webgpu/slice16.js +4 -4
package/dist/ops/webgpu/softmax16.js +2 -2
package/dist/ops/webgpu/softmax16_program.js +2 -2
package/dist/ops/webgpu/softmax16_subgroup_program.js +2 -2
package/dist/ops/webgpu/softmax16grad.js +1 -1
package/dist/ops/webgpu/sub16.js +4 -4
package/dist/ops/webgpu/sum16.js +6 -6
package/dist/ops/webgpu/transpose16.js +2 -2
package/dist/ops/webgpu/transpose16_program.js +2 -2
package/dist/ops/webgpu/transpose16_shared_program.js +3 -3
package/dist/ops/webgpu/unpack16.js +3 -3
package/dist/ops/webgpu/utils/binary_op.js +3 -3
package/dist/ops/webgpu/utils/reductions.js +4 -4
package/dist/{ops-B5yanEdW.js → ops-FJapAPfm.js} +56 -56
package/dist/{pack16-nQ6JaLo-.js → pack16-k4jq6aMX.js} +7 -7
package/dist/patches/webgpu_backend.js +7 -7
package/dist/patches/webgpu_base.js +1 -1
package/dist/patches/webgpu_program.js +8 -8
package/dist/{random_width-or-CEftb.js → random_width-UGQn4OWb.js} +33 -33
package/dist/range-CuGvVN2c.js +10 -0
package/dist/{relu-CP0ZcxWO.js → relu-Cf80uA2p.js} +1 -1
package/dist/{reshape-ByE68wS9.js → reshape-CkjKPPqB.js} +1 -1
package/dist/{resize_nearest_neighbor-B19mCEg2.js → resize_nearest_neighbor-DB8k9KN_.js} +43 -43
package/dist/{rope-Ir4mTyD1.js → rope-BmZmp9uP.js} +1 -1
package/dist/{scatter_nd_util-lvSiX8q4.js → scatter_nd_util-BY22Cc-C.js} +1 -1
package/dist/{selu_util-kbhpTdYD.js → selu_util-BuLbmbrl.js} +5 -5
package/dist/{shared-DT1TkE6w.js → shared-B7USJZgw.js} +1 -1
package/dist/{shared-dntlHIDQ.js → shared-BQboIImQ.js} +86 -86
package/dist/{slice-BfEGSH82.js → slice-Aqy7KbJh.js} +3 -3
package/dist/{slice_util-uTKwiEpW.js → slice_util-D8CQRenR.js} +7 -7
package/dist/{softmax-CA5jFsLR.js → softmax-faLoUZVT.js} +1 -1
package/dist/{split-CVLc0w--.js → split-BNz5jcGc.js} +3 -3
package/dist/{squeeze-C7Z2srUo.js → squeeze--YMgaAAf.js} +2 -2
package/dist/{stack-Cf4n9h0N.js → stack-WJK22CFn.js} +1 -1
package/dist/{step-CINUs5QB.js → step-dXR33iOg.js} +32 -32
package/dist/sum-BdplSvq_.js +11 -0
package/dist/tensor-BQqrDvpx.js +8 -0
package/dist/tensor1d-LxP9asMm.js +11 -0
package/dist/{tensor2d-Bs9wZRc7.js → tensor2d-BN1sSfQO.js} +3 -3
package/dist/{tensor4d-BARPdTaS.js → tensor4d-DVwr7pLF.js} +1 -1
package/dist/{tfjs_backend-y1cvNhLA.js → tfjs_backend-Vi4JfLzT.js} +28 -28
package/dist/{tile-mbfagpsB.js → tile-CvN_LyVr.js} +4 -4
package/dist/tokeniser/BaseTokeniser.d.ts +27 -0
package/dist/tokeniser/BaseTokeniser.js +94 -0
package/dist/tokeniser/CharTokeniser.d.ts +4 -3
package/dist/tokeniser/CharTokeniser.js +46 -32
package/dist/tokeniser/bpe.d.ts +4 -3
package/dist/tokeniser/bpe.js +60 -45
package/dist/tokeniser/type.d.ts +11 -0
package/dist/training/Adam.js +2 -2
package/dist/training/AdamExt.js +1 -1
package/dist/training/DatasetBuilder.d.ts +2 -2
package/dist/training/DatasetBuilder.js +32 -36
package/dist/training/FullTrainer.js +1 -1
package/dist/training/Trainer.d.ts +3 -3
package/dist/training/Trainer.js +2 -2
package/dist/training/sparseCrossEntropy.js +3 -3
package/dist/{transpose-ClWiBS_b.js → transpose-JawVKyZy.js} +5 -5
package/dist/{unsorted_segment_sum-BDDhB_E6.js → unsorted_segment_sum-LAbmE9G4.js} +78 -78
package/dist/utilities/dummy.js +3 -3
package/dist/utilities/multinomialCPU.js +2 -2
package/dist/utilities/packed.js +1 -1
package/dist/utilities/performance.js +1 -1
package/dist/utilities/profile.js +1 -1
package/dist/utilities/safetensors.js +2 -2
package/dist/utilities/sentences.js +5 -5
package/dist/utilities/weights.js +2 -2
package/dist/{variable-WawDEaAb.js → variable-DQ9yYgEU.js} +1 -1
package/dist/{webgpu_program-DuOXPQol.js → webgpu_program-CAE4RICo.js} +3 -3
package/dist/{webgpu_util-RxEF33Rj.js → webgpu_util-BdovYhXr.js} +1 -1
package/dist/{zeros-KnWaWf-X.js → zeros-DeiE2zTa.js} +2 -2
package/dist/{zeros_like-DvE73F4e.js → zeros_like-BAz3iKru.js} +77 -77
package/package.json +1 -1
package/dist/complex-DjxcVmoX.js +0 -11
package/dist/gather-D3JcZUaI.js +0 -9
package/dist/log_sum_exp-ngO0-4pK.js +0 -39
package/dist/range-BklejeeW.js +0 -10
package/dist/sum-DWAtNGez.js +0 -11
package/dist/tensor-DJoc7gJU.js +0 -8
package/dist/tensor1d-D11P_7Dp.js +0 -11

package/dist/tokeniser/CharTokeniser.js CHANGED Viewed

@@ -1,66 +1,80 @@
-import { E as k } from "../index-DvYrXKkX.js";
+import k, { SPECIALS as d } from "./BaseTokeniser.js";
 const u = ["<eos>", "<unk>"];
 class b extends k {
   vocabSize = 0;
   eosToken = 0;
+  bosToken = 0;
   unkToken = 0;
   vocab = [];
   cache = /* @__PURE__ */ new Map();
   _trained = !1;
-  constructor(t) {
-    if (super(), Array.isArray(t)) {
-      if (this.vocab = t, this.vocab.length > 0)
-        this.vocabSize = this.vocab.length, this.eosToken = this.vocab.indexOf("<eos>"), this.unkToken = this.vocab.indexOf(""), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf("<unk>")), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf("<pad>")), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf("_")), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf(" ")), this.unkToken === -1 && (this.unkToken = this.eosToken), this.vocab = this.vocab.map((e) => e === "<pad>" ? "" : e), this.vocab.forEach((e, n) => {
-          this.cache.set(e, n);
+  constructor(s) {
+    if (super(), Array.isArray(s)) {
+      if (this.vocab = s, this.vocab.length > 0)
+        this.vocabSize = this.vocab.length, d.forEach((t) => {
+          const e = this.vocab.indexOf(t);
+          e !== -1 && this.addSpecialToken(t, e);
+        }), this.eosToken = this.getSpecialTokenIndex("<eos>"), this.bosToken = this.getSpecialTokenIndex("<bos>") ?? this.eosToken, this.unkToken = this.getSpecialTokenIndex("") ?? -1, this.unkToken === -1 && (this.unkToken = this.vocab.indexOf("<unk>")), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf("<pad>")), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf("_")), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf(" ")), this.unkToken === -1 && (this.unkToken = this.eosToken), this.vocab = this.vocab.map((t) => t === "<pad>" ? "" : t), this.vocab.forEach((t, e) => {
+          this.cache.set(t, e);
         });
       else
         throw new Error("Vocab cannot be empty");
       this._trained = !0;
     } else
-      this.vocabSize = t, this.vocab = new Array(this.vocabSize).fill(""), this.vocab[0] = "<eos>", this.vocab[1] = "", this.eosToken = 0, this.unkToken = 1, this.cache.set("<eos>", 0), this.cache.set("", 1);
+      this.vocabSize = s, this.vocab = new Array(this.vocabSize).fill(""), this.addSpecialTokens(), this.eosToken = this.getSpecialTokenIndex("<eos>"), this.bosToken = this.getSpecialTokenIndex("<bos>") ?? this.eosToken, this.unkToken = this.getSpecialTokenIndex(""), this.vocab.forEach((t, e) => {
+        this.cache.set(t, e);
+      }), this.cache.set("", this.unkToken);
+  }
+  addToken(s, t) {
+    if (this.cache.has(s))
+      return this.cache.get(s);
+    let e;
+    if (t !== void 0 ? e = t : (e = this.vocab.indexOf("", this.unkToken + 1), e === -1 && (e = this.vocabSize)), e >= this.vocabSize)
+      throw new Error("Vocab size exceeded");
+    return this.vocab[e] = s, this.cache.set(s, e), e;
   }
   get trained() {
     return this.vocab.length === this.vocabSize && this._trained;
   }
   destroy() {
   }
-  async train(t) {
-    const e = t.map((i) => i.split("")).flat(), n = new Set(e), s = Array.from(n), h = this.vocab.indexOf("", this.unkToken + 1), o = this.vocabSize - u.length;
+  async train(s) {
+    const t = s.map((n) => n.split("")).flat(), e = new Set(t), i = Array.from(e), h = this.vocab.indexOf("", this.unkToken + 1), o = this.vocabSize - u.length;
     if (h === -1)
       return this.vocabSize;
-    if (this._trained = !0, s.length > o) {
-      const i = /* @__PURE__ */ new Map();
-      e.forEach((a) => {
-        i.set(a, (i.get(a) || 0) + 1);
-      }), s.sort((a, r) => (i.get(a) || 0) - (i.get(r) || 0)), s.splice(0, s.length - o);
+    if (this._trained = !0, i.length > o) {
+      const n = /* @__PURE__ */ new Map();
+      t.forEach((a) => {
+        n.set(a, (n.get(a) || 0) + 1);
+      }), i.sort((a, r) => (n.get(a) || 0) - (n.get(r) || 0)), i.splice(0, i.length - o);
     }
     let c = h;
     if (c !== -1) {
-      const i = new Set(this.vocab);
-      for (const a of s)
-        if (!i.has(a) && (this.vocab[c] = a, i.add(a), c = this.vocab.indexOf("", c + 1), c === -1))
+      const n = new Set(this.vocab);
+      for (const a of i)
+        if (!n.has(a) && (this.vocab[c] = a, n.add(a), c = this.vocab.indexOf("", c + 1), c === -1))
           break;
     }
-    return this.cache.clear(), this.vocab.forEach((i, a) => {
-      this.cache.set(i, a);
+    return this.cache.clear(), this.vocab.forEach((n, a) => {
+      this.cache.set(n, a);
     }), this.emit("trainStatus", "trained"), this.vocabSize;
   }
-  async tokenise(t, e) {
+  async tokenise(s, t) {
     if (!this.trained)
       throw new Error("Tokeniser not trained");
-    return t.map((s) => e ? s.split("").map((h) => this.cache.get(h) ?? this.unkToken) : s.split("").map((h) => {
+    return s.map((i) => t ? i.split("").map((h) => this.cache.get(h) ?? this.unkToken) : i.split("").map((h) => {
       const o = this.cache.get(h);
       return o !== void 0 ? this.vocab[o] : "";
     }));
   }
-  async detokenise(t) {
-    return t.map((n) => n.map((s) => this.vocab[s]).join(""));
+  async detokenise(s) {
+    return s.map((e) => e.map((i) => this.vocab[i]).join(""));
   }
-  async encode(t) {
-    return (await this.tokenise([t], !0))[0];
+  async encode(s) {
+    return (await this.tokenise([s], !0))[0];
   }
-  async decode(t) {
-    return (await this.detokenise([t]))[0];
+  async decode(s) {
+    return (await this.detokenise([s]))[0];
   }
   getVocab() {
     return this.vocab;
@@ -68,11 +82,11 @@ class b extends k {
   async getMerges() {
     return [];
   }
-  async createTrainingData(t, e = 5) {
-    const n = await this.tokenise(t, !0), s = [], h = [];
-    for (let o = 0; o < n.length - e; o++)
-      s.push(...n[o].slice(0, e)), h.push(n[o + 1][0]);
-    return [s, h];
+  async createTrainingData(s, t = 5) {
+    const e = await this.tokenise(s, !0), i = [], h = [];
+    for (let o = 0; o < e.length - t; o++)
+      i.push(...e[o].slice(0, t)), h.push(e[o + 1][0]);
+    return [i, h];
   }
 }
 export {

package/dist/tokeniser/bpe.d.ts CHANGED Viewed

@@ -1,6 +1,5 @@
-import { default as EE } from 'eventemitter3';
-import { ITokeniser } from './type';
-export default class BPETokeniser extends EE<'trainStatus'> implements ITokeniser {
+import { default as BaseTokeniser } from './BaseTokeniser';
+export default class BPETokeniser extends BaseTokeniser {
     private targetSize;
     private vocab;
     private vocabIndex;
@@ -8,10 +7,12 @@ export default class BPETokeniser extends EE<'trainStatus'> implements ITokenise
     private pretokenMap;
     constructor(vocabSize: number);
     constructor(vocab: string[], merges?: [string, string][]);
+    addToken(token: string, index?: number): number;
     destroy(): void;
     get trained(): boolean;
     get vocabSize(): number;
     get eosToken(): number;
+    get bosToken(): number;
     get unkToken(): number;
     train(text: string[]): Promise<number>;
     getVocab(): string[];

package/dist/tokeniser/bpe.js CHANGED Viewed

@@ -1,68 +1,80 @@
 import l from "../utilities/tokenParse.js";
-import { E as f } from "../index-DvYrXKkX.js";
+import d, { SPECIALS as f } from "./BaseTokeniser.js";
 function u(o, e) {
   return `${o}-::-${e}`;
 }
-function k(o) {
+function b(o) {
   const e = /* @__PURE__ */ new Map();
   for (let s = 0; s < o.length; s++) {
     const t = o[s];
-    for (let r = 0; r < t.length - 1; r++) {
-      const n = u(t[r], t[r + 1]), a = e.get(n) || {
-        a: t[r],
-        b: t[r + 1],
+    for (let n = 0; n < t.length - 1; n++) {
+      const r = u(t[n], t[n + 1]), i = e.get(r) || {
+        a: t[n],
+        b: t[n + 1],
         count: 0,
         instances: /* @__PURE__ */ new Set()
       };
-      a.count += 1, a.instances.add(s), e.set(n, a);
+      i.count += 1, i.instances.add(s), e.set(r, i);
     }
   }
   return { pairs: e, tokens: o };
 }
-function h(o, e, s, t, r) {
-  const n = u(e, s);
-  if (o.pairs.has(n)) {
-    const a = o.pairs.get(n);
-    a.count += r, r > 0 ? a.instances.add(t) : a.count <= 0 ? o.pairs.delete(n) : a.instances.delete(t);
+function h(o, e, s, t, n) {
+  const r = u(e, s);
+  if (o.pairs.has(r)) {
+    const i = o.pairs.get(r);
+    i.count += n, n > 0 ? i.instances.add(t) : i.count <= 0 ? o.pairs.delete(r) : i.instances.delete(t);
   } else
-    o.pairs.set(n, { a: e, b: s, count: r, instances: /* @__PURE__ */ new Set([t]) });
+    o.pairs.set(r, { a: e, b: s, count: n, instances: /* @__PURE__ */ new Set([t]) });
 }
-function b(o) {
+function k(o) {
   let e = null, s = 0;
   for (const t of o.pairs.values())
     t.count > s && (s = t.count, e = t);
   return e;
 }
-function d(o, e) {
+function m(o, e) {
   return o.map((s) => {
     const t = [];
-    for (let r = 0; r < s.length; r++)
-      r < s.length - 1 && s[r] === e[0] && s[r + 1] === e[1] ? (t.push(e[0] + e[1]), r++) : t.push(s[r]);
+    for (let n = 0; n < s.length; n++)
+      n < s.length - 1 && s[n] === e[0] && s[n + 1] === e[1] ? (t.push(e[0] + e[1]), n++) : t.push(s[n]);
     return t;
   });
 }
-function m(o, e) {
+function v(o, e) {
   e.instances.forEach((s) => {
-    const t = o.tokens[s], r = [];
-    for (let n = 0; n < t.length; n++)
-      if (n < t.length - 1 && t[n] === e.a && t[n + 1] === e.b) {
-        const a = e.a + e.b;
-        r.push(a), n > 0 && (h(o, t[n - 1], e.a, s, -1), h(o, t[n - 1], a, s, 1)), n++, n < t.length - 1 && (h(o, e.b, t[n + 1], s, -1), h(o, a, t[n + 1], s, 1));
+    const t = o.tokens[s], n = [];
+    for (let r = 0; r < t.length; r++)
+      if (r < t.length - 1 && t[r] === e.a && t[r + 1] === e.b) {
+        const i = e.a + e.b;
+        n.push(i), r > 0 && (h(o, t[r - 1], e.a, s, -1), h(o, t[r - 1], i, s, 1)), r++, r < t.length - 1 && (h(o, e.b, t[r + 1], s, -1), h(o, i, t[r + 1], s, 1));
       } else
-        r.push(t[n]);
-    o.tokens[s] = r;
+        n.push(t[r]);
+    o.tokens[s] = n;
   }), o.pairs.delete(u(e.a, e.b));
 }
-class S extends f {
+class T extends d {
   targetSize;
   vocab = /* @__PURE__ */ new Set();
   vocabIndex = /* @__PURE__ */ new Map();
   merges = [];
   pretokenMap = /* @__PURE__ */ new Map();
   constructor(e, s) {
-    super(), Array.isArray(e) ? (e.forEach((t, r) => {
-      this.vocab.add(t), this.vocabIndex.set(t, r);
-    }), s && (this.merges = s), this.targetSize = e.length) : (this.vocab.add("<eos>"), this.vocab.add(""), this.targetSize = e);
+    super(), Array.isArray(e) ? (e.forEach((t, n) => {
+      this.vocab.add(t), this.vocabIndex.set(t, n);
+    }), s && (this.merges = s), this.targetSize = e.length, f.forEach((t) => {
+      const n = e.indexOf(t);
+      n !== -1 && this.addSpecialToken(t, n);
+    })) : (this.addSpecialTokens(), this.targetSize = e);
+  }
+  addToken(e, s) {
+    if (this.vocab.has(e))
+      return this.vocabIndex.get(e);
+    {
+      this.vocab.add(e);
+      const t = s !== void 0 ? s : this.vocab.size - 1;
+      return this.vocabIndex.set(e, t), t;
+    }
   }
   destroy() {
     this.vocab.clear(), this.vocabIndex.clear(), this.merges = [], this.pretokenMap.clear();
@@ -76,26 +88,29 @@ class S extends f {
   get eosToken() {
     return this.vocabIndex.get("<eos>") ?? 0;
   }
+  get bosToken() {
+    return this.vocabIndex.get("<bos>") ?? 0;
+  }
   get unkToken() {
     return this.vocabIndex.get("") ?? 1;
   }
   async train(e) {
-    const s = e.map((i) => l(i)).flat(1), t = new Set(s);
-    this.vocab = /* @__PURE__ */ new Set(), this.pretokenMap.clear(), this.merges = [], this.vocab.add("<eos>"), this.vocab.add("");
-    const r = Array.from(t), n = r.map((i) => Array.from(i).map((c) => (this.vocab.add(c), c))), a = k(n);
+    const s = e.map((a) => l(a)).flat(1), t = new Set(s);
+    this.vocab = /* @__PURE__ */ new Set(), this.pretokenMap.clear(), this.merges = [], this.addSpecialTokens();
+    const n = Array.from(t), r = n.map((a) => Array.from(a).map((c) => (this.vocab.add(c), c))), i = b(r);
     for (; this.vocab.size < this.targetSize && this.merges.length < this.targetSize; ) {
-      const i = b(a);
-      if (!i)
+      const a = k(i);
+      if (!a)
         break;
-      this.merges.push([i.a, i.b]), this.vocab.add(i.a + i.b), m(a, i);
+      this.merges.push([a.a, a.b]), this.vocab.add(a.a + a.b), v(i, a);
     }
-    r.forEach((i, p) => {
-      const c = n[p];
-      this.pretokenMap.set(i, c);
+    n.forEach((a, p) => {
+      const c = r[p];
+      this.pretokenMap.set(a, c);
     }), this.vocabIndex.clear();
     let g = 0;
-    for (const i of this.vocab.keys())
-      this.vocabIndex.set(i, g++);
+    for (const a of this.vocab.keys())
+      this.vocabIndex.set(a, g++);
     return this.emit("trainStatus", "trained"), this.vocab.size;
   }
   getVocab() {
@@ -107,19 +122,19 @@ class S extends f {
   tokeniseWord(e) {
     let s = Array.from(e);
     return this.merges.forEach((t) => {
-      s = d([s], t)[0];
+      s = m([s], t)[0];
     }), this.pretokenMap.set(e, s), s;
   }
   tokeniseStrings(e) {
-    return e.map((s) => l(s).map((n) => this.pretokenMap.has(n) ? this.pretokenMap.get(n) : this.tokeniseWord(n)).flat(1));
+    return e.map((s) => l(s).map((r) => this.pretokenMap.has(r) ? this.pretokenMap.get(r) : this.tokeniseWord(r)).flat(1));
   }
   async tokenise(e, s) {
     const t = this.tokeniseStrings(e);
-    return s ? t.map((r) => r.map((n) => this.vocabIndex.get(n) ?? this.unkToken)) : t.map((r) => r.map((n) => this.vocab.has(n) ? n : ""));
+    return s ? t.map((n) => n.map((r) => this.vocabIndex.get(r) ?? this.unkToken)) : t.map((n) => n.map((r) => this.vocab.has(r) ? r : ""));
   }
   async detokenise(e) {
     const s = this.getVocab();
-    return e.map((r) => r.map((n) => s[n]).join(""));
+    return e.map((n) => n.map((r) => s[r]).join(""));
   }
   async encode(e) {
     return (await this.tokenise([e], !0))[0];
@@ -129,5 +144,5 @@ class S extends f {
   }
 }
 export {
-  S as default
+  T as default
 };

package/dist/tokeniser/type.d.ts CHANGED Viewed

@@ -1,4 +1,9 @@
 import { default as EE } from 'eventemitter3';
+export type Roles = 'user' | 'assistant' | 'system';
+export interface Conversation {
+    role: Roles;
+    content: string;
+}
 export interface ITokeniser extends EE<'trainStatus'> {
     train(text: string[]): Promise<number>;
     tokenise(text: string[], numeric?: boolean): Promise<string[][] | number[][]>;
@@ -7,8 +12,14 @@ export interface ITokeniser extends EE<'trainStatus'> {
     getMerges(): Promise<[string, string][]>;
     destroy(): void;
     encode(text: string): Promise<number[]>;
+    encodeConversation(conversation: Conversation[], completion?: boolean): Promise<number[]>;
+    encodeSequence(text: string): Promise<number[]>;
     decode(tokens: number[]): Promise<string>;
+    decodeConversation(tokens: number[]): Promise<Conversation[]>;
     vocabSize: number;
     eosToken: number;
+    bosToken: number;
     trained: boolean;
+    getSpecialTokenIndex(token: string): number | undefined;
+    isSpecialToken(index: number): boolean;
 }

package/dist/training/Adam.js CHANGED Viewed

@@ -1,7 +1,7 @@
 import { adamAdjust as b } from "../ops/adamAdjust.js";
 import { adamMoments as d } from "../ops/adamMoments.js";
-import { O as g, e as h, t as o, d as B } from "../index-DOvlwCh-.js";
-import { z as M } from "../zeros-KnWaWf-X.js";
+import { O as g, e as h, t as o, d as B } from "../index-D0RBWjq8.js";
+import { z as M } from "../zeros-DeiE2zTa.js";
 class R extends g {
   constructor(t, a, e, s, i = null) {
     super(), this.learningRate = t, this.beta1 = a, this.beta2 = e, this.lossScaling = s, this.epsilon = i, this.accBeta1 = a, this.accBeta2 = e, i === null && (this.epsilon = h().backend.epsilon());

package/dist/training/AdamExt.js CHANGED Viewed

@@ -1,4 +1,4 @@
-import { m as r, b as c, c as h, e as o } from "../index-DOvlwCh-.js";
+import { m as r, b as c, c as h, e as o } from "../index-D0RBWjq8.js";
 import { AdamOptimizer as g } from "./Adam.js";
 class y extends g {
   constructor(t, e, s, i, a) {

package/dist/training/DatasetBuilder.d.ts CHANGED Viewed

@@ -1,8 +1,8 @@
 import { Tensor } from '@tensorflow/tfjs-core';
-import { ITokeniser } from '../tokeniser/type';
+import { Conversation, ITokeniser } from '../tokeniser/type';
 import { Dataset } from '@tensorflow/tfjs-data';
 export declare const PAGE_FACTOR = 8;
-export declare function flattenTokens(textData: string[], tokenizer: ITokeniser): Promise<number[]>;
+export declare function flattenTokens(textData: Conversation[][], tokenizer: ITokeniser): Promise<number[]>;
 export declare class DatasetBuilder {
     tokenizer: ITokeniser;
     blockSize: number;

package/dist/training/DatasetBuilder.js CHANGED Viewed

@@ -1,67 +1,63 @@
-import { t as g } from "../index-DOvlwCh-.js";
-import { d as u, i as d } from "../dataset-BcwmTGYc.js";
+import { t as z } from "../index-D0RBWjq8.js";
+import { d as u, i as f } from "../dataset-DcjWqUVQ.js";
 import "../index-Cp39cXWe.js";
-function z(r) {
+function S(a) {
   return u(async () => {
-    const t = await r();
-    return d(() => t.next());
+    const t = await a();
+    return f(() => t.next());
   });
 }
-const S = 8;
-async function y(r, t) {
-  const s = await Promise.all(r.map((e) => t.encode(e))), o = t.eosToken >= 0, a = s.map((e) => o ? [...e, t.eosToken] : e).flat();
-  for (const e of a)
-    if (e < 0 || e >= t.vocabSize)
-      throw new Error(`Invalid token index ${e} found in tokenised data`);
-  return a;
+const b = 8;
+async function y(a, t) {
+  return (await Promise.all(a.map((r) => t.encodeConversation(r)))).flat();
 }
-class w {
+class x {
   tokenizer;
   blockSize;
   pageSize;
   constructor(t, s = 128) {
-    this.tokenizer = t, this.blockSize = s, this.pageSize = s * S;
+    this.tokenizer = t, this.blockSize = s, this.pageSize = s * b;
   }
   // Create dataset from text files
-  async createTextDataset(t, s = 32, o, a) {
+  async createTextDataset(t, s = 32, i, r) {
     if (t.length < this.blockSize + 1)
       throw new Error(`Not enough tokens (${t.length}) for block size ${this.blockSize}`);
-    if (o && o.size > t.length / this.pageSize / 2)
+    if (i && i.size > t.length / this.pageSize / 2)
       throw new Error("Too many masked pages - would leave insufficient training data");
-    const e = (function* () {
-      if (o && a) {
-        const i = Array.from(o);
+    const l = (function* () {
+      if (i && r) {
+        const e = Array.from(i);
         for (; ; ) {
-          const c = Math.floor(Math.random() * i.length), l = Math.floor(Math.random() * this.pageSize), n = i[c] * this.pageSize + l;
-          if (n + this.blockSize + 1 > t.length)
+          const n = Math.floor(Math.random() * e.length), h = Math.floor(Math.random() * this.pageSize), o = e[n] * this.pageSize + h;
+          if (o + this.blockSize + 1 > t.length)
             continue;
-          const h = t.slice(n, n + this.blockSize), f = t.slice(n + 1, n + this.blockSize + 1);
-          yield { xs: h, ys: f };
+          const c = t.slice(o, o + this.blockSize), g = t.slice(o + 1, o + this.blockSize + 1);
+          yield { xs: c, ys: g };
         }
       } else
         for (; ; ) {
-          const i = Math.floor(Math.random() * (t.length - this.blockSize - 1));
-          if (o) {
-            const n = Math.floor(i / this.pageSize), h = o.has(n);
-            if (h && !a || !h && a)
+          const e = Math.floor(Math.random() * (t.length - this.blockSize - 1));
+          if (i) {
+            const o = Math.floor(e / this.pageSize), c = i.has(o);
+            if (c && !r || !c && r)
               continue;
           }
-          const c = t.slice(i, i + this.blockSize), l = t.slice(i + 1, i + this.blockSize + 1);
-          yield { xs: c, ys: l };
+          const n = t.slice(e, e + this.blockSize), h = t.slice(e + 1, e + this.blockSize + 1);
+          yield { xs: n, ys: h };
         }
     }).bind(this);
-    return z(e).batch(s).map((i) => {
-      const c = i;
-      return g(() => ({
-        xs: c.xs.cast("int32"),
-        ys: c.ys.cast("int32")
+    return S(l).batch(s).map((e) => {
+      const n = e;
+      return z(() => ({
+        xs: n.xs.cast("int32"),
+        ys: n.ys.cast("int32")
         // this.tf.oneHot(batchData.ys.cast('int32'), this.tokenizer.vocabSize),
       }));
     }).prefetch(2);
   }
 }
 export {
-  w as DatasetBuilder,
-  S as PAGE_FACTOR,
+  x as DatasetBuilder,
+  b as PAGE_FACTOR,
   y as flattenTokens
 };

package/dist/training/FullTrainer.js CHANGED Viewed

@@ -1,6 +1,6 @@
 import b from "./Trainer.js";
 import L from "./Evaluator.js";
-import { d as w } from "../index-DOvlwCh-.js";
+import { d as w } from "../index-D0RBWjq8.js";
 import y from "../utilities/profile.js";
 import { createTensorStatistics as D } from "../checks/weights.js";
 const T = {

package/dist/training/Trainer.d.ts CHANGED Viewed

@@ -1,4 +1,4 @@
-import { ITokeniser } from '../tokeniser/type';
+import { Conversation, ITokeniser } from '../tokeniser/type';
 import { DatasetBuilder } from './DatasetBuilder';
 import { default as AdamExt } from './AdamExt';
 import { NamedTensorMap, TensorContainer } from '@tensorflow/tfjs-core/dist/tensor_types';
@@ -93,7 +93,7 @@ export default abstract class GPTTrainer {
         log: TrainingLogEntry;
         progress: TrainingProgress;
     }>;
-    createTrainValidationSplit(textData: string[], batchSize?: number, validationSplit?: number): Promise<{
+    createTrainValidationSplit(textData: Conversation[][], batchSize?: number, validationSplit?: number): Promise<{
         trainDataset: Dataset<{
             xs: Tensor;
             ys: Tensor;
@@ -103,6 +103,6 @@ export default abstract class GPTTrainer {
             ys: Tensor;
         }>;
     }>;
-    createDataset(textData: string[], batchSize?: number): Promise<Dataset<TensorContainer>>;
+    createDataset(textData: Conversation[][], batchSize?: number): Promise<Dataset<TensorContainer>>;
     dispose(): void;
 }

package/dist/training/Trainer.js CHANGED Viewed

@@ -1,7 +1,7 @@
 import { DatasetBuilder as f, flattenTokens as h, PAGE_FACTOR as y } from "./DatasetBuilder.js";
 import z from "./AdamExt.js";
-import { t as S, v as k, k as x, d as p, b as m } from "../index-DOvlwCh-.js";
-import { z as g } from "../zeros-KnWaWf-X.js";
+import { t as S, v as k, k as x, d as p, b as m } from "../index-D0RBWjq8.js";
+import { z as g } from "../zeros-DeiE2zTa.js";
 class M {
   constructor(t, e, s = 1e-3) {
     this.tokenizer = e, this.model = t, this.lossScaling = t.lossScaling, this.learningRate = s, this.resetOptimizer(), this.datasetBuilder = new f(e, t.config.blockSize);

package/dist/training/sparseCrossEntropy.js CHANGED Viewed

@@ -1,8 +1,8 @@
 import { gatherSub as x } from "../ops/gatherSub.js";
 import { scatterSub as L } from "../ops/scatterSub.js";
-import { a6 as C, t as u, a7 as E, c as G } from "../index-DOvlwCh-.js";
-import { s as y } from "../softmax-CA5jFsLR.js";
-import { m as z, l as v } from "../log_sum_exp-ngO0-4pK.js";
+import { a2 as C, t as u, a3 as E, c as G } from "../index-D0RBWjq8.js";
+import { s as y } from "../softmax-faLoUZVT.js";
+import { m as z, l as v } from "../log_sum_exp-VLZgbFAH.js";
 function k(t, s) {
   return u(() => {
     const n = t.shape[t.shape.length - 1], c = t.shape.slice(0, -1).reduce((o, e) => o * e, 1), h = t.shape.length > 2 ? t.reshape([c, n]) : t, p = s.shape.length > 1 ? s.reshape([c]).cast("int32") : s.cast("int32"), r = z(h, -1, !0), a = G(h, r), d = v(a, -1);

package/dist/{transpose-ClWiBS_b.js → transpose-JawVKyZy.js} RENAMED Viewed

@@ -1,5 +1,5 @@
-import { A as u, B as i, E as o, ap as $, aq as g, ar as m, l, t as x, as as p } from "./index-DOvlwCh-.js";
-import { c as k } from "./complex-DjxcVmoX.js";
+import { q as u, u as i, E as o, ap as $, aq as g, ar as m, y as l, t as x, as as p } from "./index-D0RBWjq8.js";
+import { c as k } from "./complex-DClmWqJt.js";
 function K(r) {
   const e = { input: i(r, "input", "imag") };
   return o.runKernel($, e);
@@ -15,7 +15,7 @@ function b(r) {
   return o.runKernel(m, e);
 }
 const d = /* @__PURE__ */ u({ real_: b });
-function N(r, t, e) {
+function y(r, t, e) {
   const n = i(r, "x", "transpose");
   if (t == null && (t = n.shape.map((s, a) => a).reverse()), l(n.rank === t.length, () => `Error in transpose: rank of input ${n.rank} must match length of perm ${t}.`), t.forEach((s) => {
     l(s >= 0 && s < n.rank, () => `All entries in 'perm' must be between 0 and ${n.rank - 1} but got ${t}`);
@@ -27,10 +27,10 @@ function N(r, t, e) {
     return s = o.runKernel(p, { x: s }, c), a = o.runKernel(p, { x: a }, c), e && (a = _(a)), k(s, a);
   }) : o.runKernel(p, f, c);
 }
-const y = /* @__PURE__ */ u({ transpose_: N });
+const q = /* @__PURE__ */ u({ transpose_: y });
 export {
   h as i,
   _ as n,
   d as r,
-  y as t
+  q as t
 };