@genai-fi/nanogpt 0.23.0 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/dist/{DatasetBuilder-C0iJT29K.js → DatasetBuilder-DU1G1OKX.js} +20 -23
  2. package/dist/Generator.js +2 -2
  3. package/dist/TeachableLLM.d.ts +1 -2
  4. package/dist/TeachableLLM.js +1 -1
  5. package/dist/{Trainer-DBsyWJ4s.js → Trainer-Cr7csbTD.js} +1 -1
  6. package/dist/Trainer.d.ts +2 -2
  7. package/dist/Trainer.js +1 -1
  8. package/dist/data/stream.d.ts +6 -6
  9. package/dist/data/stream.js +1 -1
  10. package/dist/data/textLoader.js +1 -1
  11. package/dist/loader/load.js +2 -2
  12. package/dist/loader/loadHF.js +1 -1
  13. package/dist/loader/loadTransformers.js +1 -1
  14. package/dist/loader/newZipLoad.js +1 -1
  15. package/dist/loader/oldZipLoad.js +1 -1
  16. package/dist/loader/save.js +1 -1
  17. package/dist/{main-BSaDGH7I.js → main-Dz72vadm.js} +1489 -1496
  18. package/dist/main.d.ts +2 -10
  19. package/dist/main.js +20 -20
  20. package/dist/models/NanoGPTV1.js +1 -1
  21. package/dist/models/NanoGPTV2.js +1 -1
  22. package/dist/models/factory.js +1 -1
  23. package/dist/models/model.js +1 -1
  24. package/dist/{stream-BjdpSNqB.js → stream-BpAwcvHz.js} +565 -561
  25. package/dist/tokeniser/CharTokeniser.js +18 -20
  26. package/dist/tokeniser/bpe.js +18 -22
  27. package/dist/training/DatasetBuilder.d.ts +4 -4
  28. package/dist/training/DatasetBuilder.js +1 -1
  29. package/dist/training/PreTrainer.js +1 -1
  30. package/dist/training/SFTTrainer.js +1 -1
  31. package/dist/training/tasks/TokenStore.d.ts +2 -1
  32. package/dist/training/tasks/TokenStore.js +8 -5
  33. package/dist/training/tasks/tokenStream.d.ts +16 -0
  34. package/dist/training/tasks/tokenStream.js +46 -0
  35. package/dist/training/validation.js +4 -2
  36. package/dist/utilities/random.d.ts +1 -0
  37. package/dist/utilities/random.js +19 -0
  38. package/package.json +1 -1
  39. package/dist/training/tasks/ConversationTask.d.ts +0 -17
  40. package/dist/training/tasks/ConversationTask.js +0 -29
  41. package/dist/training/tasks/PretrainingTask.d.ts +0 -17
  42. package/dist/training/tasks/PretrainingTask.js +0 -42
  43. package/dist/training/tasks/StartSentenceTask.d.ts +0 -18
  44. package/dist/training/tasks/StartSentenceTask.js +0 -45
  45. package/dist/training/tasks/Task.d.ts +0 -29
  46. package/dist/training/tasks/Task.js +0 -50
  47. package/dist/training/tasks/splitter.d.ts +0 -5
  48. package/dist/training/tasks/splitter.js +0 -18
@@ -1,7 +1,6 @@
1
- import { yieldIfNeeded as e } from "../utilities/yielder.js";
2
- import t, { SPECIALS as n } from "./BaseTokeniser.js";
1
+ import e, { SPECIALS as t } from "./BaseTokeniser.js";
3
2
  //#region lib/tokeniser/CharTokeniser.ts
4
- var r = ["<eos>", "<unk>"], i = class extends t {
3
+ var n = ["<eos>", "<unk>"], r = class extends e {
5
4
  vocabSize = 0;
6
5
  eosToken = 0;
7
6
  bosToken = 0;
@@ -11,7 +10,7 @@ var r = ["<eos>", "<unk>"], i = class extends t {
11
10
  _trained = !1;
12
11
  constructor(e) {
13
12
  if (super(), Array.isArray(e)) {
14
- if (this.vocab = e, this.vocab.length > 0) this.vocabSize = this.vocab.length, n.forEach((e) => {
13
+ if (this.vocab = e, this.vocab.length > 0) this.vocabSize = this.vocab.length, t.forEach((e) => {
15
14
  let t = this.vocab.indexOf(e);
16
15
  t !== -1 && this.addSpecialToken(e, t);
17
16
  }), this.eosToken = this.getSpecialTokenIndex("<eos>"), this.bosToken = this.getSpecialTokenIndex("<bos>") ?? this.eosToken, this.unkToken = this.getSpecialTokenIndex("") ?? -1, this.unkToken === -1 && (this.unkToken = this.vocab.indexOf("<unk>")), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf("<pad>")), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf("_")), this.unkToken === -1 && (this.unkToken = this.vocab.indexOf(" ")), this.unkToken === -1 && (this.unkToken = this.eosToken), this.vocab = this.vocab.map((e) => e === "<pad>" ? "" : e), this.vocab.forEach((e, t) => {
@@ -35,22 +34,21 @@ var r = ["<eos>", "<unk>"], i = class extends t {
35
34
  destroy() {
36
35
  this.cache.clear(), this.vocab = [];
37
36
  }
38
- async train(t, n, i) {
39
- this.datasetID = i;
40
- let a = /* @__PURE__ */ new Set(), o = performance.now();
41
- for (let r of t) {
42
- let t = r.cursor(), i = await t.next();
43
- for (; i !== null;) i.forEach((e) => {
44
- for (let t of e.content) a.add(t);
45
- }), o = await e(o, n, 0), i = await t.next();
46
- }
47
- let s = Array.from(a), c = this.vocab.indexOf("", this.unkToken + 1), l = this.vocabSize - r.length;
48
- if (c === -1) return this.generateID(), this.vocabSize;
49
- if (this._trained = !0, s.length > l) throw Error("too_small_vocab");
50
- let u = c;
51
- if (u !== -1) {
37
+ async train(e, t, r) {
38
+ this.datasetID = r;
39
+ let i = /* @__PURE__ */ new Set();
40
+ for (let n of e) await n.begin((e) => {
41
+ e.forEach((e) => {
42
+ for (let t of e.content) i.add(t);
43
+ });
44
+ }, t ? () => t(i.size) : void 0);
45
+ let a = Array.from(i), o = this.vocab.indexOf("", this.unkToken + 1), s = this.vocabSize - n.length;
46
+ if (o === -1) return this.generateID(), this.vocabSize;
47
+ if (this._trained = !0, a.length > s) throw Error("too_small_vocab");
48
+ let c = o;
49
+ if (c !== -1) {
52
50
  let e = new Set(this.vocab);
53
- for (let t of s) if (!e.has(t) && (this.vocab[u] = t, e.add(t), u = this.vocab.indexOf("", u + 1), u === -1)) break;
51
+ for (let t of a) if (!e.has(t) && (this.vocab[c] = t, e.add(t), c = this.vocab.indexOf("", c + 1), c === -1)) break;
54
52
  }
55
53
  return this.cache.clear(), this.vocab.forEach((e, t) => {
56
54
  this.cache.set(e, t);
@@ -85,4 +83,4 @@ var r = ["<eos>", "<unk>"], i = class extends t {
85
83
  }
86
84
  };
87
85
  //#endregion
88
- export { i as default };
86
+ export { r as default };
@@ -1,5 +1,5 @@
1
- import { yieldIfNeeded as e } from "../utilities/yielder.js";
2
- import t, { SPECIALS as n } from "./BaseTokeniser.js";
1
+ import e, { SPECIALS as t } from "./BaseTokeniser.js";
2
+ import { yieldIfNeeded as n } from "../utilities/yielder.js";
3
3
  import r from "../utilities/tokenParse.js";
4
4
  //#region lib/tokeniser/bpe.ts
5
5
  function i(e, t) {
@@ -58,16 +58,16 @@ function l(e, t) {
58
58
  e.tokens[n] = i;
59
59
  }), e.pairs.delete(i(t.a, t.b));
60
60
  }
61
- var u = class extends t {
61
+ var u = class extends e {
62
62
  targetSize;
63
63
  vocab = /* @__PURE__ */ new Set();
64
64
  vocabIndex = /* @__PURE__ */ new Map();
65
65
  merges = [];
66
66
  pretokenMap = /* @__PURE__ */ new Map();
67
- constructor(e, t) {
67
+ constructor(e, n) {
68
68
  super(), Array.isArray(e) ? (e.forEach((e, t) => {
69
69
  this.vocab.add(e), this.vocabIndex.set(e, t);
70
- }), t && (this.merges = t), this.targetSize = e.length, n.forEach((t) => {
70
+ }), n && (this.merges = n), this.targetSize = e.length, t.forEach((t) => {
71
71
  let n = e.indexOf(t);
72
72
  n !== -1 && this.addSpecialToken(t, n);
73
73
  })) : (this.addSpecialTokens(), this.targetSize = e);
@@ -84,7 +84,7 @@ var u = class extends t {
84
84
  this.vocab.clear(), this.vocabIndex.clear(), this.merges = [], this.pretokenMap.clear();
85
85
  }
86
86
  get trained() {
87
- return this.vocab.size > n.length && this.vocab.size <= this.targetSize;
87
+ return this.vocab.size > t.length && this.vocab.size <= this.targetSize;
88
88
  }
89
89
  get vocabSize() {
90
90
  return this.vocab.size;
@@ -98,26 +98,22 @@ var u = class extends t {
98
98
  get unkToken() {
99
99
  return this.vocabIndex.get("") ?? 1;
100
100
  }
101
- async train(t = [], n, i) {
101
+ async train(e = [], t, i) {
102
102
  this.datasetID = i;
103
- let o = performance.now(), c = /* @__PURE__ */ new Set();
103
+ let o = /* @__PURE__ */ new Set();
104
104
  this.vocab = /* @__PURE__ */ new Set(), this.pretokenMap.clear(), this.merges = [], this.addSpecialTokens();
105
- for (let i of t) {
106
- let t = i.cursor(), a = await t.next();
107
- for (; a !== null;) {
108
- for (let e of a) {
109
- let t = r(e.content);
110
- for (let e of t) c.has(e) || (c.add(e), Array.from(e).forEach((e) => this.vocab.add(e)));
111
- }
112
- o = await e(o, n, this.vocab.size), a = await t.next();
105
+ for (let n of e) await n.begin((e) => {
106
+ for (let t of e) {
107
+ let e = r(t.content);
108
+ for (let t of e) o.has(t) || (o.add(t), Array.from(t).forEach((e) => this.vocab.add(e)));
113
109
  }
114
- }
115
- let u = Array.from(c), d = u.map((e) => Array.from(e)), f = a(d);
116
- if (o = await e(o, n, this.vocab.size), this.vocab.size >= this.targetSize) throw console.warn("Initial vocab size is greater than or equal to target size. No merges will be performed.", this.vocab.size, this.targetSize), Error("too_small_vocab");
110
+ }, t ? () => t(this.vocab.size) : void 0);
111
+ let c = performance.now(), u = Array.from(o), d = u.map((e) => Array.from(e)), f = a(d);
112
+ if (c = await n(c, t, this.vocab.size), this.vocab.size >= this.targetSize) throw console.warn("Initial vocab size is greater than or equal to target size. No merges will be performed.", this.vocab.size, this.targetSize), Error("too_small_vocab");
117
113
  for (; this.vocab.size < this.targetSize && this.merges.length < this.targetSize;) {
118
- let t = s(f);
119
- if (!t) break;
120
- this.merges.push([t.a, t.b]), this.vocab.add(t.a + t.b), l(f, t), o = await e(o, n, this.vocab.size);
114
+ let e = s(f);
115
+ if (!e) break;
116
+ this.merges.push([e.a, e.b]), this.vocab.add(e.a + e.b), l(f, e), c = await n(c, t, this.vocab.size);
121
117
  }
122
118
  u.forEach((e, t) => {
123
119
  let n = d[t];
@@ -7,11 +7,11 @@ export declare function flattenTokensWithMask(textData: Conversation[][], tokeni
7
7
  tokens: Uint16Array;
8
8
  mask: Uint8Array;
9
9
  };
10
- export declare function shuffle(array: Uint32Array): Uint32Array;
10
+ export declare function shuffle(array: Uint32Array | Uint16Array): Uint32Array | Uint16Array;
11
11
  export interface DatasetState {
12
- shuffledShards: Uint32Array;
13
- shuffledIndexes: Uint32Array;
14
- lastShardIndexes: Uint32Array;
12
+ shuffledShards: Uint16Array;
13
+ shuffledIndexes: Uint16Array;
14
+ lastShardIndexes: Uint16Array;
15
15
  currentShard: Uint16Array | null;
16
16
  nextShard: Uint16Array | null;
17
17
  currentMask: Uint8Array | null;
@@ -1,2 +1,2 @@
1
- import { a as e, i as t, n, r, t as i } from "../DatasetBuilder-C0iJT29K.js";
1
+ import { a as e, i as t, n, r, t as i } from "../DatasetBuilder-DU1G1OKX.js";
2
2
  export { i as DatasetBuilder, n as flattenTokens, r as flattenTokensWithMask, t as moveToNext, e as shuffle };
@@ -1,4 +1,4 @@
1
- import { t as e } from "../DatasetBuilder-C0iJT29K.js";
1
+ import { t as e } from "../DatasetBuilder-DU1G1OKX.js";
2
2
  import t from "./BasicTrainer.js";
3
3
  //#region lib/training/PreTrainer.ts
4
4
  var n = {
@@ -1,4 +1,4 @@
1
- import { t as e } from "../DatasetBuilder-C0iJT29K.js";
1
+ import { t as e } from "../DatasetBuilder-DU1G1OKX.js";
2
2
  import t from "./BasicTrainer.js";
3
3
  //#region lib/training/SFTTrainer.ts
4
4
  var n = {
@@ -31,7 +31,8 @@ export declare class TokenStore {
31
31
  getMask(index: number): Promise<Uint8Array | undefined>;
32
32
  getShardCount(): number;
33
33
  getTokenCount(): number;
34
- appendShard(shard: Uint16Array, mask?: Uint8Array): Promise<void>;
34
+ finish(): Promise<void>;
35
+ appendShard(shard: Uint16Array, mask?: Uint8Array): void;
35
36
  dispose(): Promise<void>;
36
37
  clear(): Promise<void>;
37
38
  }
@@ -7,12 +7,12 @@ var e = "llm-tokenstore", t = class {
7
7
  masks;
8
8
  shardCount = 0;
9
9
  lastShardLength = -1;
10
- _shardSize = 4e3 * 1024;
10
+ _shardSize = 8e3 * 1024;
11
11
  dirHandle = null;
12
12
  opfsAvailable = !1;
13
13
  lru = /* @__PURE__ */ new Map();
14
14
  maxCachedShards;
15
- constructor(t, n, r, i = 8, a = 4e3 * 1024) {
15
+ constructor(t, n, r, i = 2, a = 8e3 * 1024) {
16
16
  this.tokeniserId = t, this.datasetId = n, this.maxCachedShards = Math.max(1, Math.trunc(i)), this._shardSize = Math.max(1, Math.trunc(a)), this.name = r ?? e;
17
17
  }
18
18
  get shardSize() {
@@ -148,12 +148,15 @@ var e = "llm-tokenstore", t = class {
148
148
  getTokenCount() {
149
149
  return this.shardCount === 0 ? 0 : (this.shardCount - 1) * this._shardSize + this.lastShardLength;
150
150
  }
151
- async appendShard(e, t) {
151
+ async finish() {
152
+ await this.writeManifest();
153
+ }
154
+ appendShard(e, t) {
152
155
  if (this.lastShardLength >= 0 && this.lastShardLength < this._shardSize) throw Error("Previous shard was not full");
153
156
  let n = this.shards.length;
154
157
  this.shards.push(e), !this.masks && t && (this.masks = Array(n).fill(null)), this.masks && this.masks.push(t ?? null), this.shardCount = this.shards.length, this.lastShardLength = e.length, this.touchLRU(n);
155
158
  try {
156
- this.opfsAvailable && this.dirHandle && (await this.writeShardToOPFS(n, e), t && await this.writeMaskToOPFS(n, t), await this.writeManifest());
159
+ this.opfsAvailable && this.dirHandle && (this.writeShardToOPFS(n, e), t && this.writeMaskToOPFS(n, t));
157
160
  } catch (e) {
158
161
  console.error(e), this.maxCachedShards = 2 ** 53 - 1;
159
162
  }
@@ -208,7 +211,7 @@ async function a(e, r, i, a) {
208
211
  let o = n.get(e);
209
212
  if (o && o.tokeniserId === r && o.datasetId === i) return o;
210
213
  o && (await o.dispose(), n.delete(e));
211
- let s = new t(r, i, e, a?.maxCachedShards ?? 8, a?.shardSize ?? 4e3 * 1024);
214
+ let s = new t(r, i, e, a?.maxCachedShards, a?.shardSize);
212
215
  return a?.noOPFS || await s.init(), n.set(e, s), s;
213
216
  }
214
217
  //#endregion
@@ -0,0 +1,16 @@
1
+ import { ConversationStream, ITokeniser } from '../../../main';
2
+ import { TokenStore } from './TokenStore';
3
+ interface TokensFromTasksOptions {
4
+ masking?: boolean;
5
+ maxCachedShards?: number;
6
+ noOPFS?: boolean;
7
+ shardSize?: number;
8
+ validationSplit?: number;
9
+ validationSeed?: string | number;
10
+ cb?: (tokens: number) => void;
11
+ }
12
+ export declare function tokensFromStreams(tasks: ConversationStream[], tokenizer: ITokeniser, options?: TokensFromTasksOptions): Promise<{
13
+ trainingTokens: TokenStore;
14
+ validationTokens?: TokenStore;
15
+ }>;
16
+ export {};
@@ -0,0 +1,46 @@
1
+ import { createTokenStore as e, deleteTokenStore as t } from "./TokenStore.js";
2
+ import { seededRng as n } from "../../utilities/random.js";
3
+ //#region lib/training/tasks/tokenStream.ts
4
+ function r(e, t, n, r, i, a) {
5
+ let o = n.encodeConversation(e, !1, !!a);
6
+ if (o) {
7
+ let e = Array.isArray(o) ? o : o.tokens;
8
+ r.total += e.length;
9
+ let n = t[t.length - 1], s = a ? a[a.length - 1] : null;
10
+ if (r.offset + e.length > n.length) {
11
+ let c = n.length - r.offset;
12
+ n.set(e.slice(0, c), r.offset);
13
+ let l = e.length - c;
14
+ if (l > i) throw Error(`Estimated tokens (${i}) is too small for the next batch of tokens (${l}).`);
15
+ let u = new Uint16Array(i);
16
+ if (u.set(e.slice(c), 0), t.push(u), a && s && !Array.isArray(o)) {
17
+ s.set(o.mask.slice(0, c).map((e) => +!!e), r.offset);
18
+ let e = new Uint8Array(u.length);
19
+ e.set(o.mask.slice(c).map((e) => +!!e), 0), a.push(e);
20
+ }
21
+ r.offset = e.length - c;
22
+ } else n.set(e, r.offset), s && !Array.isArray(o) && s.set(o.mask.map((e) => +!!e), r.offset), r.offset += e.length;
23
+ }
24
+ }
25
+ async function i(i, a, o) {
26
+ await t("training-tokens");
27
+ let s = await e("training-tokens", a.id, a.datasetID ?? "", o);
28
+ await t("validation-tokens");
29
+ let c = o?.validationSplit && o.validationSplit > 0 ? await e("validation-tokens", a.id, a.datasetID ?? "", o) : void 0, l = [new Uint16Array(s.shardSize)], u = o?.masking ? [new Uint8Array(s.shardSize)] : null, d = {
30
+ offset: 0,
31
+ total: 0
32
+ }, f = o?.validationSplit && o.validationSplit > 0 ? [new Uint16Array(c.shardSize)] : void 0, p = o?.masking && f ? [new Uint8Array(c.shardSize)] : null, m = {
33
+ offset: 0,
34
+ total: 0
35
+ }, h = 0, g = o?.cb, _ = o?.validationSeed === void 0 ? Math.random : n(o.validationSeed);
36
+ for (; h < i.length;) await i[h++].begin((e) => {
37
+ let t = o?.validationSplit && o.validationSplit > 0 && _() < o.validationSplit, n = t ? f : l, i = t ? m : d, h = t ? p : u, g = t ? c : s;
38
+ r(e, n, a, i, g.shardSize, h || void 0), n.length > 1 && (g.appendShard(n[0], h ? h[0] : void 0), n.shift(), h && h.shift());
39
+ }, g ? () => g(d.total) : void 0);
40
+ return l.length === 1 && (l[0] = l[0].subarray(0, d.offset), s.appendShard(l[0], u ? u[0].subarray(0, d.offset) : void 0)), f && f.length === 1 && (f[0] = f[0].subarray(0, m.offset), c.appendShard(f[0], p ? p[0].subarray(0, m.offset) : void 0)), await s.finish(), c && await c.finish(), {
41
+ trainingTokens: s,
42
+ validationTokens: f ? c : void 0
43
+ };
44
+ }
45
+ //#endregion
46
+ export { i as tokensFromStreams };
@@ -1,8 +1,10 @@
1
1
  import { TokenStore as e, createTokenStore as t } from "./tasks/TokenStore.js";
2
2
  //#region lib/training/validation.ts
3
3
  async function n(e, n) {
4
- let r = await t("training-tokens", n.id, n.datasetID ?? ""), i = e.map((e) => r.appendShard(e));
5
- return await Promise.all(i), r;
4
+ let r = await t("training-tokens", n.id, n.datasetID ?? "");
5
+ return e.forEach((e) => {
6
+ r.appendShard(e);
7
+ }), await r.finish(), r;
6
8
  }
7
9
  async function r(t, r, i, a, o) {
8
10
  let s = t instanceof e ? t : await n(t, i), c = r instanceof e ? r : await n(r, i), l = s.getTokenCount(), { dataset: u, state: d } = await a.createTextDataset(s, { batchSize: o }), { dataset: f, state: p } = await a.createTextDataset(c, {
@@ -0,0 +1 @@
1
+ export declare function seededRng(seed: string | number): () => number;
@@ -0,0 +1,19 @@
1
+ //#region lib/utilities/random.ts
2
+ function e(e) {
3
+ let t = 1779033703 ^ e.length;
4
+ for (let n = 0; n < e.length; n++) t = Math.imul(t ^ e.charCodeAt(n), 3432918353), t = t << 13 | t >>> 19;
5
+ return () => (t = Math.imul(t ^ t >>> 16, 2246822507), t = Math.imul(t ^ t >>> 13, 3266489909), (t ^= t >>> 16) >>> 0);
6
+ }
7
+ function t(e) {
8
+ let t = e >>> 0;
9
+ return () => {
10
+ t += 1831565813;
11
+ let e = Math.imul(t ^ t >>> 15, t | 1);
12
+ return e ^= e + Math.imul(e ^ e >>> 7, e | 61), ((e ^ e >>> 14) >>> 0) / 4294967296;
13
+ };
14
+ }
15
+ function n(n) {
16
+ return t(typeof n == "number" ? n >>> 0 : e(n)());
17
+ }
18
+ //#endregion
19
+ export { n as seededRng };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@genai-fi/nanogpt",
3
- "version": "0.23.0",
3
+ "version": "0.24.0",
4
4
  "type": "module",
5
5
  "main": "dist/main.js",
6
6
  "types": "dist/main.d.ts",
@@ -1,17 +0,0 @@
1
- import { Conversation, ConversationStream, ITokeniser } from '../../../main';
2
- import { Task } from './Task';
3
- export default class ConversationTask extends Task {
4
- private streams;
5
- private streamIndex;
6
- private currentCursor;
7
- get length(): number;
8
- constructor(conversations: ConversationStream[]);
9
- hasMoreConversations(): boolean;
10
- nextConversation(): Promise<Conversation[] | null>;
11
- nextTokens(tokeniser: ITokeniser): Promise<number[] | null>;
12
- nextTokens(tokeniser: ITokeniser, masking: boolean): Promise<{
13
- tokens: number[];
14
- mask: boolean[];
15
- } | null>;
16
- estimateTokens(tokeniser: ITokeniser): Promise<number>;
17
- }
@@ -1,29 +0,0 @@
1
- import { Task as e } from "./Task.js";
2
- //#region lib/training/tasks/ConversationTask.ts
3
- var t = class extends e {
4
- streams;
5
- streamIndex = 0;
6
- currentCursor = null;
7
- get length() {
8
- return this.streams.length;
9
- }
10
- constructor(e) {
11
- super(), this.streams = e;
12
- }
13
- hasMoreConversations() {
14
- return this.streamIndex < this.streams.length;
15
- }
16
- async nextConversation() {
17
- return this.streamIndex < this.streams.length ? (this.currentCursor ||= this.streams[this.streamIndex].cursor(), await this.currentCursor.next() || (this.streamIndex++, this.currentCursor = null, this.nextConversation())) : null;
18
- }
19
- async nextTokens(e, t) {
20
- let n = await this.nextConversation();
21
- return n ? e.encodeConversation(n, !1, t) : null;
22
- }
23
- async estimateTokens(e) {
24
- let t = await this.streams[0].cursor().next();
25
- return t ? e.encodeConversation(t).length * this.length : 0;
26
- }
27
- };
28
- //#endregion
29
- export { t as default };
@@ -1,17 +0,0 @@
1
- import { Conversation, ITokeniser } from '../../../main';
2
- import { Task } from './Task';
3
- export default class PretrainingTask extends Task {
4
- private rawText;
5
- private index;
6
- get length(): number;
7
- constructor(texts: string[]);
8
- hasMoreConversations(): boolean;
9
- nextConversation(): Promise<Conversation[] | null>;
10
- nextTokens(tokeniser: ITokeniser): Promise<number[] | null>;
11
- nextTokens(tokeniser: ITokeniser, masking: boolean): Promise<{
12
- tokens: number[];
13
- mask: boolean[];
14
- } | null>;
15
- shuffle(): void;
16
- estimateTokens(tokeniser: ITokeniser): Promise<number>;
17
- }
@@ -1,42 +0,0 @@
1
- import { Task as e } from "./Task.js";
2
- //#region lib/training/tasks/PretrainingTask.ts
3
- var t = class extends e {
4
- rawText;
5
- index = 0;
6
- get length() {
7
- return this.rawText.length;
8
- }
9
- constructor(e) {
10
- super(), this.rawText = e;
11
- }
12
- hasMoreConversations() {
13
- return this.index < this.rawText.length;
14
- }
15
- async nextConversation() {
16
- if (this.index >= this.rawText.length) return null;
17
- let e = {
18
- role: "assistant",
19
- content: this.rawText[this.index]
20
- };
21
- return this.index++, [e];
22
- }
23
- async nextTokens(e, t) {
24
- if (this.index >= this.rawText.length) return null;
25
- let n = e.encodeSequence(this.rawText[this.index]);
26
- return this.index++, t ? {
27
- tokens: n,
28
- mask: Array(n.length).fill(!0)
29
- } : n;
30
- }
31
- shuffle() {
32
- this.index = 0;
33
- }
34
- async estimateTokens(e) {
35
- return e.encodeConversation([{
36
- role: "assistant",
37
- content: this.rawText[0]
38
- }]).length * this.length;
39
- }
40
- };
41
- //#endregion
42
- export { t as default };
@@ -1,18 +0,0 @@
1
- import { Conversation, ITokeniser } from '../../../main';
2
- import { Task } from './Task';
3
- export default class StartSentenceTask extends Task {
4
- private rawText;
5
- private index;
6
- get length(): number;
7
- constructor(texts: string[]);
8
- hasMoreConversations(): boolean;
9
- nextConversation(): Promise<Conversation[] | null>;
10
- nextTokens(tokeniser: ITokeniser): Promise<number[] | null>;
11
- nextTokens(tokeniser: ITokeniser, masking: boolean): Promise<{
12
- tokens: number[];
13
- mask: boolean[];
14
- } | null>;
15
- shuffle(): void;
16
- private conversationFromString;
17
- estimateTokens(tokeniser: ITokeniser): Promise<number>;
18
- }
@@ -1,45 +0,0 @@
1
- import { Task as e } from "./Task.js";
2
- //#region lib/training/tasks/StartSentenceTask.ts
3
- var t = class extends e {
4
- rawText;
5
- index = 0;
6
- get length() {
7
- return this.rawText.length;
8
- }
9
- constructor(e) {
10
- super(), this.rawText = e;
11
- }
12
- hasMoreConversations() {
13
- return this.index < this.rawText.length;
14
- }
15
- async nextConversation() {
16
- if (this.index >= this.rawText.length) return null;
17
- let e = this.rawText[this.index];
18
- return this.index++, this.conversationFromString(e);
19
- }
20
- async nextTokens(e, t) {
21
- let n = await this.nextConversation();
22
- return n ? e.encodeConversation(n, !1, t) : null;
23
- }
24
- shuffle() {
25
- this.index = 0;
26
- }
27
- conversationFromString(e) {
28
- let t = e.indexOf(".");
29
- return t === -1 ? [{
30
- role: "assistant",
31
- content: this.rawText[this.index]
32
- }] : [{
33
- role: "user",
34
- content: e.slice(0, t + 1).trim()
35
- }, {
36
- role: "assistant",
37
- content: e.slice(t + 1).trim()
38
- }];
39
- }
40
- async estimateTokens(e) {
41
- return (await e.encodeConversation(this.conversationFromString(this.rawText[0]))).length * this.length;
42
- }
43
- };
44
- //#endregion
45
- export { t as default };
@@ -1,29 +0,0 @@
1
- import { Conversation, ITokeniser } from '../../../main';
2
- import { TokenStore } from './TokenStore';
3
- export declare abstract class Task {
4
- abstract get length(): number;
5
- abstract hasMoreConversations(): boolean;
6
- abstract nextConversation(): Promise<Conversation[] | null>;
7
- abstract nextTokens(tokeniser: ITokeniser): Promise<number[] | null>;
8
- abstract nextTokens(tokeniser: ITokeniser, masking: boolean): Promise<{
9
- tokens: number[];
10
- mask: boolean[];
11
- } | null>;
12
- abstract nextTokens(tokeniser: ITokeniser, masking?: boolean): Promise<number[] | {
13
- tokens: number[];
14
- mask: boolean[];
15
- } | null>;
16
- }
17
- interface TokensFromTasksOptions {
18
- masking?: boolean;
19
- maxCachedShards?: number;
20
- noOPFS?: boolean;
21
- shardSize?: number;
22
- validationSplit?: number;
23
- cb?: (tokens: number) => void;
24
- }
25
- export declare function tokensFromTasks(tasks: Task[], tokenizer: ITokeniser, options?: TokensFromTasksOptions): Promise<{
26
- trainingTokens: TokenStore;
27
- validationTokens?: TokenStore;
28
- }>;
29
- export {};
@@ -1,50 +0,0 @@
1
- import { yieldIfNeeded as e } from "../../utilities/yielder.js";
2
- import { createTokenStore as t, deleteTokenStore as n } from "./TokenStore.js";
3
- //#region lib/training/tasks/Task.ts
4
- var r = class {};
5
- async function i(e, t, n, r, i, a) {
6
- for (let o of e) {
7
- let e = await o.nextTokens(n, a ? !0 : void 0);
8
- if (e) {
9
- let n = Array.isArray(e) ? e : e.tokens;
10
- r.total += n.length;
11
- let o = t[t.length - 1], s = a ? a[a.length - 1] : null;
12
- if (r.offset + n.length > o.length) {
13
- let c = o.length - r.offset;
14
- o.set(n.slice(0, c), r.offset);
15
- let l = n.length - c;
16
- if (l > i) throw Error(`Estimated tokens (${i}) is too small for the next batch of tokens (${l}).`);
17
- let u = new Uint16Array(i);
18
- if (u.set(n.slice(c), 0), t.push(u), a && s && !Array.isArray(e)) {
19
- s.set(e.mask.slice(0, c).map((e) => +!!e), r.offset);
20
- let t = new Uint8Array(u.length);
21
- t.set(e.mask.slice(c).map((e) => +!!e), 0), a.push(t);
22
- }
23
- r.offset = n.length - c;
24
- } else o.set(n, r.offset), s && !Array.isArray(e) && s.set(e.mask.map((e) => +!!e), r.offset), r.offset += n.length;
25
- }
26
- }
27
- }
28
- async function a(r, a, o) {
29
- await n("training-tokens");
30
- let s = await t("training-tokens", a.id, a.datasetID ?? "", o);
31
- await n("validation-tokens");
32
- let c = o?.validationSplit && o.validationSplit > 0 ? await t("validation-tokens", a.id, a.datasetID ?? "", o) : void 0, l = [new Uint16Array(s.shardSize)], u = o?.masking ? [new Uint8Array(s.shardSize)] : null, d = {
33
- offset: 0,
34
- total: 0
35
- }, f = o?.validationSplit && o.validationSplit > 0 ? [new Uint16Array(c.shardSize)] : void 0, p = o?.masking && f ? [new Uint8Array(c.shardSize)] : null, m = {
36
- offset: 0,
37
- total: 0
38
- }, h = performance.now();
39
- for (;;) {
40
- let t = o?.validationSplit && o.validationSplit > 0 && Math.random() < o.validationSplit, n = t ? f : l, g = t ? m : d, _ = t ? p : u, v = t ? c : s;
41
- if (await i(r, n, a, g, v.shardSize, _ || void 0), n.length > 1 && (v.appendShard(n[0], _ ? _[0] : void 0), n.shift(), _ && _.shift()), r.every((e) => !e.hasMoreConversations())) break;
42
- h = await e(h, o?.cb, d.total);
43
- }
44
- return l.length === 1 && (l[0] = l[0].subarray(0, d.offset), await s.appendShard(l[0], u ? u[0].subarray(0, d.offset) : void 0)), f && f.length === 1 && (f[0] = f[0].subarray(0, m.offset), await c.appendShard(f[0], p ? p[0].subarray(0, m.offset) : void 0)), {
45
- trainingTokens: s,
46
- validationTokens: f ? c : void 0
47
- };
48
- }
49
- //#endregion
50
- export { r as Task, a as tokensFromTasks };
@@ -1,5 +0,0 @@
1
- import { Task } from './Task';
2
- export default function splitValidation(tasks: Task[], validationSplit: number): Promise<{
3
- training: Task;
4
- validation: Task;
5
- }>;
@@ -1,18 +0,0 @@
1
- import { n as e } from "../../stream-BjdpSNqB.js";
2
- import t from "./ConversationTask.js";
3
- //#region lib/training/tasks/splitter.ts
4
- async function n(n, r) {
5
- if (r <= 0 || r >= 1) throw Error("validationSplit must be between 0 and 1");
6
- let i = [], a = [];
7
- for (let e of n) for (; e.hasMoreConversations();) {
8
- let t = await e.nextConversation();
9
- if (!t) break;
10
- Math.random() < r ? a.push(t) : i.push(t);
11
- }
12
- return {
13
- training: new t([new e(i)]),
14
- validation: new t([new e(a)])
15
- };
16
- }
17
- //#endregion
18
- export { n as default };