@genai-fi/nanogpt 0.23.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +78 -281
- package/dist/{DatasetBuilder-C0iJT29K.js → DatasetBuilder-DU1G1OKX.js} +20 -23
- package/dist/{RealDiv-CNsvC4AU.js → RealDiv-CSnvtN2E.js} +20 -20
- package/dist/{Reshape-dnm9bO3B.js → Reshape-BlylqwWy.js} +12 -12
- package/dist/TeachableLLM.d.ts +10 -15
- package/dist/TeachableLLM.js +201 -2
- package/dist/api/responses.d.ts +81 -0
- package/dist/api/responses.js +169 -0
- package/dist/api/training.d.ts +70 -0
- package/dist/api/training.js +205 -0
- package/dist/data/docx.js +9 -3036
- package/dist/data/stream.d.ts +8 -8
- package/dist/data/stream.js +1 -1
- package/dist/data/textLoader.d.ts +1 -1
- package/dist/data/textLoader.js +2 -2
- package/dist/data.d.ts +3 -0
- package/dist/data.js +12 -0
- package/dist/{dist-BqAU9-yi.js → dist-CwK5S7Ls.js} +2168 -2168
- package/dist/{gpgpu_math-DBYEAAdI.js → gpgpu_math-20tPK8LM.js} +458 -458
- package/dist/{Generator.d.ts → inference/Generator.d.ts} +16 -44
- package/dist/inference/Generator.js +271 -0
- package/dist/inference/tokenisePrompt.d.ts +4 -0
- package/dist/inference/tokenisePrompt.js +13 -0
- package/dist/inference/types.d.ts +44 -8
- package/dist/inference/utilities.d.ts +9 -0
- package/dist/inference/utilities.js +20 -0
- package/dist/jszip.min-DKa1Rjyn.js +3033 -0
- package/dist/{kernel_funcs_utils-D-mATnGR.js → kernel_funcs_utils-ql8Y8qPn.js} +96 -93
- package/dist/layers/MLP.d.ts +1 -1
- package/dist/layers/PositionEmbedding.d.ts +2 -1
- package/dist/layers/PositionEmbedding.js +1 -1
- package/dist/layers/RMSNorm.d.ts +1 -1
- package/dist/layers/TiedEmbedding.js +1 -1
- package/dist/layers.d.ts +4 -0
- package/dist/layers.js +14 -0
- package/dist/loader/load.js +58 -2
- package/dist/loader/loadHF.d.ts +1 -1
- package/dist/loader/loadHF.js +17 -2
- package/dist/loader/loadTransformers.js +46 -2
- package/dist/loader/newZipLoad.js +25 -2
- package/dist/loader/oldZipLoad.d.ts +1 -1
- package/dist/loader/oldZipLoad.js +37 -2
- package/dist/loader/save.js +75 -2
- package/dist/loader/types.d.ts +3 -3
- package/dist/main.d.ts +34 -43
- package/dist/main.js +12327 -20
- package/dist/{matMulGelu-BAIgQaRx.js → matMulGelu-CBoqTZM7.js} +2 -2
- package/dist/models/NanoGPTV1.js +95 -2
- package/dist/models/NanoGPTV2.js +86 -2
- package/dist/models/factory.js +13 -2
- package/dist/models/model.js +76 -2
- package/dist/models.d.ts +4 -0
- package/dist/models.js +14 -0
- package/dist/ops/dot16.js +1 -1
- package/dist/ops/matMulGelu.js +1 -1
- package/dist/ops/webgl/adamAdjust.js +1 -1
- package/dist/ops/webgl/fusedSoftmax.js +2 -2
- package/dist/ops/webgl/gelu.js +2 -2
- package/dist/ops/webgl/log.js +5 -5
- package/dist/ops/webgl/matMulGelu.js +1 -1
- package/dist/ops/webgl/matMulMul.js +1 -1
- package/dist/{stream-BjdpSNqB.js → stream-BpAwcvHz.js} +565 -561
- package/dist/{tfjs_backend-CydPRQTc.js → tfjs_backend-h5weiy1O.js} +36 -36
- package/dist/tokenise.d.ts +4 -0
- package/dist/tokenise.js +15 -0
- package/dist/tokeniser/CharTokeniser.js +18 -20
- package/dist/tokeniser/bpe.js +18 -22
- package/dist/training/BasicTrainer.d.ts +5 -10
- package/dist/training/BasicTrainer.js +80 -88
- package/dist/training/DatasetBuilder.d.ts +4 -4
- package/dist/training/DatasetBuilder.js +1 -1
- package/dist/training/PreTrainer.js +1 -1
- package/dist/training/SFTTrainer.js +1 -1
- package/dist/training/configure.d.ts +3 -0
- package/dist/training/configure.js +32 -0
- package/dist/training/factory.d.ts +6 -0
- package/dist/training/factory.js +8 -0
- package/dist/training/prepareData.d.ts +22 -0
- package/dist/training/prepareData.js +49 -0
- package/dist/training/tasks/TokenStore.d.ts +2 -1
- package/dist/training/tasks/TokenStore.js +8 -5
- package/dist/training/tasks/tokenStream.d.ts +17 -0
- package/dist/training/tasks/tokenStream.js +46 -0
- package/dist/training/types.d.ts +14 -1
- package/dist/training/validateOptions.d.ts +2 -0
- package/dist/training/validateOptions.js +19 -0
- package/dist/training/validation.js +4 -2
- package/dist/utilities/arrayShape.d.ts +1 -0
- package/dist/utilities/arrayShape.js +8 -0
- package/dist/utilities/random.d.ts +1 -0
- package/dist/utilities/random.js +19 -0
- package/dist/utilities/waitForModel.d.ts +1 -1
- package/dist/v4-BK7K-jy_.js +30 -0
- package/package.json +8 -2
- package/dist/Generator.js +0 -2
- package/dist/Trainer-DBsyWJ4s.js +0 -228
- package/dist/Trainer.d.ts +0 -45
- package/dist/Trainer.js +0 -2
- package/dist/main-BSaDGH7I.js +0 -13274
- package/dist/training/tasks/ConversationTask.d.ts +0 -17
- package/dist/training/tasks/ConversationTask.js +0 -29
- package/dist/training/tasks/PretrainingTask.d.ts +0 -17
- package/dist/training/tasks/PretrainingTask.js +0 -42
- package/dist/training/tasks/StartSentenceTask.d.ts +0 -18
- package/dist/training/tasks/StartSentenceTask.js +0 -45
- package/dist/training/tasks/Task.d.ts +0 -29
- package/dist/training/tasks/Task.js +0 -50
- package/dist/training/tasks/splitter.d.ts +0 -5
- package/dist/training/tasks/splitter.js +0 -18
|
@@ -1,31 +1,16 @@
|
|
|
1
|
-
import { Conversation, ITokeniser } from '
|
|
1
|
+
import { Conversation, ITokeniser } from '../tokeniser/type';
|
|
2
2
|
import { default as EE } from 'eventemitter3';
|
|
3
|
-
import { default as Model, ModelForwardAttributes } from '
|
|
4
|
-
import {
|
|
3
|
+
import { default as Model, ModelForwardAttributes } from '../models/model';
|
|
4
|
+
import { IGenerateOptions, GeneratorConversation, IGeneratorOutput } from './types';
|
|
5
5
|
export declare function isConversation(data: unknown): data is Conversation[];
|
|
6
|
-
export interface IGenerateOptions extends GenerateOptions {
|
|
7
|
-
maxLength?: number;
|
|
8
|
-
noCache?: boolean;
|
|
9
|
-
allowSpecial?: boolean;
|
|
10
|
-
nonConversational?: boolean;
|
|
11
|
-
continuation?: boolean;
|
|
12
|
-
}
|
|
13
6
|
export interface IGenerator extends EE<'start' | 'stop' | 'tokens' | 'reset'> {
|
|
14
|
-
generate(prompt: Conversation[], options?: IGenerateOptions): Promise<
|
|
15
|
-
generate(options?: IGenerateOptions): Promise<
|
|
16
|
-
step(prompt: Conversation[], options?: IGenerateOptions): Promise<
|
|
17
|
-
step(options?: IGenerateOptions): Promise<
|
|
7
|
+
generate(prompt: Conversation[], options?: IGenerateOptions): Promise<GeneratorConversation[]>;
|
|
8
|
+
generate(options?: IGenerateOptions): Promise<GeneratorConversation[]>;
|
|
9
|
+
step(prompt: Conversation[], options?: IGenerateOptions): Promise<GeneratorConversation[]>;
|
|
10
|
+
step(options?: IGenerateOptions): Promise<GeneratorConversation[]>;
|
|
18
11
|
stop(): void;
|
|
19
|
-
getConversation():
|
|
20
|
-
|
|
21
|
-
getProbabilitiesData(): number[][][];
|
|
22
|
-
getEmbeddingsData(): {
|
|
23
|
-
name: string;
|
|
24
|
-
tensor: number[][];
|
|
25
|
-
}[][];
|
|
26
|
-
getTokens(): number[];
|
|
27
|
-
getLastLoss(): number | null;
|
|
28
|
-
getLastMultinomialRand(): number | null;
|
|
12
|
+
getConversation(): GeneratorConversation[];
|
|
13
|
+
getRawOutput(): IGeneratorOutput[];
|
|
29
14
|
dispose(): void;
|
|
30
15
|
reset(): void;
|
|
31
16
|
}
|
|
@@ -42,18 +27,13 @@ export default class Generator extends EE<'start' | 'stop' | 'tokens' | 'reset'>
|
|
|
42
27
|
private outputConversation;
|
|
43
28
|
private actualTokeniser;
|
|
44
29
|
private lastToken;
|
|
45
|
-
private attentionData;
|
|
46
|
-
private probabilitiesData;
|
|
47
|
-
private embeddingsData;
|
|
48
|
-
private tokens;
|
|
49
30
|
private lastLoss;
|
|
50
|
-
private
|
|
31
|
+
private rawOutput;
|
|
51
32
|
private jobQueue;
|
|
52
33
|
private processingJob;
|
|
53
34
|
private startTime;
|
|
35
|
+
private tokenCount;
|
|
54
36
|
constructor(model: Model<ModelForwardAttributes>, tokeniser: ITokeniser);
|
|
55
|
-
private tokenisePrompt;
|
|
56
|
-
private processResponse;
|
|
57
37
|
/** Generate logits and select a token. */
|
|
58
38
|
private _generateToken;
|
|
59
39
|
/** Generate multiple tokens in a loop and produce text */
|
|
@@ -62,21 +42,13 @@ export default class Generator extends EE<'start' | 'stop' | 'tokens' | 'reset'>
|
|
|
62
42
|
reset(): void;
|
|
63
43
|
dispose(): void;
|
|
64
44
|
private initialise;
|
|
65
|
-
step(prompt: Conversation[], options?: IGenerateOptions): Promise<
|
|
66
|
-
step(options?: IGenerateOptions): Promise<
|
|
67
|
-
generate(prompt: Conversation[], options?: IGenerateOptions): Promise<
|
|
68
|
-
generate(options?: IGenerateOptions): Promise<
|
|
45
|
+
step(prompt: Conversation[], options?: IGenerateOptions): Promise<GeneratorConversation[]>;
|
|
46
|
+
step(options?: IGenerateOptions): Promise<GeneratorConversation[]>;
|
|
47
|
+
generate(prompt: Conversation[], options?: IGenerateOptions): Promise<GeneratorConversation[]>;
|
|
48
|
+
generate(options?: IGenerateOptions): Promise<GeneratorConversation[]>;
|
|
69
49
|
private startJob;
|
|
70
50
|
getQueueLength(): number;
|
|
71
51
|
stop(): void;
|
|
72
52
|
getConversation(): Conversation[];
|
|
73
|
-
|
|
74
|
-
getProbabilitiesData(): number[][][];
|
|
75
|
-
getEmbeddingsData(): {
|
|
76
|
-
name: string;
|
|
77
|
-
tensor: number[][];
|
|
78
|
-
}[][];
|
|
79
|
-
getTokens(): number[];
|
|
80
|
-
getLastLoss(): number | null;
|
|
81
|
-
getLastMultinomialRand(): number | null;
|
|
53
|
+
getRawOutput(): IGeneratorOutput[];
|
|
82
54
|
}
|
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
import { t as e } from "../eventemitter3-D_qV3Lof.js";
|
|
2
|
+
import { SPECIALS as t } from "../tokeniser/BaseTokeniser.js";
|
|
3
|
+
import n from "../tokeniser/CharTokeniser.js";
|
|
4
|
+
import { A as r, Ct as i, I as a, Y as o, _r as s, an as c, di as l, oi as u, yt as d } from "../dist-Da20xy8E.js";
|
|
5
|
+
import f from "../utilities/multinomialCPU.js";
|
|
6
|
+
import p from "../utilities/topP.js";
|
|
7
|
+
import { sparseSoftmaxCrossEntropy as m } from "../training/sparseCrossEntropy.js";
|
|
8
|
+
import h from "./tokenisePrompt.js";
|
|
9
|
+
import { getTokenConfidence as g } from "./utilities.js";
|
|
10
|
+
//#region lib/inference/Generator.ts
|
|
11
|
+
function _(e) {
|
|
12
|
+
return Array.isArray(e);
|
|
13
|
+
}
|
|
14
|
+
var v = [
|
|
15
|
+
...t,
|
|
16
|
+
...Array.from({ length: 95 }, (e, t) => String.fromCharCode(t + 32)),
|
|
17
|
+
..."áéíóúüñ¿¡",
|
|
18
|
+
..."äöÄÖÅå",
|
|
19
|
+
..."αβγδεζηθικλμνξοπρστυφχψωΑΒΓΔΕΖΗΘΙΚΛΜΝΞΟΠΡΣΤΥΦΧΨΩ",
|
|
20
|
+
..."абвгдеёжзийклмнопрстуфхцчшщъыьэюяАБВГДЕЁЖЗИЙКЛМНОПРСТУФХЦЧШЩЪЫЬЭЮЯ"
|
|
21
|
+
];
|
|
22
|
+
function y(e, t) {
|
|
23
|
+
return e.length === t ? e : e.length > t ? e.slice(0, t) : e.concat(Array(t - e.length).fill(""));
|
|
24
|
+
}
|
|
25
|
+
var b = class extends e {
|
|
26
|
+
model;
|
|
27
|
+
tokeniser;
|
|
28
|
+
active = !1;
|
|
29
|
+
cache = null;
|
|
30
|
+
initialPrompt = null;
|
|
31
|
+
outputConversation = [];
|
|
32
|
+
actualTokeniser;
|
|
33
|
+
lastToken = -1;
|
|
34
|
+
lastLoss = null;
|
|
35
|
+
rawOutput = [];
|
|
36
|
+
jobQueue = [];
|
|
37
|
+
processingJob = !1;
|
|
38
|
+
startTime = null;
|
|
39
|
+
tokenCount = 0;
|
|
40
|
+
constructor(e, t) {
|
|
41
|
+
super(), this.model = e, this.tokeniser = t, this.actualTokeniser = t;
|
|
42
|
+
}
|
|
43
|
+
async _generateToken(e, t, n) {
|
|
44
|
+
let s = n?.temperature ?? 1, h = n?.topK, _ = n?.topP, v = n?.usePadding ?? !1, y = {
|
|
45
|
+
training: !1,
|
|
46
|
+
attentionScores: n?.outputAttention ? { attentionOut: [] } : void 0,
|
|
47
|
+
cache: t,
|
|
48
|
+
outputEmbeddings: !!n?.outputHiddenStates
|
|
49
|
+
}, [b, x] = l(() => {
|
|
50
|
+
let t = e, r = t.shape[1], i = r <= this.model.config.blockSize ? t : t.slice([0, r - this.model.config.blockSize], [t.shape[0], this.model.config.blockSize]), o = v ? this.model.config.blockSize - i.shape[1] : 0, c = o > 0 ? d(i, [[0, 0], [0, o]]) : i, l = this.model.forward(y, c), f = l.shape[1] - 1 - o, p = l.slice([
|
|
51
|
+
0,
|
|
52
|
+
f,
|
|
53
|
+
0
|
|
54
|
+
], [
|
|
55
|
+
l.shape[0],
|
|
56
|
+
1,
|
|
57
|
+
l.shape[2]
|
|
58
|
+
]), h;
|
|
59
|
+
if (n?.targets) {
|
|
60
|
+
let e = n.targets.shift();
|
|
61
|
+
if (e !== void 0) {
|
|
62
|
+
let t = a([[e]], [1, 1], "int32"), n = m(p, t);
|
|
63
|
+
h = n.mean(), t.dispose(), n.dispose();
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
return y.attentionScores?.attentionOut && y.attentionScores.attentionOut.forEach((e, t) => {
|
|
67
|
+
e.shape[1] !== 1 && (y.attentionScores.attentionOut[t] = u(e.slice([
|
|
68
|
+
0,
|
|
69
|
+
f,
|
|
70
|
+
0
|
|
71
|
+
], [
|
|
72
|
+
e.shape[0],
|
|
73
|
+
1,
|
|
74
|
+
e.shape[2]
|
|
75
|
+
])), e.dispose());
|
|
76
|
+
}), l.dispose(), [p.div(s).squeeze([1]), h];
|
|
77
|
+
}), S, C, w, T = Math.random();
|
|
78
|
+
if (_) {
|
|
79
|
+
let e = o(b), t = await e.array();
|
|
80
|
+
e.dispose();
|
|
81
|
+
let r = p(t, _);
|
|
82
|
+
(n?.outputScores || n?.outputConfidence) && (C = t), S = f(r, T);
|
|
83
|
+
} else if (h) {
|
|
84
|
+
let { values: e, indices: t } = r(b, h), n = i(e, 1);
|
|
85
|
+
S = c(t, n, 1), e.dispose(), t.dispose(), n.dispose();
|
|
86
|
+
} else if (S = i(b, 1), n?.outputScores || n?.outputConfidence) {
|
|
87
|
+
let e = o(b);
|
|
88
|
+
C = await e.array(), e.dispose();
|
|
89
|
+
}
|
|
90
|
+
if (y.embeddings) {
|
|
91
|
+
let e = (n?.outputHiddenStates === "all" ? y.embeddings : y.embeddings.filter((e) => e.name.startsWith("block_output_"))).map(async (e) => {
|
|
92
|
+
let t = e.tensor.shape[1], r = e.tensor.slice([
|
|
93
|
+
0,
|
|
94
|
+
t - 1,
|
|
95
|
+
0
|
|
96
|
+
], [
|
|
97
|
+
e.tensor.shape[0],
|
|
98
|
+
1,
|
|
99
|
+
e.tensor.shape[2]
|
|
100
|
+
]);
|
|
101
|
+
e.tensor.dispose();
|
|
102
|
+
let i = r.squeeze([1]);
|
|
103
|
+
if (r.dispose(), n?.outputHiddenStates === "softmax") {
|
|
104
|
+
let t = this.model.project(i);
|
|
105
|
+
i.dispose();
|
|
106
|
+
let n = o(t, -1);
|
|
107
|
+
t.dispose();
|
|
108
|
+
let r = {
|
|
109
|
+
name: e.name,
|
|
110
|
+
tensor: await n.array()
|
|
111
|
+
};
|
|
112
|
+
return n.dispose(), r;
|
|
113
|
+
} else if (n?.outputHiddenStates === "logits") {
|
|
114
|
+
let t = this.model.project(i);
|
|
115
|
+
i.dispose();
|
|
116
|
+
let n = {
|
|
117
|
+
name: e.name,
|
|
118
|
+
tensor: await t.array()
|
|
119
|
+
};
|
|
120
|
+
return t.dispose(), n;
|
|
121
|
+
} else {
|
|
122
|
+
let t = await i.array();
|
|
123
|
+
return i.dispose(), {
|
|
124
|
+
name: e.name,
|
|
125
|
+
tensor: t
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
});
|
|
129
|
+
w = await Promise.all(e);
|
|
130
|
+
}
|
|
131
|
+
let E = S.reshape([1, 1]);
|
|
132
|
+
S.dispose(), S = E;
|
|
133
|
+
let D = (await S.array())[0][0], O = this.actualTokeniser.decode([D]);
|
|
134
|
+
this.lastToken = D;
|
|
135
|
+
let k = !n?.allowSpecial && this.tokeniser.isSpecialToken(D), A = {
|
|
136
|
+
outputTensor: S,
|
|
137
|
+
token: D,
|
|
138
|
+
text: O,
|
|
139
|
+
confidence: n?.outputConfidence && C ? g(C[0]) : null,
|
|
140
|
+
score: n?.outputScore && C ? C[0][D] : null,
|
|
141
|
+
logits: n?.outputLogits ? (await b.array())[0] : null,
|
|
142
|
+
scores: n?.outputScores && C ? C[0] : null,
|
|
143
|
+
hiddenStates: w ? w.map((e) => e.tensor[0]) : null,
|
|
144
|
+
attention: n?.outputAttention ? await Promise.all(y.attentionScores?.attentionOut?.map((e) => e.array()) ?? []) : null,
|
|
145
|
+
loss: this.lastLoss,
|
|
146
|
+
multinomialRand: T,
|
|
147
|
+
terminated: k
|
|
148
|
+
};
|
|
149
|
+
if (b.dispose(), x) {
|
|
150
|
+
let e = await x.array();
|
|
151
|
+
x.dispose(), A.loss = e;
|
|
152
|
+
}
|
|
153
|
+
return this.rawOutput.push(A), (!n?.chunkSize || this.tokenCount++ % n.chunkSize === 0) && (this.emit("tokens", A), n?._onChunk && await n._onChunk(A)), A;
|
|
154
|
+
}
|
|
155
|
+
async _generate(e, t) {
|
|
156
|
+
let n = !1;
|
|
157
|
+
this.outputConversation.length === 0 || this.outputConversation[this.outputConversation.length - 1]._completed || this.outputConversation[this.outputConversation.length - 1].role !== "assistant" && e?.nonConversational !== !0 || this.outputConversation[this.outputConversation.length - 1].role !== "text" && e?.nonConversational === !0 ? (this.outputConversation.push({
|
|
158
|
+
role: e?.nonConversational === !0 ? "text" : "assistant",
|
|
159
|
+
content: "",
|
|
160
|
+
_timestamp: Date.now()
|
|
161
|
+
}), n = !0, this.resetCache(!e?.noCache)) : (this.lastToken < 0 || t) && this.resetCache(!e?.noCache);
|
|
162
|
+
let r = this.lastToken >= 0 && this.cache ? a([this.lastToken], [1, 1], "int32") : await h(this.actualTokeniser, this.model.config.blockSize, t ? n ? this.outputConversation.slice(0, -1) : this.outputConversation : void 0, e), i = e?.maxLength ?? 1e3;
|
|
163
|
+
for (let t = 0; t < i && this.active; t++) {
|
|
164
|
+
let n = await this._generateToken(r, this.cache ? this.cache : void 0, {
|
|
165
|
+
...e,
|
|
166
|
+
usePadding: !this.cache
|
|
167
|
+
});
|
|
168
|
+
if (this.cache) r.dispose(), r = n.outputTensor;
|
|
169
|
+
else {
|
|
170
|
+
let e = r;
|
|
171
|
+
r = s([r, n.outputTensor], 1), e.dispose();
|
|
172
|
+
}
|
|
173
|
+
let a = this.outputConversation[this.outputConversation.length - 1];
|
|
174
|
+
if (this.cache || n.outputTensor.dispose(), n.terminated) {
|
|
175
|
+
a._completed = !0;
|
|
176
|
+
break;
|
|
177
|
+
}
|
|
178
|
+
t === i - 1 && i > 1 && (a._completed = !0, n.terminated = !0), a.content += n.text, a._output ||= [], a._output.push(n);
|
|
179
|
+
}
|
|
180
|
+
return r.dispose(), this.outputConversation;
|
|
181
|
+
}
|
|
182
|
+
resetCache(e) {
|
|
183
|
+
this.cache && (this.cache.forEach((e) => {
|
|
184
|
+
e && (e.k && e.k.dispose(), e.v && e.v.dispose(), e.k = void 0, e.v = void 0, e.cumulativeLength = 0, e.length = 0);
|
|
185
|
+
}), e || (this.cache = null)), this.lastToken = -1;
|
|
186
|
+
}
|
|
187
|
+
reset() {
|
|
188
|
+
this.resetCache(), this.outputConversation = [], this.initialPrompt = null, this.rawOutput = [], this.lastLoss = null, this.emit("reset");
|
|
189
|
+
}
|
|
190
|
+
dispose() {
|
|
191
|
+
this.reset();
|
|
192
|
+
}
|
|
193
|
+
initialise(e, t) {
|
|
194
|
+
if (this.cache && t?.noCache && this.reset(), this.initialPrompt = e || null, this.lastToken === -1 ? this.outputConversation = (this.initialPrompt || []).slice() : e && e.length > this.outputConversation.length && (this.outputConversation = (this.initialPrompt || []).slice(), this.resetCache()), !this.cache && !t?.noCache && (this.model.config.modelType !== "GenAI_NanoGPT_v1" || this.model.config.useRope)) {
|
|
195
|
+
let e = Array(this.model.config.nLayer);
|
|
196
|
+
for (let t = 0; t < this.model.config.nLayer; t++) e[t] = {
|
|
197
|
+
k: void 0,
|
|
198
|
+
v: void 0,
|
|
199
|
+
length: 0,
|
|
200
|
+
cumulativeLength: 0
|
|
201
|
+
};
|
|
202
|
+
this.cache = e, this.lastToken = -1;
|
|
203
|
+
}
|
|
204
|
+
let r = this.tokeniser.trained ? this.tokeniser : new n(y(v, this.tokeniser.vocabSize));
|
|
205
|
+
this.actualTokeniser = r, t?.loraName ? this.model.attachLoRA(t.loraName) : this.model.hasLoRA() && this.model.detachLoRA();
|
|
206
|
+
}
|
|
207
|
+
async step(e, t) {
|
|
208
|
+
let n = {
|
|
209
|
+
...t,
|
|
210
|
+
maxLength: 1
|
|
211
|
+
};
|
|
212
|
+
return _(e) ? this.generate(e, n) : this.generate({
|
|
213
|
+
...e,
|
|
214
|
+
...n
|
|
215
|
+
});
|
|
216
|
+
}
|
|
217
|
+
async generate(e, t) {
|
|
218
|
+
let n;
|
|
219
|
+
if (Array.isArray(e) ? n = e : typeof e == "object" && (t = e), this.processingJob) {
|
|
220
|
+
if (this.jobQueue.length > 10) throw Error("Job queue is too long, rejecting new job");
|
|
221
|
+
return new Promise((e, r) => {
|
|
222
|
+
this.jobQueue.push({
|
|
223
|
+
prompt: n,
|
|
224
|
+
options: t,
|
|
225
|
+
resolve: e,
|
|
226
|
+
reject: r
|
|
227
|
+
});
|
|
228
|
+
});
|
|
229
|
+
}
|
|
230
|
+
this.processingJob = !0, this.startTime = Date.now();
|
|
231
|
+
try {
|
|
232
|
+
let e = await this.startJob(n, t);
|
|
233
|
+
if (this.processingJob = !1, this.jobQueue.length > 0) {
|
|
234
|
+
let e = this.jobQueue.shift();
|
|
235
|
+
this.generate(e.prompt || [], e.options).then(e.resolve).catch(e.reject);
|
|
236
|
+
}
|
|
237
|
+
return e;
|
|
238
|
+
} catch (e) {
|
|
239
|
+
throw this.processingJob = !1, e;
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
async startJob(e, t) {
|
|
243
|
+
this.initialise(e, t), this.active = !0, this.model.metaData.generationSettings = t, t?.maxLength !== 1 && this.emit("start");
|
|
244
|
+
let n = await this._generate(t, !!e);
|
|
245
|
+
if (this.active = !1, this.startTime !== null) {
|
|
246
|
+
let e = Date.now(), n = e - this.startTime;
|
|
247
|
+
this.startTime = null, this.model.metaData.actionLog = this.model.metaData.actionLog || [], this.model.metaData.actionLog.push({
|
|
248
|
+
action: "generate",
|
|
249
|
+
timestamp: e,
|
|
250
|
+
duration: n,
|
|
251
|
+
tokensProcessed: this.rawOutput.length,
|
|
252
|
+
options: t || {}
|
|
253
|
+
});
|
|
254
|
+
}
|
|
255
|
+
return this.emit("stop"), n;
|
|
256
|
+
}
|
|
257
|
+
getQueueLength() {
|
|
258
|
+
return this.jobQueue.length;
|
|
259
|
+
}
|
|
260
|
+
stop() {
|
|
261
|
+
this.active = !1;
|
|
262
|
+
}
|
|
263
|
+
getConversation() {
|
|
264
|
+
return this.outputConversation;
|
|
265
|
+
}
|
|
266
|
+
getRawOutput() {
|
|
267
|
+
return this.rawOutput;
|
|
268
|
+
}
|
|
269
|
+
};
|
|
270
|
+
//#endregion
|
|
271
|
+
export { b as default, _ as isConversation };
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
import { Conversation, ITokeniser } from '../tokeniser/type';
|
|
2
|
+
import { IGenerateOptions } from './types';
|
|
3
|
+
import { Tensor } from '@tensorflow/tfjs-core';
|
|
4
|
+
export default function tokenisePrompt(tokeniser: ITokeniser, blockSize: number, prompt?: Conversation[], options?: IGenerateOptions): Promise<Tensor>;
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import { I as e } from "../dist-Da20xy8E.js";
|
|
2
|
+
//#region lib/inference/tokenisePrompt.ts
|
|
3
|
+
async function t(t, n, r, i) {
|
|
4
|
+
if (r) {
|
|
5
|
+
let a = r.length > 0 && r[r.length - 1].role === "text", o;
|
|
6
|
+
return o = i?.nonConversational ? a && i?.continuation ? [t.bosToken, ...t.encode(r[r.length - 1].content)] : t.encodeAsSequence(r, !0) : t.encodeConversation(r, !0), o.length > n && (o = o.slice(-n)), e([o], [1, o.length], "int32");
|
|
7
|
+
} else {
|
|
8
|
+
let n = i?.nonConversational ? void 0 : t.getSpecialTokenIndex("<|assistant_start|>"), r = n === void 0 ? [t.bosToken] : [t.bosToken, n];
|
|
9
|
+
return e([r], [1, r.length], "int32");
|
|
10
|
+
}
|
|
11
|
+
}
|
|
12
|
+
//#endregion
|
|
13
|
+
export { t as default };
|
|
@@ -1,16 +1,52 @@
|
|
|
1
1
|
import { Conversation } from '../tokeniser/type';
|
|
2
|
+
import { Tensor } from '@tensorflow/tfjs-core';
|
|
3
|
+
export interface IGeneratorOutput {
|
|
4
|
+
outputTensor: Tensor;
|
|
5
|
+
token: number;
|
|
6
|
+
text: string;
|
|
7
|
+
confidence: number | null;
|
|
8
|
+
score: number | null;
|
|
9
|
+
logits: number[] | null;
|
|
10
|
+
scores: number[] | null;
|
|
11
|
+
hiddenStates: number[][] | null;
|
|
12
|
+
attention: number[][][][] | null;
|
|
13
|
+
loss: number | null;
|
|
14
|
+
multinomialRand: number | null;
|
|
15
|
+
terminated: boolean;
|
|
16
|
+
}
|
|
2
17
|
export interface GeneratorConversation extends Conversation {
|
|
3
18
|
_completed?: boolean;
|
|
4
19
|
_timestamp?: number;
|
|
20
|
+
_output?: IGeneratorOutput[];
|
|
5
21
|
}
|
|
6
|
-
export interface
|
|
22
|
+
export interface IGenerateOptions {
|
|
23
|
+
input?: Conversation[] | string; /** Optional prompt. */
|
|
24
|
+
previous_response_id?: string;
|
|
7
25
|
temperature?: number;
|
|
8
|
-
topK?: number;
|
|
9
|
-
topP?: number;
|
|
26
|
+
topK?: number; /** Select the top K best tokens. */
|
|
27
|
+
topP?: number; /** Select the top P best tokens (nucleus sampling). */
|
|
10
28
|
usePadding?: boolean;
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
29
|
+
outputAttention?: boolean; /** All attention weights. */
|
|
30
|
+
outputScores?: boolean; /** Softmax scores of the final logits (all tokens). */
|
|
31
|
+
outputConfidence?: boolean; /** Entropy-based confidence in [0, 1] from the softmax distribution. */
|
|
32
|
+
outputScore?: boolean; /** Selected token score (probability) */
|
|
33
|
+
outputLogits?: boolean; /** Final output logits. */
|
|
34
|
+
outputHiddenStates?: 'embedding' | 'logits' | 'softmax' | 'all';
|
|
35
|
+
outputLoss?: boolean;
|
|
36
|
+
outputMultinomialRand?: boolean;
|
|
37
|
+
targets?: number[]; /** Optional target tokens for loss calculation. */
|
|
38
|
+
loraName?: string; /** Optional LoRA name to use for inference. */
|
|
39
|
+
maxLength?: number; /** Maximum length of the generated text. */
|
|
40
|
+
noCache?: boolean; /** Do not use key/value cache for attention. */
|
|
41
|
+
allowSpecial?: boolean; /** Keep special tokens in the output. */
|
|
42
|
+
nonConversational?: boolean; /** Do not use turn taking tokens. */
|
|
43
|
+
continuation?: boolean; /** Continue the previous response without adding a new turn. */
|
|
44
|
+
chunkSize?: number; /** Number of tokens to generate before returning a partial response. */
|
|
45
|
+
background?: boolean; /** Run the generation in the background and return immediately. */
|
|
46
|
+
_onChunk?: (output: IGeneratorOutput) => void | Promise<void>;
|
|
47
|
+
}
|
|
48
|
+
export interface IGeneratorResponse {
|
|
49
|
+
output: GeneratorConversation[] | null;
|
|
50
|
+
id: string;
|
|
51
|
+
done: boolean;
|
|
16
52
|
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
import { IGeneratorOutput } from './types';
|
|
2
|
+
/**
|
|
3
|
+
* Confidence derived from normalized entropy of the probability distribution.
|
|
4
|
+
* Returns a value in [0, 1], where 1 is fully confident (delta-like distribution)
|
|
5
|
+
* and 0 is maximally uncertain (uniform distribution).
|
|
6
|
+
*/
|
|
7
|
+
export declare function getTokenConfidence(probabilities: number[]): number;
|
|
8
|
+
export declare function getAttention(output: IGeneratorOutput, layer: number, head: number): number[] | null;
|
|
9
|
+
export declare function getHiddenState(output: IGeneratorOutput, layer: number): number[] | null;
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
//#region lib/inference/utilities.ts
|
|
2
|
+
function e(e) {
|
|
3
|
+
if (!e.length) return 0;
|
|
4
|
+
let t = 0;
|
|
5
|
+
for (let n of e) n > 0 && (t -= n * Math.log(n));
|
|
6
|
+
let n = Math.log(e.length);
|
|
7
|
+
if (n <= 0) return 1;
|
|
8
|
+
let r = t / n;
|
|
9
|
+
return Math.min(1, Math.max(0, 1 - r));
|
|
10
|
+
}
|
|
11
|
+
function t(e, t, n) {
|
|
12
|
+
if (!e.attention || !e.attention[t] || !e.attention[t][n]) return null;
|
|
13
|
+
let r = e.attention[t][n].length;
|
|
14
|
+
return e.attention[t][n][r - 1];
|
|
15
|
+
}
|
|
16
|
+
function n(e, t) {
|
|
17
|
+
return !e.hiddenStates || !e.hiddenStates[t] ? null : e.hiddenStates[t];
|
|
18
|
+
}
|
|
19
|
+
//#endregion
|
|
20
|
+
export { t as getAttention, n as getHiddenState, e as getTokenConfidence };
|