luminav 1.1.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
luminav/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .luminav import LuminaV, __version__, __author__, __license__
2
+
3
+ __all__ = ["LuminaV", "__version__", "__author__", "__license__"]
luminav/luminav.py ADDED
@@ -0,0 +1,651 @@
1
+ # Copyright (c) 2026 Lumina Moon and Contributors.
2
+ # Licensed under the Apache License, Version 2.0 (see LICENSE for details).
3
+
4
+ """
5
+ LuminaV: We Were Too Broke for AdamW So We Trapped Gradients in a
6
+ Hyperbolic Straitjacket and Hired a Traffic Cop to Slap Them
7
+
8
+ Official Repo : https://huggingface.co/cloverx-id/LuminaV-Optimizer-Paper
9
+ Software DOI : 10.57967/hf/10365
10
+ Paper PDF : https://huggingface.co/cloverx-id/XoneLM-1.0-Paper/blob/main/LuminaV.pdf
11
+ Paper DOI : 10.57967/hf/10270
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import logging
17
+ import math
18
+ from typing import Callable, Dict, List, Optional, Tuple
19
+
20
+ import torch
21
+ from torch import Tensor
22
+ from torch.optim import Optimizer
23
+
24
+ __all__ = ["LuminaV"]
25
+ __version__ = "1.1.5"
26
+ __author__ = "Silver Moon (cloverxion)"
27
+ __organization__ = "Lumina Moon"
28
+ __license__ = "Apache-2.0"
29
+
30
+ logger = logging.getLogger("LuminaV")
31
+
32
+
33
+ def _supports_native_bf16() -> bool:
34
+ return torch.cuda.is_available() and torch.cuda.is_bf16_supported()
35
+
36
+
37
+ HAS_TRITON = False
38
+ try:
39
+ import triton
40
+ import triton.language as tl
41
+ HAS_TRITON = True
42
+ except ImportError:
43
+ HAS_TRITON = False
44
+
45
+ if HAS_TRITON:
46
+ @triton.jit
47
+ def _prng(seed, idx):
48
+ x = idx * 1103515245 + seed + 12345
49
+ x = x ^ (x >> 16)
50
+ return x
51
+
52
+ @triton.jit
53
+ def _store_param(p_ptr, offsets, p_val, mask, seed, dtype_mode: tl.constexpr, use_sr: tl.constexpr):
54
+ if use_sr:
55
+ if dtype_mode == 1: # bfloat16
56
+ p_int = p_val.to(tl.int32, bitcast=True)
57
+ noise = (_prng(seed, offsets) & 0x7FFFFFFF) & 0xFFFF
58
+ p_int = (p_int + noise) & ~0xFFFF
59
+ p_out = p_int.to(tl.float32, bitcast=True).to(tl.bfloat16)
60
+ tl.store(p_ptr + offsets, p_out, mask=mask)
61
+ return
62
+ elif dtype_mode == 2: # float16
63
+ p_int = p_val.to(tl.int32, bitcast=True)
64
+ noise = (_prng(seed, offsets) & 0x7FFFFFFF) & 0x1FFF
65
+ p_int = (p_int + noise) & ~0x1FFF
66
+ p_out = p_int.to(tl.float32, bitcast=True).to(tl.float16)
67
+ tl.store(p_ptr + offsets, p_out, mask=mask)
68
+ return
69
+
70
+ if dtype_mode == 1:
71
+ tl.store(p_ptr + offsets, p_val.to(tl.bfloat16), mask=mask)
72
+ elif dtype_mode == 2:
73
+ tl.store(p_ptr + offsets, p_val.to(tl.float16), mask=mask)
74
+ else:
75
+ tl.store(p_ptr + offsets, p_val, mask=mask)
76
+
77
+ @triton.jit
78
+ def _triton_tanh_fast(x):
79
+ return 2.0 * tl.sigmoid(2.0 * x) - 1.0
80
+
81
+ @triton.jit
82
+ def _lumina_v2_pass1_kernel(
83
+ grad_ptr, exp_avg_ptr, exp_avg_sq_ptr, mask_sum_ptr,
84
+ n_elements, beta1, beta2, c1, c2, BLOCK_SIZE: tl.constexpr
85
+ ):
86
+ pid = tl.program_id(axis=0)
87
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
88
+ mask = offsets < n_elements
89
+
90
+ g = tl.load(grad_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
91
+ m = tl.load(exp_avg_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
92
+ v = tl.load(exp_avg_sq_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
93
+
94
+ m_new = beta1 * m + (1.0 - beta1) * g
95
+ nes_m = beta1 * m_new + (1.0 - beta1) * g
96
+ diff = g - m_new
97
+ v_new = beta2 * v + (1.0 - beta2) * (diff * diff)
98
+ v_new = tl.minimum(v_new, 65000.0)
99
+
100
+ sigma = tl.sqrt(tl.maximum(v_new, 0.0)) * c1 + c2
101
+ u = _triton_tanh_fast(nes_m / (sigma + 1e-6))
102
+ m_mask = tl.where((u * g) > 0.0, 1.0, 0.0)
103
+
104
+ tl.store(exp_avg_ptr + offsets, m_new, mask=mask)
105
+ tl.store(exp_avg_sq_ptr + offsets, v_new, mask=mask)
106
+
107
+ block_sum = tl.sum(tl.where(mask, m_mask, 0.0), axis=0)
108
+ tl.atomic_add(mask_sum_ptr, block_sum)
109
+
110
+ @triton.jit
111
+ def _lumina_v2_pass2_kernel(
112
+ p_ptr, grad_ptr, exp_avg_ptr, exp_avg_sq_ptr, mask_sum_ptr,
113
+ n_elements, beta1, c1, c2, lr, weight_decay, clamp_min,
114
+ seed, dtype_mode: tl.constexpr, use_sr: tl.constexpr, BLOCK_SIZE: tl.constexpr
115
+ ):
116
+ pid = tl.program_id(axis=0)
117
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
118
+ mask = offsets < n_elements
119
+
120
+ m_sum = tl.load(mask_sum_ptr)
121
+ m_bar = tl.minimum(tl.maximum(m_sum / n_elements, clamp_min), 1.0)
122
+
123
+ p = tl.load(p_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
124
+ g = tl.load(grad_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
125
+ m = tl.load(exp_avg_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
126
+ v = tl.load(exp_avg_sq_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
127
+
128
+ nes_m = beta1 * m + (1.0 - beta1) * g
129
+ sigma = tl.sqrt(tl.maximum(v, 0.0)) * c1 + c2
130
+ u = _triton_tanh_fast(nes_m / (sigma + 1e-6))
131
+ m_mask = tl.where((u * g) > 0.0, 1.0, 0.0)
132
+
133
+ if weight_decay != 0.0:
134
+ p = p * (1.0 - lr * weight_decay)
135
+
136
+ delta_theta = (u * m_mask) / m_bar
137
+ p_updated = p - lr * delta_theta
138
+ _store_param(p_ptr, offsets, p_updated, mask, seed, dtype_mode, use_sr)
139
+
140
+ @triton.jit
141
+ def _lumina_v2_single_pass_kernel(
142
+ p_ptr, grad_ptr, exp_avg_ptr, exp_avg_sq_ptr,
143
+ n_elements, beta1, beta2, c1, c2, lr, weight_decay,
144
+ seed, dtype_mode: tl.constexpr, use_sr: tl.constexpr, BLOCK_SIZE: tl.constexpr
145
+ ):
146
+ pid = tl.program_id(axis=0)
147
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
148
+ mask = offsets < n_elements
149
+
150
+ p = tl.load(p_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
151
+ g = tl.load(grad_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
152
+ m = tl.load(exp_avg_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
153
+ v = tl.load(exp_avg_sq_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
154
+
155
+ m_new = beta1 * m + (1.0 - beta1) * g
156
+ nes_m = beta1 * m_new + (1.0 - beta1) * g
157
+ diff = g - m_new
158
+ v_new = beta2 * v + (1.0 - beta2) * (diff * diff)
159
+ v_new = tl.minimum(v_new, 65000.0)
160
+
161
+ sigma = tl.sqrt(tl.maximum(v_new, 0.0)) * c1 + c2
162
+ u = _triton_tanh_fast(nes_m / (sigma + 1e-6))
163
+
164
+ tl.store(exp_avg_ptr + offsets, m_new, mask=mask)
165
+ tl.store(exp_avg_sq_ptr + offsets, v_new, mask=mask)
166
+
167
+ if weight_decay != 0.0:
168
+ p = p * (1.0 - lr * weight_decay)
169
+
170
+ p_updated = p - lr * u
171
+ _store_param(p_ptr, offsets, p_updated, mask, seed, dtype_mode, use_sr)
172
+
173
+ @triton.jit
174
+ def _lumina_v1_pass1_kernel(
175
+ grad_ptr, exp_avg_ptr, rms_sum_ptr,
176
+ n_elements, beta1, BLOCK_SIZE: tl.constexpr
177
+ ):
178
+ pid = tl.program_id(axis=0)
179
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
180
+ mask = offsets < n_elements
181
+
182
+ g = tl.load(grad_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
183
+ m = tl.load(exp_avg_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
184
+
185
+ m_new = beta1 * m + (1.0 - beta1) * g
186
+ nes_m = beta1 * m_new + (1.0 - beta1) * g
187
+ tl.store(exp_avg_ptr + offsets, m_new, mask=mask)
188
+
189
+ sq_val = nes_m * nes_m
190
+ block_sq_sum = tl.sum(tl.where(mask, sq_val, 0.0), axis=0)
191
+ tl.atomic_add(rms_sum_ptr, block_sq_sum)
192
+
193
+ @triton.jit
194
+ def _lumina_v1_pass2_kernel(
195
+ grad_ptr, exp_avg_ptr, rms_sum_ptr, mask_sum_ptr,
196
+ n_elements, beta1, tau, eps, c2, alpha_ss, BLOCK_SIZE: tl.constexpr
197
+ ):
198
+ pid = tl.program_id(axis=0)
199
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
200
+ mask = offsets < n_elements
201
+
202
+ sq_sum = tl.load(rms_sum_ptr)
203
+ rms = tl.sqrt(sq_sum / n_elements + eps)
204
+ sigma = tau * rms + c2
205
+
206
+ g = tl.load(grad_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
207
+ m = tl.load(exp_avg_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
208
+
209
+ nes_m = beta1 * m + (1.0 - beta1) * g
210
+ z = nes_m / (sigma + 1e-6)
211
+ z_soft = z / (1.0 + alpha_ss * tl.abs(z))
212
+ u = _triton_tanh_fast(z_soft)
213
+ m_mask = tl.where((u * g) > 0.0, 1.0, 0.0)
214
+
215
+ block_sum = tl.sum(tl.where(mask, m_mask, 0.0), axis=0)
216
+ tl.atomic_add(mask_sum_ptr, block_sum)
217
+
218
+ @triton.jit
219
+ def _lumina_v1_pass3_update_kernel(
220
+ p_ptr, grad_ptr, exp_avg_ptr, rms_sum_ptr, mask_sum_ptr,
221
+ n_elements, beta1, tau, eps, c2, alpha_ss, lr, weight_decay, clamp_min,
222
+ seed, dtype_mode: tl.constexpr, use_sr: tl.constexpr, cautious: tl.constexpr, BLOCK_SIZE: tl.constexpr
223
+ ):
224
+ pid = tl.program_id(axis=0)
225
+ offsets = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE)
226
+ mask = offsets < n_elements
227
+
228
+ sq_sum = tl.load(rms_sum_ptr)
229
+ rms = tl.sqrt(sq_sum / n_elements + eps)
230
+ sigma = tau * rms + c2
231
+
232
+ m_sum = tl.load(mask_sum_ptr)
233
+ m_bar = tl.minimum(tl.maximum(m_sum / n_elements, clamp_min), 1.0)
234
+
235
+ p = tl.load(p_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
236
+ g = tl.load(grad_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
237
+ m = tl.load(exp_avg_ptr + offsets, mask=mask, other=0.0).to(tl.float32)
238
+
239
+ nes_m = beta1 * m + (1.0 - beta1) * g
240
+ z = nes_m / (sigma + 1e-6)
241
+ z_soft = z / (1.0 + alpha_ss * tl.abs(z))
242
+ u = _triton_tanh_fast(z_soft)
243
+ m_mask = tl.where((u * g) > 0.0, 1.0, 0.0)
244
+
245
+ if weight_decay != 0.0:
246
+ p = p * (1.0 - lr * weight_decay)
247
+
248
+ delta_theta = (u * m_mask) / m_bar if cautious else u
249
+ p_updated = p - lr * delta_theta
250
+ _store_param(p_ptr, offsets, p_updated, mask, seed, dtype_mode, use_sr)
251
+
252
+
253
+ def _sr_update(p: Tensor, update: Tensor, lr: float, use_sr: bool):
254
+ if use_sr and p.dtype == torch.bfloat16:
255
+ p_fp32 = p.float().add_(update, alpha=-lr)
256
+ int_view = p_fp32.view(torch.int32)
257
+ noise = torch.randint(0, 1 << 16, p.shape, dtype=torch.int32, device=p.device)
258
+ p.copy_(((int_view + noise) & ~0xFFFF).view(torch.float32).to(torch.bfloat16))
259
+ elif use_sr and p.dtype == torch.float16:
260
+ p_fp32 = p.float().add_(update, alpha=-lr)
261
+ int_view = p_fp32.view(torch.int32)
262
+ noise = torch.randint(0, 1 << 13, p.shape, dtype=torch.int32, device=p.device)
263
+ p.copy_(((int_view + noise) & ~0x1FFF).view(torch.float32).to(torch.float16))
264
+ else:
265
+ p.add_(update, alpha=-lr)
266
+
267
+
268
+ class LuminaV(Optimizer):
269
+ def __init__(
270
+ self,
271
+ params,
272
+ lr: float = 8e-4,
273
+ betas: Tuple[float, float] = (0.9, 0.999),
274
+ eps: float = 1e-8,
275
+ weight_decay: float = 8e-2,
276
+ tau: float = 0.8,
277
+ alpha_ss: float = 0.5,
278
+ cautious: bool = True,
279
+ cautious_clamp_min: float = 0.2,
280
+ buffer: int = 2,
281
+ stochastic_rounding: bool = True,
282
+ execution: str = "auto",
283
+ ):
284
+ if lr < 0.0:
285
+ raise ValueError(f"Invalid learning rate: {lr}")
286
+ if not 0.0 <= betas[0] < 1.0:
287
+ raise ValueError(f"Invalid beta1 parameter: {betas[0]}")
288
+ if not 0.0 <= betas[1] < 1.0:
289
+ raise ValueError(f"Invalid beta2 parameter: {betas[1]}")
290
+ if eps <= 0.0:
291
+ raise ValueError(f"Invalid epsilon value: {eps}")
292
+ if weight_decay < 0.0:
293
+ raise ValueError(f"Invalid weight_decay value: {weight_decay}")
294
+ if tau <= 0.0:
295
+ raise ValueError(f"Invalid tau parameter: {tau}")
296
+ if not 0.0 < cautious_clamp_min <= 1.0:
297
+ raise ValueError(f"Invalid cautious_clamp_min: {cautious_clamp_min}")
298
+ if buffer not in (1, 2):
299
+ raise ValueError(f"Buffer count must be 1 (Single) or 2 (Dual), got: {buffer}")
300
+
301
+ defaults = dict(
302
+ lr=lr, betas=betas, eps=eps, weight_decay=weight_decay,
303
+ tau=tau, alpha_ss=alpha_ss, cautious=cautious,
304
+ cautious_clamp_min=cautious_clamp_min, buffer=buffer,
305
+ stochastic_rounding=stochastic_rounding, execution=execution
306
+ )
307
+ super().__init__(params, defaults)
308
+ self._scratch_tensors: Dict[torch.device, torch.Tensor] = {}
309
+
310
+ def _get_scratch_buffer(self, device: torch.device, count: int) -> torch.Tensor:
311
+ if device not in self._scratch_tensors or self._scratch_tensors[device].numel() < count:
312
+ self._scratch_tensors[device] = torch.zeros((count,), device=device, dtype=torch.float32)
313
+ buf = self._scratch_tensors[device][:count]
314
+ buf.zero_()
315
+ return buf
316
+
317
+ @torch.no_grad()
318
+ def step(self, closure: Optional[Callable[[], float]] = None) -> Optional[float]:
319
+ loss = None
320
+ if closure is not None:
321
+ with torch.enable_grad():
322
+ loss = closure()
323
+
324
+ for group in self.param_groups:
325
+ params_with_grad = []
326
+ grads = []
327
+ exp_avgs = []
328
+ exp_avg_sqs = []
329
+ steps = []
330
+ buf_count = group["buffer"]
331
+
332
+ for p in group["params"]:
333
+ if p.grad is None:
334
+ continue
335
+ if p.grad.is_sparse:
336
+ raise RuntimeError("LuminaV does not support sparse gradients.")
337
+
338
+ params_with_grad.append(p)
339
+ grads.append(p.grad)
340
+ state = self.state[p]
341
+
342
+ if len(state) == 0:
343
+ state["step"] = 0
344
+ state["exp_avg"] = torch.zeros_like(p, memory_format=torch.contiguous_format)
345
+ if buf_count == 2:
346
+ state["exp_avg_sq"] = torch.zeros_like(p, memory_format=torch.contiguous_format)
347
+
348
+ exp_avgs.append(state["exp_avg"])
349
+ if buf_count == 2:
350
+ if "exp_avg_sq" not in state:
351
+ state["exp_avg_sq"] = torch.zeros_like(p, memory_format=torch.contiguous_format)
352
+ exp_avg_sqs.append(state["exp_avg_sq"])
353
+
354
+ state["step"] += 1
355
+ steps.append(state["step"])
356
+
357
+ if not params_with_grad:
358
+ continue
359
+
360
+ lr = group["lr"]
361
+ beta1, beta2 = group["betas"]
362
+ eps = group["eps"]
363
+ weight_decay = group["weight_decay"]
364
+ tau = group["tau"]
365
+ alpha_ss = group["alpha_ss"]
366
+ cautious = group["cautious"]
367
+ clamp_min = group["cautious_clamp_min"]
368
+ use_sr = group["stochastic_rounding"]
369
+ exec_mode = group["execution"]
370
+
371
+ all_cuda = all(p.is_cuda for p in params_with_grad)
372
+
373
+ if exec_mode == "auto":
374
+ if HAS_TRITON and all_cuda:
375
+ exec_mode = "triton"
376
+ elif hasattr(torch, "_foreach_mul_"):
377
+ exec_mode = "foreach"
378
+ else:
379
+ exec_mode = "single"
380
+
381
+ if exec_mode == "triton" and not (HAS_TRITON and all_cuda):
382
+ exec_mode = "foreach"
383
+
384
+ if exec_mode == "triton" and HAS_TRITON and all_cuda:
385
+ try:
386
+ if buf_count == 2:
387
+ self._triton_step_v2(params_with_grad, grads, exp_avgs, exp_avg_sqs, steps, lr, beta1, beta2, eps, weight_decay, tau, cautious, clamp_min, use_sr)
388
+ else:
389
+ self._triton_step_v1(params_with_grad, grads, exp_avgs, steps, lr, beta1, eps, weight_decay, tau, alpha_ss, cautious, clamp_min, use_sr)
390
+ except Exception as e:
391
+ logger.warning(f"Triton kernel fallback to C++ foreach engine: {e}")
392
+ self._grouped_foreach_step(params_with_grad, grads, exp_avgs, exp_avg_sqs, steps, lr, beta1, beta2, eps, weight_decay, tau, alpha_ss, cautious, clamp_min, buf_count, use_sr)
393
+ elif exec_mode == "foreach":
394
+ self._grouped_foreach_step(params_with_grad, grads, exp_avgs, exp_avg_sqs, steps, lr, beta1, beta2, eps, weight_decay, tau, alpha_ss, cautious, clamp_min, buf_count, use_sr)
395
+ else:
396
+ if buf_count == 2:
397
+ self._single_step_v2(params_with_grad, grads, exp_avgs, exp_avg_sqs, steps, lr, beta1, beta2, eps, weight_decay, tau, cautious, clamp_min, use_sr)
398
+ else:
399
+ self._single_step_v1(params_with_grad, grads, exp_avgs, steps, lr, beta1, eps, weight_decay, tau, alpha_ss, cautious, clamp_min, use_sr)
400
+
401
+ return loss
402
+
403
+ def _grouped_foreach_step(self, params, grads, exp_avgs, exp_avg_sqs, steps, lr, beta1, beta2, eps, weight_decay, tau, alpha_ss, cautious, clamp_min, buf_count, use_sr):
404
+ groups: Dict[Tuple[torch.device, torch.dtype], List[int]] = {}
405
+ for idx, p in enumerate(params):
406
+ key = (p.device, p.dtype)
407
+ if key not in groups:
408
+ groups[key] = []
409
+ groups[key].append(idx)
410
+
411
+ for (dev, dt), indices in groups.items():
412
+ sub_params = [params[i] for i in indices]
413
+ sub_grads = [grads[i] for i in indices]
414
+ sub_exp_avgs = [exp_avgs[i] for i in indices]
415
+ sub_steps = [steps[i] for i in indices]
416
+
417
+ if buf_count == 2:
418
+ sub_exp_avg_sqs = [exp_avg_sqs[i] for i in indices]
419
+ self._foreach_step_v2(sub_params, sub_grads, sub_exp_avgs, sub_exp_avg_sqs, sub_steps, lr, beta1, beta2, eps, weight_decay, tau, cautious, clamp_min, use_sr)
420
+ else:
421
+ self._foreach_step_v1(sub_params, sub_grads, sub_exp_avgs, sub_steps, lr, beta1, eps, weight_decay, tau, alpha_ss, cautious, clamp_min, use_sr)
422
+
423
+ def _triton_step_v2(self, params, grads, exp_avgs, exp_avg_sqs, steps, lr, beta1, beta2, eps, weight_decay, tau, cautious, clamp_min, use_sr):
424
+ BLOCK_SIZE = 1024
425
+ device = params[0].device
426
+ scratch = self._get_scratch_buffer(device, len(params))
427
+
428
+ for i in range(len(params)):
429
+ p, grad, exp_avg, exp_avg_sq, step = params[i], grads[i], exp_avgs[i], exp_avg_sqs[i], steps[i]
430
+ is_orig_contig = p.is_contiguous()
431
+ p_contig = p if is_orig_contig else p.contiguous()
432
+ grad_contig = grad if grad.is_contiguous() else grad.contiguous()
433
+ exp_avg_contig = exp_avg if exp_avg.is_contiguous() else exp_avg.contiguous()
434
+ exp_avg_sq_contig = exp_avg_sq if exp_avg_sq.is_contiguous() else exp_avg_sq.contiguous()
435
+
436
+ effective_eps = max(eps, 1e-4) if p.dtype == torch.float16 else eps
437
+ bc1 = 1.0 - (beta1**step)
438
+ bc2 = 1.0 - (beta2**step)
439
+ c1 = (bc1 * tau) / math.sqrt(bc2)
440
+ c2 = effective_eps * bc1 * tau
441
+
442
+ if p.dtype == torch.bfloat16:
443
+ dtype_mode = 1
444
+ elif p.dtype == torch.float16:
445
+ dtype_mode = 2
446
+ else:
447
+ dtype_mode = 0
448
+
449
+ seed = torch.randint(0, 2147483647, (1,)).item()
450
+ n_elements = p.numel()
451
+ grid = (triton.cdiv(n_elements, BLOCK_SIZE),)
452
+
453
+ if not cautious:
454
+ _lumina_v2_single_pass_kernel[grid](
455
+ p_contig, grad_contig, exp_avg_contig, exp_avg_sq_contig,
456
+ n_elements, beta1, beta2, c1, c2, lr, weight_decay,
457
+ seed, dtype_mode, use_sr, BLOCK_SIZE=BLOCK_SIZE
458
+ )
459
+ else:
460
+ mask_sum_ptr = scratch[i : i + 1]
461
+ mask_sum_ptr.zero_()
462
+ _lumina_v2_pass1_kernel[grid](
463
+ grad_contig, exp_avg_contig, exp_avg_sq_contig, mask_sum_ptr,
464
+ n_elements, beta1, beta2, c1, c2, BLOCK_SIZE=BLOCK_SIZE
465
+ )
466
+ _lumina_v2_pass2_kernel[grid](
467
+ p_contig, grad_contig, exp_avg_contig, exp_avg_sq_contig, mask_sum_ptr,
468
+ n_elements, beta1, c1, c2, lr, weight_decay, clamp_min,
469
+ seed, dtype_mode, use_sr, BLOCK_SIZE=BLOCK_SIZE
470
+ )
471
+
472
+ if not is_orig_contig:
473
+ p.copy_(p_contig)
474
+
475
+ def _triton_step_v1(self, params, grads, exp_avgs, steps, lr, beta1, eps, weight_decay, tau, alpha_ss, cautious, clamp_min, use_sr):
476
+ BLOCK_SIZE = 1024
477
+ device = params[0].device
478
+ scratch_rms = self._get_scratch_buffer(device, len(params) * 2)
479
+
480
+ for i in range(len(params)):
481
+ p, grad, exp_avg, step = params[i], grads[i], exp_avgs[i], steps[i]
482
+ is_orig_contig = p.is_contiguous()
483
+ p_contig = p if is_orig_contig else p.contiguous()
484
+ grad_contig = grad if grad.is_contiguous() else grad.contiguous()
485
+ exp_avg_contig = exp_avg if exp_avg.is_contiguous() else exp_avg.contiguous()
486
+
487
+ effective_eps = max(eps, 1e-4) if p.dtype == torch.float16 else eps
488
+ bc1 = 1.0 - (beta1**step)
489
+ c2 = effective_eps * bc1 * tau
490
+
491
+ if p.dtype == torch.bfloat16:
492
+ dtype_mode = 1
493
+ elif p.dtype == torch.float16:
494
+ dtype_mode = 2
495
+ else:
496
+ dtype_mode = 0
497
+
498
+ seed = torch.randint(0, 2147483647, (1,)).item()
499
+ n_elements = p.numel()
500
+ grid = (triton.cdiv(n_elements, BLOCK_SIZE),)
501
+ rms_sum_ptr = scratch_rms[2 * i : 2 * i + 1]
502
+ mask_sum_ptr = scratch_rms[2 * i + 1 : 2 * i + 2]
503
+ rms_sum_ptr.zero_()
504
+ mask_sum_ptr.zero_()
505
+
506
+ _lumina_v1_pass1_kernel[grid](grad_contig, exp_avg_contig, rms_sum_ptr, n_elements, beta1, BLOCK_SIZE=BLOCK_SIZE)
507
+ _lumina_v1_pass2_kernel[grid](grad_contig, exp_avg_contig, rms_sum_ptr, mask_sum_ptr, n_elements, beta1, tau, eps, c2, alpha_ss, BLOCK_SIZE=BLOCK_SIZE)
508
+ _lumina_v1_pass3_update_kernel[grid](
509
+ p_contig, grad_contig, exp_avg_contig, rms_sum_ptr, mask_sum_ptr,
510
+ n_elements, beta1, tau, eps, c2, alpha_ss, lr, weight_decay, clamp_min,
511
+ seed, dtype_mode, use_sr, cautious, BLOCK_SIZE=BLOCK_SIZE
512
+ )
513
+
514
+ if not is_orig_contig:
515
+ p.copy_(p_contig)
516
+
517
+ def _single_step_v2(self, params, grads, exp_avgs, exp_avg_sqs, steps, lr, beta1, beta2, eps, weight_decay, tau, cautious, clamp_min, use_sr):
518
+ for i in range(len(params)):
519
+ p, grad, exp_avg, exp_avg_sq, step = params[i], grads[i], exp_avgs[i], exp_avg_sqs[i], steps[i]
520
+ effective_eps = max(eps, 1e-4) if p.dtype == torch.float16 else eps
521
+ bc1 = 1.0 - (beta1**step)
522
+ bc2 = 1.0 - (beta2**step)
523
+ c1 = (bc1 * tau) / math.sqrt(bc2)
524
+ c2 = effective_eps * bc1 * tau
525
+
526
+ if weight_decay != 0.0:
527
+ p.mul_(1.0 - lr * weight_decay)
528
+
529
+ exp_avg.mul_(beta1).add_(grad, alpha=1.0 - beta1)
530
+ nes_m = torch.mul(exp_avg, beta1).add_(grad, alpha=1.0 - beta1)
531
+ grad_diff = grad - exp_avg
532
+
533
+ if p.dtype == torch.float16:
534
+ diff_f32 = grad_diff.float()
535
+ v_f32 = exp_avg_sq.float().mul_(beta2).addcmul_(diff_f32, diff_f32, value=1.0 - beta2).clamp_(0.0, 65000.0)
536
+ exp_avg_sq.copy_(v_f32.to(dtype=p.dtype))
537
+ else:
538
+ exp_avg_sq.mul_(beta2).addcmul_(grad_diff, grad_diff, value=1.0 - beta2)
539
+
540
+ sigma = exp_avg_sq.float().sqrt().mul_(c1).add_(c2)
541
+ update = torch.tanh(nes_m.float() / (sigma + 1e-6)).to(dtype=p.dtype)
542
+
543
+ if cautious:
544
+ mask = (update * grad > 0).to(dtype=grad.dtype)
545
+ mask_scale = mask.float().mean().clamp_(min=clamp_min, max=1.0).to(dtype=grad.dtype)
546
+ update = update.mul_(mask).div_(mask_scale)
547
+
548
+ _sr_update(p, update, lr, use_sr)
549
+
550
+ def _single_step_v1(self, params, grads, exp_avgs, steps, lr, beta1, eps, weight_decay, tau, alpha_ss, cautious, clamp_min, use_sr):
551
+ for i in range(len(params)):
552
+ p, grad, exp_avg, step = params[i], grads[i], exp_avgs[i], steps[i]
553
+ if weight_decay != 0.0:
554
+ p.mul_(1.0 - lr * weight_decay)
555
+
556
+ exp_avg.mul_(beta1).add_(grad, alpha=1.0 - beta1)
557
+ nes_m = torch.mul(exp_avg, beta1).add_(grad, alpha=1.0 - beta1)
558
+ effective_eps = max(eps, 1e-4) if p.dtype == torch.float16 else eps
559
+ bc1 = 1.0 - (beta1**step)
560
+ c2 = effective_eps * bc1 * tau
561
+
562
+ rms = torch.sqrt(nes_m.float().square().mean() + effective_eps)
563
+ sigma = rms * tau + c2
564
+ z = nes_m.float() / (sigma + 1e-6)
565
+ z_soft = z / (1.0 + alpha_ss * torch.abs(z))
566
+ update = torch.tanh(z_soft).to(dtype=p.dtype)
567
+
568
+ if cautious:
569
+ mask = (update * grad > 0).to(dtype=grad.dtype)
570
+ mask_scale = mask.float().mean().clamp_(min=clamp_min, max=1.0).to(dtype=grad.dtype)
571
+ update = update.mul_(mask).div_(mask_scale)
572
+
573
+ _sr_update(p, update, lr, use_sr)
574
+
575
+ def _foreach_step_v2(self, params, grads, exp_avgs, exp_avg_sqs, steps, lr, beta1, beta2, eps, weight_decay, tau, cautious, clamp_min, use_sr):
576
+ if weight_decay != 0.0:
577
+ torch._foreach_mul_(params, 1.0 - lr * weight_decay)
578
+
579
+ torch._foreach_mul_(exp_avgs, beta1)
580
+ torch._foreach_add_(exp_avgs, grads, alpha=1.0 - beta1)
581
+
582
+ nes_m_list = torch._foreach_mul(exp_avgs, beta1)
583
+ torch._foreach_add_(nes_m_list, grads, alpha=1.0 - beta1)
584
+
585
+ grad_diff_list = torch._foreach_sub(grads, exp_avgs)
586
+ is_fp16 = params[0].dtype == torch.float16
587
+ if is_fp16:
588
+ for i in range(len(params)):
589
+ d_f32 = grad_diff_list[i].float()
590
+ v_f32 = exp_avg_sqs[i].float().mul_(beta2).addcmul_(d_f32, d_f32, value=1.0 - beta2).clamp_(0.0, 65000.0)
591
+ exp_avg_sqs[i].copy_(v_f32.to(dtype=torch.float16))
592
+ else:
593
+ torch._foreach_mul_(exp_avg_sqs, beta2)
594
+ torch._foreach_addcmul_(exp_avg_sqs, grad_diff_list, grad_diff_list, value=1.0 - beta2)
595
+
596
+ effective_eps = max(eps, 1e-4) if is_fp16 else eps
597
+ bias_correction1 = [1.0 - (beta1**st) for st in steps]
598
+ bias_correction2 = [1.0 - (beta2**st) for st in steps]
599
+ c1_list = [(bc1 * tau) / math.sqrt(bc2) for bc1, bc2 in zip(bias_correction1, bias_correction2)]
600
+ c2_list = [effective_eps * bc1 * tau for bc1 in bias_correction1]
601
+
602
+ updates = []
603
+ for i in range(len(params)):
604
+ sigma = exp_avg_sqs[i].float().sqrt().mul_(c1_list[i]).add_(c2_list[i])
605
+ u = torch.tanh(nes_m_list[i].float() / (sigma + 1e-6)).to(dtype=params[i].dtype)
606
+ if cautious:
607
+ mask = (u * grads[i] > 0).to(dtype=grads[i].dtype)
608
+ scale = mask.float().mean().clamp_(min=clamp_min, max=1.0).to(dtype=grads[i].dtype)
609
+ u = u.mul(mask).div(scale)
610
+ updates.append(u)
611
+
612
+ if use_sr and params[0].dtype in (torch.bfloat16, torch.float16):
613
+ for i in range(len(params)):
614
+ _sr_update(params[i], updates[i], lr, use_sr=True)
615
+ else:
616
+ torch._foreach_add_(params, updates, alpha=-lr)
617
+
618
+ def _foreach_step_v1(self, params, grads, exp_avgs, steps, lr, beta1, eps, weight_decay, tau, alpha_ss, cautious, clamp_min, use_sr):
619
+ if weight_decay != 0.0:
620
+ torch._foreach_mul_(params, 1.0 - lr * weight_decay)
621
+
622
+ torch._foreach_mul_(exp_avgs, beta1)
623
+ torch._foreach_add_(exp_avgs, grads, alpha=1.0 - beta1)
624
+
625
+ nes_m_list = torch._foreach_mul(exp_avgs, beta1)
626
+ torch._foreach_add_(nes_m_list, grads, alpha=1.0 - beta1)
627
+
628
+ is_fp16 = params[0].dtype == torch.float16
629
+ effective_eps = max(eps, 1e-4) if is_fp16 else eps
630
+ bias_correction1 = [1.0 - (beta1**st) for st in steps]
631
+ c2_list = [effective_eps * bc1 * tau for bc1 in bias_correction1]
632
+
633
+ updates = []
634
+ for i in range(len(params)):
635
+ m = nes_m_list[i]
636
+ rms = torch.sqrt(m.float().square().mean() + effective_eps)
637
+ sigma = rms * tau + c2_list[i]
638
+ z = m.float() / (sigma + 1e-6)
639
+ z_soft = z / (1.0 + alpha_ss * torch.abs(z))
640
+ u = torch.tanh(z_soft).to(dtype=params[i].dtype)
641
+ if cautious:
642
+ mask = (u * grads[i] > 0).to(dtype=grads[i].dtype)
643
+ scale = mask.float().mean().clamp_(min=clamp_min, max=1.0).to(dtype=grads[i].dtype)
644
+ u = u.mul(mask).div(scale)
645
+ updates.append(u)
646
+
647
+ if use_sr and params[0].dtype in (torch.bfloat16, torch.float16):
648
+ for i in range(len(params)):
649
+ _sr_update(params[i], updates[i], lr, use_sr=True)
650
+ else:
651
+ torch._foreach_add_(params, updates, alpha=-lr)
@@ -0,0 +1,202 @@
1
+ Metadata-Version: 2.4
2
+ Name: luminav
3
+ Version: 1.1.5
4
+ Summary: A master-free, memory-efficient adaptive optimizer for low-precision deep learning
5
+ Author: Silver Moon (cloverxion)
6
+ Maintainer: Lumina Moon
7
+ License: Apache-2.0
8
+ Project-URL: Homepage, https://huggingface.co/cloverx-id/LuminaV-Optimizer-Paper
9
+ Project-URL: Author Profile, https://huggingface.co/cloverxion
10
+ Project-URL: Organization, https://huggingface.co/Lumina-Moon
11
+ Project-URL: Software DOI, https://doi.org/10.57967/hf/10365
12
+ Project-URL: Paper DOI, https://doi.org/10.57967/hf/10270
13
+ Project-URL: Bug Tracker, https://huggingface.co/cloverx-id/LuminaV-Optimizer-Paper/discussions
14
+ Project-URL: Changelog, https://huggingface.co/cloverx-id/LuminaV-Optimizer-Paper/blob/main/CHANGELOG.md
15
+ Keywords: pytorch,optimizer,low-vram,fp16,bf16,triton,deep-learning,llm,transformer
16
+ Classifier: Development Status :: 4 - Beta
17
+ Classifier: Intended Audience :: Science/Research
18
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
19
+ Classifier: License :: OSI Approved :: Apache Software License
20
+ Classifier: Programming Language :: Python :: 3
21
+ Classifier: Programming Language :: Python :: 3.10
22
+ Classifier: Programming Language :: Python :: 3.11
23
+ Classifier: Programming Language :: Python :: 3.12
24
+ Classifier: Programming Language :: Python :: 3.13
25
+ Classifier: Programming Language :: Python :: 3.14
26
+ Classifier: Operating System :: OS Independent
27
+ Requires-Python: >=3.10
28
+ Description-Content-Type: text/markdown
29
+ License-File: LICENSE
30
+ Requires-Dist: torch>=2.0.0
31
+ Provides-Extra: triton
32
+ Requires-Dist: triton; extra == "triton"
33
+ Dynamic: license-file
34
+
35
+ # LuminaV Optimizer
36
+
37
+ > **We Were Too Broke for AdamW So We Trapped Gradients in a Hyperbolic Straitjacket and Hired a Traffic Cop to Slap Them**
38
+
39
+ [![PyPI - Version](https://img.shields.io/pypi/v/luminav.svg?style=flat-square)](https://pypi.org/project/luminav/)
40
+ [![Software DOI](https://img.shields.io/badge/Software_DOI-10.57967%2Fhf%2F10365-purple?style=flat-square)](https://doi.org/10.57967/hf/10365)
41
+ [![Paper DOI](https://img.shields.io/badge/Paper_DOI-10.57967%2Fhf%2F10270-darkblue?style=flat-square)](https://doi.org/10.57967/hf/10270)
42
+ [![Python Version](https://img.shields.io/badge/python-3.10+-blue?style=flat-square)](https://pypi.org/project/luminav/)
43
+ [![License](https://img.shields.io/badge/License-Apache%202.0-yellow?style=flat-square)](LICENSE)
44
+
45
+ ---
46
+
47
+ ## Official Repository & Research Paper
48
+
49
+ - **Official Repository & Issue Tracker:** [https://huggingface.co/cloverx-id/LuminaV-Optimizer-Paper](https://huggingface.co/cloverx-id/LuminaV-Optimizer-Paper)
50
+ - **Primary Research Paper Archive:** [https://huggingface.co/cloverx-id/XoneLM-1.0-Paper](https://huggingface.co/cloverx-id/XoneLM-1.0-Paper)
51
+ - **Software DOI:** [10.57967/hf/10365](https://doi.org/10.57967/hf/10365)
52
+ - **Paper DOI:** [10.57967/hf/10270](https://doi.org/10.57967/hf/10270)
53
+
54
+ ---
55
+
56
+ ## What's New in v1.1.5 (Latest Release)
57
+
58
+ The **v1.1.5** release introduces the **Zero-VRAM FP16 Numerical Safety Shield**, resolving early pretraining instability and eliminating `NaN` collapses in pure half-precision training without allocating 4-byte master weights:
59
+
60
+ - **Dynamic Epsilon Floor for FP16:** Automatically floors effective epsilon to `max(eps, 1e-4)` for `torch.float16` parameters, preventing the bias-corrected scale factor `c2` from underflowing into IEEE-754 subnormal/zero limits (~5.96e-8) and eliminating step-1 `0.0 / 0.0 = NaN` errors on zero-gradient parameters (e.g., unselected vocabulary tokens, padding, or dropout paths).
61
+ - **On-Chip Second-Moment Clamping:** Added an upper-bound register ceiling (`65,000.0`) to the centered innovation variance in Triton kernels and PyTorch loops before storing into FP16 pointers. This prevents unclipped initial gradient spikes (`|g| > 256`) from exceeding the FP16 maximum dynamic range (65,504) and permanently saturating state buffers to `+inf`.
62
+ - **Zero-Sigma Division Guard:** Embedded an explicit `sigma + 1e-6` protection across all Triton kernel passes and PyTorch fallbacks to ensure division safety during cold-start iterations.
63
+ - **Native Hardware Capability Detection:** Integrated `_supports_native_bf16()` to automatically detect Nvidia Ampere SM80+ architectures for optimal native execution.
64
+ - **Empirically Proven on Tesla T4:** Empirically verified by pretraining `SmolLM-135M` from scratch on Wikimedia Wikipedia in pure FP16 on an Nvidia Tesla T4 GPU for 100 consecutive steps with zero `NaN` occurrences—even with external gradient clipping completely disabled.
65
+
66
+ ---
67
+
68
+ ## Overview
69
+
70
+ **LuminaV** is a master-free, memory-efficient adaptive optimizer engineered specifically for deep learning workloads running directly in low precision (`FP16` / `BF16`) without maintaining redundant 4-byte FP32 master weights.
71
+
72
+ By combining **Centered Innovation Variance**, **Hyperbolic Tangent (tanh) Coordinate Bounding**, a **Directional Traffic-Cop Mask**, and **On-Chip Bitwise Stochastic Rounding**, LuminaV eliminates the standard 16-byte-per-parameter memory tax imposed by AdamW while preventing weight stagnation and numerical explosions.
73
+
74
+ ---
75
+
76
+ ## Key Features
77
+
78
+ 1. **Zero Master-Weight Copies:** Directly mutates parameter weights in native `FP16` or `BF16`, eliminating the redundant 4-byte FP32 master weight allocation.
79
+ 2. **On-Chip Bitwise Stochastic Rounding (SR):** Implements in-register bitcast hashing in Triton to provide unbiased stochastic rounding, preventing weight stagnation during fine-grained updates or learning rate decay.
80
+ 3. **Hyperbolic tanh Bounding Envelope:** Maps normalized momentum through a `(-1.0, 1.0)` transfer function, guaranteeing coordinate updates cannot explode beyond the step learning rate.
81
+ 4. **The Traffic-Cop Directional Gate:** Dynamically eliminates coordinate updates whenever historical momentum conflicts with the incoming mini-batch gradient direction (`u_t · g_t <= 0`).
82
+ 5. **Centered Innovation Variance:** Tracks centered innovation dispersion `(g_t - m_t)^2` rather than uncentered raw second moments, suppressing variance inflation during confident descent.
83
+ 6. **Dual Execution Engine:** Fully accelerated custom OpenAI Triton kernels for CUDA devices, paired with vectorized C++ `torch._foreach` multi-tensor fallbacks.
84
+
85
+ ---
86
+
87
+ ## Installation
88
+
89
+ Install directly via PyPI:
90
+
91
+ ```bash
92
+ pip install luminav
93
+ ```
94
+
95
+ For GPU acceleration via OpenAI Triton:
96
+
97
+ ```bash
98
+ pip install luminav[triton]
99
+ ```
100
+
101
+ Or install in editable mode from source:
102
+
103
+ ```bash
104
+ git clone https://huggingface.co/cloverx-id/LuminaV-Optimizer-Paper
105
+ cd LuminaV-Optimizer-Paper
106
+ pip install -e .
107
+ ```
108
+
109
+ ---
110
+
111
+ ## Quickstart
112
+
113
+ ```python
114
+ import torch
115
+ import torch.nn as nn
116
+ from luminav import LuminaV
117
+
118
+ # 1. Instantiate model directly in native half precision (e.g. BF16 or FP16)
119
+ device = "cuda" if torch.cuda.is_available() else "cpu"
120
+ model = nn.Linear(1024, 1024).to(device=device, dtype=torch.float16)
121
+
122
+ # 2. Initialize LuminaV
123
+ optimizer = LuminaV(
124
+ model.parameters(),
125
+ lr=8e-4,
126
+ betas=(0.9, 0.999),
127
+ eps=1e-8,
128
+ weight_decay=0.08,
129
+ tau=0.8,
130
+ buffer=2, # 2 = Dual-Buffer (Standard), 1 = Single-Buffer (Low VRAM)
131
+ stochastic_rounding=True,
132
+ execution="auto"
133
+ )
134
+
135
+ # 3. Standard training loop
136
+ data = torch.randn(32, 1024, device=device, dtype=torch.float16)
137
+ target = torch.randn(32, 1024, device=device, dtype=torch.float16)
138
+ criterion = nn.MSELoss()
139
+
140
+ optimizer.zero_grad(set_to_none=True)
141
+ output = model(data)
142
+ loss = criterion(output, target)
143
+ loss.backward()
144
+
145
+ optimizer.step()
146
+ ```
147
+
148
+ ---
149
+
150
+ ## Parameter Reference
151
+
152
+ | Parameter | Type | Default | Description |
153
+ | :--- | :--- | :--- | :--- |
154
+ | `params` | `iterable` | *Required* | Iterable of parameters to optimize or dicts defining parameter groups. |
155
+ | `lr` | `float` | `8e-4` | Learning rate (eta). |
156
+ | `betas` | `Tuple[float, float]` | `(0.9, 0.999)` | Coefficients (beta1, beta2) for running momentum and centered innovation variance. |
157
+ | `eps` | `float` | `1e-8` | Numerical stability term (epsilon). Automatically floored to `1e-4` in FP16 to prevent subnormal underflow. |
158
+ | `weight_decay` | `float` | `8e-2` | Decoupled weight decay coefficient (lambda). |
159
+ | `tau` | `float` | `0.8` | Analytical bias correction temperature parameter (tau). |
160
+ | `alpha_ss` | `float` | `0.5` | Softsign dampening factor (alpha_ss) used in single-buffer mode (`buffer=1`). |
161
+ | `cautious` | `bool` | `True` | If `True`, enables Traffic-Cop directional verification masking. |
162
+ | `cautious_clamp_min` | `float` | `0.2` | Safety floor density clamp (gamma_min) preventing division by zero in masked normalization. |
163
+ | `buffer` | `int` | `2` | Buffer mode: `2` (Dual-buffer tracking m_t and v_t) or `1` (Single-buffer scalar RMS tracking). |
164
+ | `stochastic_rounding` | `bool` | `True` | Enables bitwise stochastic rounding on native FP16/BF16 weights. |
165
+ | `execution` | `str` | `"auto"` | Execution engine: `"auto"`, `"triton"`, `"foreach"`, or `"single"`. |
166
+
167
+ ---
168
+
169
+ ## Citation
170
+
171
+ If you utilize LuminaV in your research or applications, please cite both the foundational paper and this software implementation:
172
+
173
+ ```bibtex
174
+ # 1. To cite the official research paper & theoretical mechanics
175
+ @misc{luminamoon2026luminav_paper,
176
+ author = {{Silver Moon (cloverxion)}},
177
+ organization = {Lumina Moon},
178
+ title = {{LuminaV: We Were Too Broke for AdamW So We Trapped Gradients in a Hyperbolic Straitjacket and Hired a Traffic Cop to Slap Them}},
179
+ year = {2026},
180
+ publisher = {Hugging Face},
181
+ doi = {10.57967/hf/10270},
182
+ url = {https://huggingface.co/cloverx-id/XoneLM-1.0-Paper}
183
+ }
184
+
185
+ # 2. To cite this software implementation & standalone codebase
186
+ @software{luminamoon2026luminav_code,
187
+ author = {{Silver Moon (cloverxion)}},
188
+ organization = {Lumina Moon},
189
+ title = {{LuminaV Optimizer: Official PyTorch Implementation}},
190
+ year = {2026},
191
+ publisher = {Hugging Face},
192
+ version = {1.1.5},
193
+ doi = {10.57967/hf/10365},
194
+ url = {https://huggingface.co/cloverx-id/LuminaV-Optimizer-Paper}
195
+ }
196
+ ```
197
+
198
+ ---
199
+
200
+ ## License
201
+
202
+ Apache License 2.0. See [LICENSE](LICENSE) for full terms.
@@ -0,0 +1,7 @@
1
+ luminav/__init__.py,sha256=9QXx2WxRhVYNY5FGYkTXGKumRiS6dOvVE5qLlRxaYUc,134
2
+ luminav/luminav.py,sha256=HhZG-1igoI4Q_mpiwtNBOzOWSK16l8prKA3_L_pN7Zg,28360
3
+ luminav-1.1.5.dist-info/licenses/LICENSE,sha256=vfIk0HOJipiGkxSJLASwBoUIc4mkDsspxhxb_5DpKVQ,762
4
+ luminav-1.1.5.dist-info/METADATA,sha256=5wL-XC_rpHWSXOFB8iBmCeRO3QhUlloGDG_bcLeSW9U,10086
5
+ luminav-1.1.5.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
6
+ luminav-1.1.5.dist-info/top_level.txt,sha256=8yzfkXhUGr1S4izcM8vOu7nw7MfZRRwNAJu3mto91O0,8
7
+ luminav-1.1.5.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,17 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ Copyright (c) 2026 Lumina Moon and Contributors.
6
+
7
+ Licensed under the Apache License, Version 2.0 (the "License");
8
+ you may not use this file except in compliance with the License.
9
+ You may obtain a copy of the License at
10
+
11
+ http://www.apache.org/licenses/LICENSE-2.0
12
+
13
+ Unless required by applicable law or agreed to in writing, software
14
+ distributed under the License is distributed on an "AS IS" BASIS,
15
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
16
+ See the License for the specific language governing permissions and
17
+ limitations under the License.
@@ -0,0 +1 @@
1
+ luminav