denpex 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- denpex-1.0.0.data/data/share/denpex/denpex_patterns.json +1 -0
- denpex-1.0.0.dist-info/METADATA +17 -0
- denpex-1.0.0.dist-info/RECORD +9 -0
- denpex-1.0.0.dist-info/WHEEL +5 -0
- denpex-1.0.0.dist-info/entry_points.txt +2 -0
- denpex-1.0.0.dist-info/top_level.txt +3 -0
- denpex.py +2017 -0
- denpex_local.py +271 -0
- denpex_telemetry.py +417 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":1,"generated":"2026-06-21T02:55:14.406Z","patterns":[{"type":"CUDA_OOM","re":"torch\\.cuda\\.OutOfMemoryError|CUDA out of memory|RuntimeError.*CUDA.*out of memory","flags":"i","confidence":95,"summary":"GPU ran out of memory during training.","action":"Reduce batch size, enable gradient checkpointing, or use PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True","resumeFrom":"latest checkpoint"},{"type":"NCCL_TIMEOUT","re":"NCCL timeout|Watchdog caught collective operation timeout|ncclTimeout","flags":"i","confidence":90,"summary":"NCCL collective timed out. One rank stopped participating.","action":"Check failing rank with nvidia-smi, kill zombie processes, and restart with NCCL_TIMEOUT=6000","resumeFrom":"latest checkpoint"},{"type":"NCCL_CASCADE","re":"NCCL timeout detected on \\d+\\/\\d+ ranks|\\d+ ranks.*waiting at.*barrier","flags":"i","confidence":93,"summary":"NCCL timeout cascade from single failed rank.","action":"Find root cause rank, fix the issue, and resume from checkpoint.","resumeFrom":"latest checkpoint"},{"type":"OOM_FRAGMENTATION","re":"expandable_segments|memory fragmentation|reserved.*allocated.*free","flags":"i","confidence":92,"summary":"GPU memory is fragmented.","action":"Set PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True and restart.","resumeFrom":"latest checkpoint"},{"type":"NAN_LOSS","re":"loss.{0,12}(=|:|is|became|to).{0,5}nan|nan.{0,8}loss|GradScaler.*overflow","flags":"i","confidence":92,"summary":"Training loss became NaN.","action":"Enable gradient clipping, lower learning rate 5-10x, check for bad data batch.","resumeFrom":"last checkpoint before divergence"},{"type":"CHECKPOINT_CORRUPTION","re":"checkpoint.*corrupt|failed to load.*checkpoint|UnpicklingError|PytorchStreamReader.*failed","flags":"i","confidence":92,"summary":"Checkpoint file is corrupted.","action":"Resume from the previous checkpoint.","resumeFrom":"previous (older) checkpoint"},{"type":"NVLS_ERROR","re":"Failed to bind NVLink SHARP \\(NVLS\\) Multicast memory|NVSwitch multicast resources\\/slot ids are used|transport\\/nvls\\.cc[^\\n]{0,40}Cuda failure 1 'invalid argument'","flags":"i","confidence":90,"summary":"NVLink SHARP (NVLS) multicast setup failed (slot exhaustion or invalid argument).","action":"Disable NVLS with NCCL_NVLS_ENABLE=0; on NCCL 2.29.x downgrade to 2.28.x. Check Fabric Manager multicast slot usage.","resumeFrom":"latest checkpoint"},{"type":"GPU_XID_MMU","re":"Xid[^\\n]{0,40}\\b31\\b[^\\n]{0,80}(MMU Fault|FAULT_PDE)","flags":"i","confidence":90,"summary":"GPU MMU page fault (Xid 31), often an NVLS/Fabric Manager cascade.","action":"Stop nvidia-fabricmanager, run nvidia-smi --gpu-reset, then restart fabricmanager; reboot if it persists. Never restart Fabric Manager under load.","resumeFrom":"latest checkpoint"},{"type":"GPU_OFF_BUS","re":"Xid[^\\n]{0,40}\\b79\\b|GPU has fallen off the bus|fallen off the bus","flags":"i","confidence":96,"summary":"GPU fell off the PCIe bus (Xid 79) - critical hardware/connectivity fault.","action":"Reboot the node (a reset alone rarely recovers it). Check power/thermals; reseat or RMA the GPU if it recurs.","resumeFrom":"latest checkpoint"},{"type":"GSP_RPC_TIMEOUT","re":"Xid[^\\n]{0,40}\\b119\\b|GSP[^\\n]{0,30}RPC[^\\n]{0,30}timeout","flags":"i","confidence":90,"summary":"GPU System Processor (GSP) stopped responding (Xid 119).","action":"Reset the GPU (nvidia-smi --gpu-reset) or reboot. Update driver/GSP firmware to a qualified version.","resumeFrom":"latest checkpoint"},{"type":"GPU_XID_ECC","re":"Xid[^\\n]{0,40}\\b(48|94|95)\\b|double-bit ECC|uncorrectable ECC|row.?remap","flags":"i","confidence":92,"summary":"GPU ECC error (Xid 48/94/95) or memory row-remapping event.","action":"Drain the node and reset the GPU to apply any pending row remap. Track the remap-count trend; RMA if remaps accumulate.","resumeFrom":"latest checkpoint"},{"type":"NVLINK_ERROR","re":"Xid[^\\n]{0,40}\\b(145|149)\\b|NVLink[^\\n]{0,20}error","flags":"i","confidence":88,"summary":"NVLink error (Xid 145/149) - link degraded or failed.","action":"Compare per-GPU NVLink bandwidth (all 8 low = NVSwitch; one low = that link). Check nvidia-smi nvlink -e; drain/reset the implicated component.","resumeFrom":"latest checkpoint"},{"type":"ROCE_GID_FAIL","re":"read failed in ncclIbRoceGetVersionNum","flags":"i","confidence":90,"summary":"NCCL could not read the RoCE GID (zero-GID / macvlan layout).","action":"Set NCCL_IB_GID_INDEX to a valid RoCE v2 entry (show_gids); upgrade to NCCL 2.26.2+ which skips zero-GID entries.","resumeFrom":"latest checkpoint"},{"type":"NCCL_NET_TOPOLOGY","re":"Could not find NET with id 0|Could not find a path for pattern","flags":"i","confidence":88,"summary":"NCCL network topology lookup failed (NIC fusion remapped NET/0).","action":"Set NCCL_NET_MERGE_LEVEL=LOC or upgrade to NCCL 2.27.x; avoid NCCL 2.26.2 on partial-node IB allocations.","resumeFrom":"latest checkpoint"},{"type":"CUDA_FABRIC_IMPORT","re":"cuMemImportFromShareableHandle|Cuda failure 101 'invalid device ordinal'","flags":"i","confidence":90,"summary":"Fabric handle import failed (CUDA 101) - CUDA driver 570.00 bug on MNNVL/GH200.","action":"Upgrade the CUDA driver beyond 570.00. Ensure consistent device enumeration across containers.","resumeFrom":"latest checkpoint"},{"type":"ROCE_QP_TIMEOUT","re":"ibv_modify_qp failed with error Connection timed out|ibv_modify_qp[^\\n]{0,40}errno 110","flags":"i","confidence":88,"summary":"RDMA queue-pair setup timed out (errno 110), often during checkpoint save at scale on RoCE.","action":"Tune RoCE PFC/ECN; verify NCCL_IB_GID_INDEX; raise NCCL_IB_TIMEOUT; stagger/shard checkpoint writes.","resumeFrom":"latest checkpoint"},{"type":"INFLIGHT_PARAM","re":"Cannot partition a param in flight|still have inflight params","flags":"i","confidence":92,"summary":"DeepSpeed ZeRO-3 parameter still in flight (save or backward).","action":"Align save interval to gradient_accumulation_steps and upgrade DeepSpeed; for dynamic/RLHF graphs keep the per-step param set consistent.","resumeFrom":"latest checkpoint"},{"type":"CHECKPOINT_RACE","re":"FileExistsError[^\\n]{0,30}Errno 17|File exists[^\\n]{0,30}copytree|copytree[^\\n]{0,30}File exists","flags":"i","confidence":90,"summary":"ZeRO-3 NVMe-offload checkpoint race: ranks write the same directory.","action":"Upgrade DeepSpeed to a version using per-rank checkpoint subdirectories; clear stale offloaded_tensors/ between runs.","resumeFrom":"previous (older) checkpoint"},{"type":"HOST_RAM_OOM","re":"exits with return code = ?-9|return code = ?-9|killed[^\\n]{0,20}signal 9","flags":"i","confidence":80,"summary":"Process OOM-killed on host RAM (SIGKILL -9), often ZeRO CPU/NVMe offload.","action":"Set pin_memory:false in the offload config, reduce stage3_max_live_parameters, disable CPU offload, or add nodes.","resumeFrom":"latest checkpoint"},{"type":"LOSS_SCALE_UNDERFLOW","re":"Current loss scale already at minimum","flags":"i","confidence":92,"summary":"fp16 loss scale hit minimum from continuous overflow.","action":"Switch to bf16 (A100/H100), or reduce loss_scale_window and learning rate. Match training precision to pretraining (bf16).","resumeFrom":"last checkpoint before divergence"},{"type":"CUDA_UNKNOWN","re":"Cuda failure 999 'unknown error'","flags":"i","confidence":78,"summary":"Generic CUDA unknown error (999), usually a GPU in a bad state or a version mismatch.","action":"Run with NCCL_DEBUG=INFO; scan dmesg for Xid on the failing rank; align NCCL/torch/CUDA versions; reset the GPU/node.","resumeFrom":"latest checkpoint"},{"type":"ROCE_MTU_MISMATCH","re":"Got completion with error 12.*vendor err 129|completion with error 12.*vendor_err 129","flags":"i","confidence":90,"summary":"RoCE MTU mismatch between NIC and switch causing RDMA QP fatal errors.","action":"Set MTU to 4096 on both NICs and switch ports (sudo ip link set dev ib0 mtu 4096); set NCCL_IB_RETRY_CNT=7 and NCCL_IB_TIMEOUT=22.","resumeFrom":"latest checkpoint"},{"type":"MISMATCHED_COLLECTIVE","re":"Mismatched collective detected","flags":"i","confidence":92,"summary":"Different ranks executing different NCCL collective operations — a programming error in the training loop.","action":"Ensure all ranks call the same collectives in the same order; remove conditional collectives or add dummy collectives on the non-active path; set NCCL_DEBUG_SUBSYS=COLL to trace.","resumeFrom":"latest checkpoint"},{"type":"OFI_MEMLOCK","re":"NET\\/OFI Unable to register memory.*RC: ?12|Unable to register memory.*RC: ?12","flags":"i","confidence":90,"summary":"Insufficient memlock limit for RDMA memory registration — OFI transport cannot pin GPU memory.","action":"Set memlock to unlimited in /etc/security/limits.conf (* soft/hard memlock unlimited); for Docker pass --ulimit memlock=-1:-1; for K8s add IPC_LOCK capability.","resumeFrom":"latest checkpoint"},{"type":"SHM_EXHAUSTION","re":"posix_fallocate failed.*No space left on device|ncclShmemOpen.*cannot allocate","flags":"i","confidence":92,"summary":"Docker /dev/shm too small (default 64MB) for NCCL shared memory IPC.","action":"Pass --shm-size=8g to docker run; for K8s use emptyDir with medium: Memory; or use --ipc=host.","resumeFrom":"latest checkpoint"},{"type":"ACS_GDR_FAIL","re":"Got completion with error 4.*vendor err 81|completion with error 4.*vendor_err 81","flags":"i","confidence":88,"summary":"ACS disabled in BIOS breaks GPU Direct RDMA — RDMA transactions between GPU and NIC rejected.","action":"Enable ACS in BIOS under PCIe settings; or disable GDR with NCCL_NET_GDR_LEVEL=0.","resumeFrom":"latest checkpoint"},{"type":"NCCL_TOPOLOGY_XML","re":"Attribute busid of node nic not found","flags":"i","confidence":88,"summary":"NCCL topology XML missing NIC PCI bus ID — cannot associate NIC with NUMA node/GPU.","action":"Set NCCL_TOPO_FILE to a manually created topology XML; or set NCCL_IB_HCA explicitly to the correct device name.","resumeFrom":"latest checkpoint"},{"type":"NCCL_BUFFSIZE","re":"NCCL_BUFFSIZE set by environment.*Watchdog.*timeout|NCCL_BUFFSIZE.*timeout","flags":"i","confidence":82,"summary":"NCCL_BUFFSIZE too large for Socket transport — exceeds OS socket buffer limits causing stall.","action":"Remove custom NCCL_BUFFSIZE and use default; or increase OS socket buffers (net.core.rmem_max, wmem_max).","resumeFrom":"latest checkpoint"},{"type":"BF16_NORM_UNDERFLOW","re":"assert all_groups_norm > 0|all_groups_norm.*bf16","flags":"i","confidence":90,"summary":"bf16 gradient norm underflow — norms can round to zero in bf16 precision (7 mantissa bits).","action":"Upgrade DeepSpeed to >=0.13.5 which computes norms in fp32; or use fp16; or increase learning rate.","resumeFrom":"latest checkpoint"},{"type":"MOE_LEAF_MISSING","re":"NCCL timeout.*MoE|hang.*MoE.*ZeRO-3|ZeRO.*hang.*expert","flags":"i","confidence":85,"summary":"MoE model + ZeRO-3 hang — MoE blocks need leaf module marking for parameter gathering.","action":"Mark MoE blocks as leaf modules: deepspeed.zero.set_z3_leaf_modules(model, [MoEBlockClass]); or use ZeRO stage 2.","resumeFrom":"latest checkpoint"},{"type":"OVERLAP_CONTIG_NAN","re":"grad_norm.*nan.*overlap_comm|overlap_comm.*True.*contiguous_gradients.*True.*nan","flags":"i","confidence":90,"summary":"NaN from gradient buffer reuse race when both overlap_comm and contiguous_gradients are True in DeepSpeed ZeRO-3.","action":"Disable one: set overlap_comm: false or contiguous_gradients: false in DeepSpeed config.","resumeFrom":"last checkpoint before divergence"}],"encyclopedia":[{"slug":"cpu-offloading-overhead","title":"CPU Offloading Overhead","category":"Memory","text":"CPU Offloading Overhead Memory Training is 2-5x slower with offloading CPU memory usage is high during training GPU SM utilization is low cpu-offloading memory performance deepspeed fsdp pcie Training is much slower with CPU offloading GPU utilization is low CPU memory is heavily used CPU-GPU transfer bandwidth is bottleneck Offloading too much data CPU memory not fast enough PCIe bandwidth limits offloading performance","anchorText":"CPU Offloading Overhead Training is 2-5x slower with offloading CPU memory usage is high during training GPU SM utilization is low cpu-offloading memory performance deepspeed fsdp pcie CPU-GPU transfer bandwidth is bottleneck Offloading too much data CPU memory not fast enough PCIe bandwidth limits offloading performance ","action":"Reduce offloaded layers/parameters","steps":["Reduce offloaded layers/parameters","Use NVMe offload for less frequently accessed data","Profile to find optimal offload ratio","Use faster CPU memory (DDR5)","Consider gradient checkpointing instead of full offload"]},{"slug":"nan-detection-and-skip","title":"NaN Detection and Skip","category":"Training Stability","text":"NaN Detection and Skip Training Stability Loss is NaN for one step but then recovers Gradient norm reports NaN Parameter updates are skipped due to NaN nan-detection gradscaler mixed-precision skip-update training-stability Training continues after NaN occurs Loss is NaN for some steps but recovers NaN detection prevents training crash Gradient scaler detects NaN and skips update NaN detected in loss but optimizer step is skipped NaN in one rank doesn't propagate to others Mixed precision underflow causes NaN","anchorText":"NaN Detection and Skip Loss is NaN for one step but then recovers Gradient norm reports NaN Parameter updates are skipped due to NaN nan-detection gradscaler mixed-precision skip-update training-stability Gradient scaler detects NaN and skips update NaN detected in loss but optimizer step is skipped NaN in one rank doesn't propagate to others Mixed precision underflow causes NaN ","action":"Use GradScaler with default NaN detection: scaler.step(optimizer) skips on NaN","steps":["Use GradScaler with default NaN detection: scaler.step(optimizer) skips on NaN","Monitor NaN frequency: too many indicates real problem","Combine with gradient clipping: scaler.unscale_(optimizer); clip_grad_norm_()","Use bf16 instead of fp16 to avoid underflow NaN","Log NaN events to understand root cause"]},{"slug":"nccl-broadcast-hang","title":"NCCL Broadcast Hang","category":"Distributed Training","text":"NCCL Broadcast Hang Distributed Training NCCL WARN broadcast timeout NCCL hangs at all ranks except one torch.distributed.broadcast() hangs forever nccl broadcast hang distributed collective initialization Training hangs at broadcast operation One rank doesn't respond to broadcast Distributed init fails at broadcast step One rank stuck or slow Network partition between nodes Rank not initialized for broadcast Broadcast source rank has issue","anchorText":"NCCL Broadcast Hang NCCL WARN broadcast timeout NCCL hangs at all ranks except one torch.distributed.broadcast() hangs forever nccl broadcast hang distributed collective initialization One rank stuck or slow Network partition between nodes Rank not initialized for broadcast Broadcast source rank has issue ","action":"Identify slow rank with NCCL_DEBUG=INFO","steps":["Identify slow rank with NCCL_DEBUG=INFO","Check network between nodes during broadcast","Verify all ranks are initialized: torch.distributed.is_initialized()","Use broadcast timeout: dist.broadcast(tensor, src, group=group) with timeout","Restart training if broadcast hangs persistently"]},{"slug":"augmentation-pipeline-error","title":"Augmentation Pipeline Error","category":"Data Pipeline","text":"Augmentation Pipeline Error Data Pipeline Augmentation produces images with wrong shape or dtype Transforms fail with PIL/torchvision errors Augmented batches have NaN values augmentation data-pipeline transforms image vision albumentations Training accuracy is lower than expected Augmented images look corrupted Training crashes intermittently with augmentation errors Transform receives wrong input type or shape Random seed not set consistently across workers Augmentation order or parameters not appropriate for data Num_workers conflict with main process augmentation","anchorText":"Augmentation Pipeline Error Augmentation produces images with wrong shape or dtype Transforms fail with PIL/torchvision errors Augmented batches have NaN values augmentation data-pipeline transforms image vision albumentations Transform receives wrong input type or shape Random seed not set consistently across workers Augmentation order or parameters not appropriate for data Num_workers conflict with main process augmentation ","action":"Set deterministic augmentation: torch.manual_seed(42)","steps":["Set deterministic augmentation: torch.manual_seed(42)","Validate augmentation output shape and dtype","Use Albumentations or torchvision transforms with proper error handling","Test augmentation pipeline on sample data before training","Use Denpex to validate augmentation pipeline correctness"]},{"slug":"raid-storage-failure","title":"RAID Storage Failure","category":"Infrastructure","text":"RAID Storage Failure Infrastructure dmesg: I/O error, dev sda, sector X mdadm: disk failure detected Storage performance is much slower than usual raid storage hardware failure backup data-integrity Training crashes with disk read/write errors Storage performance degrades Some training data becomes inaccessible Hard drive failure in RAID array RAID controller failure Storage subsystem degradation Disk SMART errors preceding failure","anchorText":"RAID Storage Failure dmesg: I/O error, dev sda, sector X mdadm: disk failure detected Storage performance is much slower than usual raid storage hardware failure backup data-integrity Hard drive failure in RAID array RAID controller failure Storage subsystem degradation Disk SMART errors preceding failure ","action":"Monitor RAID health: mdadm --detail /dev/md0","steps":["Monitor RAID health: mdadm --detail /dev/md0","Set up SMART monitoring: smartctl -a /dev/sda","Replace failing disks immediately","Backup training data regularly","Use cloud storage as backup"]},{"slug":"activation-memory-spike","title":"Activation Memory Spike","category":"Memory","text":"Activation Memory Spike Memory OOM on batches with long sequences Memory spikes correlate with specific input shapes OOM is harder to reproduce activation-memory attention sequence-length peak-memory flash-attention OOM on specific batches or sequence lengths Memory spikes during attention or large matmul OOM is non-deterministic Attention has quadratic memory in sequence length Specific batch has unusually long sequences Activation memory varies with input shape No memory headroom for peaks","anchorText":"Activation Memory Spike OOM on batches with long sequences Memory spikes correlate with specific input shapes OOM is harder to reproduce activation-memory attention sequence-length peak-memory flash-attention Attention has quadratic memory in sequence length Specific batch has unusually long sequences Activation memory varies with input shape No memory headroom for peaks ","action":"Profile peak memory: torch.cuda.max_memory_allocated()","steps":["Profile peak memory: torch.cuda.max_memory_allocated()","Limit sequence length: tokenizer(max_length=512)","Enable gradient checkpointing for attention layers","Use Flash Attention for memory-efficient attention","Batch similar-length sequences to reduce memory variance"]},{"slug":"python-path-conflict","title":"Python Path Conflict","category":"Environment","text":"Python Path Conflict Environment Module loaded from wrong location Different module version than expected ImportError: cannot import name from installed version python path import environment module version Wrong module version is loaded ImportError: cannot import name from different version Different behavior in different environments PYTHONPATH includes wrong directory Site-packages order puts wrong version first Conda and pip packages conflict Virtual environment not activated","anchorText":"Python Path Conflict Module loaded from wrong location Different module version than expected ImportError: cannot import name from installed version python path import environment module version PYTHONPATH includes wrong directory Site-packages order puts wrong version first Conda and pip packages conflict Virtual environment not activated ","action":"Check Python path: python -c 'import sys; print(sys.path)'","steps":["Check Python path: python -c 'import sys; print(sys.path)'","Use virtual environments: python -m venv venv","Activate environment before running: source venv/bin/activate","Avoid mixing conda and pip","Use absolute imports: from package.module import function"]},{"slug":"lr-not-warmup-decay","title":"LR Without Warmup or Decay","category":"Training Stability","text":"LR Without Warmup or Decay Training Stability Loss decreases very slowly Loss jumps at the start of training Final loss is higher than with proper schedule learning-rate warmup decay scheduler convergence training-stability Training converges slowly Final loss is higher than expected Training is unstable at start Constant LR is too high causing oscillation Constant LR is too low causing slow convergence No warmup causes initial loss spikes No decay means final LR is same as initial","anchorText":"LR Without Warmup or Decay Loss decreases very slowly Loss jumps at the start of training Final loss is higher than with proper schedule learning-rate warmup decay scheduler convergence training-stability Constant LR is too high causing oscillation Constant LR is too low causing slow convergence No warmup causes initial loss spikes No decay means final LR is same as initial ","action":"Add linear warmup: torch.optim.lr_scheduler.LinearLR(optimizer, start_factor=0.01, total_iters=warmup_steps)","steps":["Add linear warmup: torch.optim.lr_scheduler.LinearLR(optimizer, start_factor=0.01, total_iters=warmup_steps)","Add cosine decay: torch.optim.lr_scheduler.CosineAnnealingLR","Combine warmup and decay for best results","Use 1cycle policy for fast convergence: torch.optim.lr_scheduler.OneCycleLR","Tune LR schedule with learning rate range test"]},{"slug":"nccl-cuda-failure","title":"NCCL CUDA Failure","category":"Communication","text":"NCCL CUDA Failure Communication NCCL WARN CUDA error RuntimeError: NCCL error in: Cuda failure in collective nccl cuda error communication distributed gpu NCCL operations fail with CUDA error Training crashes with CUDA error in NCCL context NCCL collective returns error code GPU has CUDA error GPU memory corruption Driver issues CUDA toolkit incompatibility with NCCL","anchorText":"NCCL CUDA Failure NCCL WARN CUDA error RuntimeError: NCCL error in: Cuda failure in collective nccl cuda error communication distributed gpu GPU has CUDA error GPU memory corruption Driver issues CUDA toolkit incompatibility with NCCL ","action":"Check CUDA error first: nvidia-smi and torch.cuda.is_available()","steps":["Check CUDA error first: nvidia-smi and torch.cuda.is_available()","Reset GPU: nvidia-smi --gpu-reset","Verify CUDA and NCCL compatibility","Update CUDA toolkit and NCCL to compatible versions","Use Denpex to correlate NCCL errors with CUDA errors"]},{"slug":"training-restart-stuck","title":"Training Restart Stuck","category":"Reliability","text":"Training Restart Stuck Reliability New training process hangs at init GPU shows memory used by previous process SLURM/K8s shows old process still running restart stuck cleanup reliability zombie cuda-context Restarted training hangs at startup Previous training's resources still held GPU memory not released from previous run Previous training process not fully killed CUDA context not released Distributed training group not destroyed File handles not closed properly","anchorText":"Training Restart Stuck New training process hangs at init GPU shows memory used by previous process SLURM/K8s shows old process still running restart stuck cleanup reliability zombie cuda-context Previous training process not fully killed CUDA context not released Distributed training group not destroyed File handles not closed properly ","action":"Kill all related processes: pkill -9 -f python","steps":["Kill all related processes: pkill -9 -f python","Reset GPU: nvidia-smi --gpu-reset","Wait for old training to fully terminate","Use proper cleanup in training script: try/finally with destroy_process_group()","Implement health check before starting new training"]},{"slug":"lr-finder-result-misuse","title":"Learning Rate Finder Result Misuse","category":"Training Stability","text":"Learning Rate Finder Result Misuse Training Stability Loss explodes at finder-recommended LR Training is unstable at finder-recommended LR No improvement from using finder result learning-rate lr-finder hyperparameter tuning training-stability Learning rate from finder is too high Training diverges after using finder recommendation Final loss is worse than expected Learning rate finder picks highest stable LR which is at edge of stability Using peak LR instead of middle of decreasing range No warmup after high LR Single batch test not representative","anchorText":"Learning Rate Finder Result Misuse Loss explodes at finder-recommended LR Training is unstable at finder-recommended LR No improvement from using finder result learning-rate lr-finder hyperparameter tuning training-stability Learning rate finder picks highest stable LR which is at edge of stability Using peak LR instead of middle of decreasing range No warmup after high LR Single batch test not representative ","action":"Use middle of the decreasing loss range, not the minimum","steps":["Use middle of the decreasing loss range, not the minimum","Reduce finder-recommended LR by 3-10x for stability","Always add warmup after finder result","Use multiple seeds for reliability","Test on full training run, not just finder"]},{"slug":"memory-summary-tool","title":"Memory Summary Tool Usage","category":"Memory","text":"Memory Summary Tool Usage Memory Memory summary shows large reserved memory Snapshot shows many small allocations Memory cache size is large memory-summary debugging profiling cuda tools Memory summary is hard to interpret Memory usage reported doesn't match nvidia-smi Cannot identify what's using memory Memory summary shows allocated, reserved, and active memory Reserved includes cached memory that can be released Active memory is what tensors currently use Snapshot traces allocation history","anchorText":"Memory Summary Tool Usage Memory summary shows large reserved memory Snapshot shows many small allocations Memory cache size is large memory-summary debugging profiling cuda tools Memory summary shows allocated, reserved, and active memory Reserved includes cached memory that can be released Active memory is what tensors currently use Snapshot traces allocation history ","action":"Use torch.cuda.memory_summary() to see memory breakdown","steps":["Use torch.cuda.memory_summary() to see memory breakdown","Use torch.cuda.memory._record_memory_history() for detailed trace","Save snapshot: torch.cuda.memory._snapshot()","Use active memory not reserved for accurate usage","Compare with nvidia-smi for GPU-level view"]},{"slug":"huggingface-tokenizers-error","title":"HuggingFace Tokenizers Rust Error","category":"Environment","text":"HuggingFace Tokenizers Rust Error Environment TokenizersError: Error in tokenization pyo3_runtime.PanicException OSError: Failed to load native tokenization library tokenizers rust native huggingface nlp environment Tokenization fails with Rust panic Tokenizers library throws opaque error Fast tokenizer fails but slow works Rust tokenizers library not properly installed Tokenizers binary incompatible with Python Memory issue in Rust tokenizers Bug in specific tokenizer implementation","anchorText":"HuggingFace Tokenizers Rust Error TokenizersError: Error in tokenization pyo3_runtime.PanicException OSError: Failed to load native tokenization library tokenizers rust native huggingface nlp environment Rust tokenizers library not properly installed Tokenizers binary incompatible with Python Memory issue in Rust tokenizers Bug in specific tokenizer implementation ","action":"Install rust tokenizers: pip install tokenizers","steps":["Install rust tokenizers: pip install tokenizers","Reinstall transformers: pip install --force-reinstall transformers","Use slow tokenizer: use_fast=False","Update tokenizers: pip install -U tokenizers","Check tokenizers library version compatibility"]},{"slug":"torchdata-pipeline-error","title":"TorchData Pipeline Error","category":"Data Pipeline","text":"TorchData Pipeline Error Data Pipeline RuntimeError: DataPipe error DataPipe: cannot iterate over closed iterator Worker process exited unexpectedly torchdata datapipes data-pipeline streaming environment DataPipe fails to iterate Data loading is slow with TorchData DataPipe workers crash DataPipe not properly configured Incompatible DataPipe operations Worker process crashes during data loading Memory issue with large pipeline buffers","anchorText":"TorchData Pipeline Error RuntimeError: DataPipe error DataPipe: cannot iterate over closed iterator Worker process exited unexpectedly torchdata datapipes data-pipeline streaming environment DataPipe not properly configured Incompatible DataPipe operations Worker process crashes during data loading Memory issue with large pipeline buffers ","action":"Check DataPipe documentation for correct usage","steps":["Check DataPipe documentation for correct usage","Test DataPipe pipeline with simple data first","Use torchdata>=0.5.0 for newer features","Configure worker count and buffer sizes","Test with single process first to isolate issues"]},{"slug":"nccl-p2p-disable","title":"NCCL P2P Disabled","category":"Communication","text":"NCCL P2P Disabled Communication NCCL P2P is disabled NCCL using shared memory for intra-node GPU-to-GPU transfer is slow nccl p2p intra-node bandwidth performance Intra-node communication is slow NCCL falls back to shared memory or PCIe Training performance is poor on multi-GPU nodes P2P disabled by environment variable GPU pair doesn't support P2P PCIe topology doesn't allow P2P Driver issue with P2P","anchorText":"NCCL P2P Disabled NCCL P2P is disabled NCCL using shared memory for intra-node GPU-to-GPU transfer is slow nccl p2p intra-node bandwidth performance P2P disabled by environment variable GPU pair doesn't support P2P PCIe topology doesn't allow P2P Driver issue with P2P ","action":"Enable P2P: export NCCL_P2P_LEVEL=SYS","steps":["Enable P2P: export NCCL_P2P_LEVEL=SYS","Check P2P support: nvidia-smi topo -m","Test P2P: python -c 'import torch; print(torch.cuda.can_device_access_peer(0,1))'","Update GPU driver to enable P2P","Use NVLink for P2P when available"]},{"slug":"gradient-accumulation-bn-issue","title":"Gradient Accumulation BatchNorm Issue","category":"Training Stability","text":"Gradient Accumulation BatchNorm Issue Training Stability Model accuracy is lower with gradient accumulation BN momentum is too low for effective accumulation BN stats computed on sub-batches not effective batch batchnorm gradient-accumulation normalization training-stability sync-batchnorm Model doesn't converge with gradient accumulation and BN Validation accuracy is poor BN running stats are wrong with gradient accumulation BN normalizes per sub-batch not per effective batch BN running stats accumulated with smaller effective samples BN behavior changes with effective batch size BN momentum not accounting for accumulation steps","anchorText":"Gradient Accumulation BatchNorm Issue Model accuracy is lower with gradient accumulation BN momentum is too low for effective accumulation BN stats computed on sub-batches not effective batch batchnorm gradient-accumulation normalization training-stability sync-batchnorm BN normalizes per sub-batch not per effective batch BN running stats accumulated with smaller effective samples BN behavior changes with effective batch size BN momentum not accounting for accumulation steps ","action":"Replace BN with GroupNorm or LayerNorm when using gradient accumulation","steps":["Replace BN with GroupNorm or LayerNorm when using gradient accumulation","Use SyncBatchNorm with DDP for proper BN stats","Calculate BN momentum as base_momentum ** accumulation_steps","Test with no accumulation first to verify model works","Use InstanceNorm for very small effective batch sizes"]},{"slug":"hdf5-data-corruption","title":"HDF5 Data Corruption","category":"Data Pipeline","text":"HDF5 Data Corruption Data Pipeline OSError: unable to read file (bad signature) h5py.H5Error: file corrupted RuntimeError: corrupted file hdf5 h5py data-corruption file-format data-pipeline Training crashes with HDF5 error Data loading fails intermittently Model produces garbage output Disk failure during HDF5 write HDF5 file truncated HDF5 metadata corruption Concurrent writes to same file","anchorText":"HDF5 Data Corruption OSError: unable to read file (bad signature) h5py.H5Error: file corrupted RuntimeError: corrupted file hdf5 h5py data-corruption file-format data-pipeline Disk failure during HDF5 write HDF5 file truncated HDF5 metadata corruption Concurrent writes to same file ","action":"Use h5py with write mode that flushes: f.flush() and os.fsync()","steps":["Use h5py with write mode that flushes: f.flush() and os.fsync()","Use HDF5's chunked storage for partial read recovery","Enable h5py error handling: h5py.get_config().track_order=True","Verify file integrity: h5dump file.h5","Use Denpex to detect corrupted HDF5 files in dataset"]},{"slug":"gpu-memory-clock-throttle","title":"GPU Memory Clock Throttle","category":"Hardware","text":"GPU Memory Clock Throttle Hardware nvidia-smi shows memory clock lower than max Performance varies with batch size Memory-bound operations are slow gpu memory-clock throttle performance hardware Training is slower than expected GPU memory clock is lower than rated Memory bandwidth is below maximum GPU memory clock throttles when memory is not fully utilized Idle memory causes clock throttling Power management reduces memory clock","anchorText":"GPU Memory Clock Throttle nvidia-smi shows memory clock lower than max Performance varies with batch size Memory-bound operations are slow gpu memory-clock throttle performance hardware GPU memory clock throttles when memory is not fully utilized Idle memory causes clock throttling Power management reduces memory clock ","action":"Increase batch size to fully utilize memory","steps":["Increase batch size to fully utilize memory","Set GPU to performance mode: nvidia-smi -pm 1","Lock memory clock: nvidia-smi -lmc 5001","Use larger memory access patterns","Monitor memory clock with nvidia-smi"]},{"slug":"checkpoint-version-older","title":"Checkpoint Saved with Older Version","category":"Data Integrity","text":"Checkpoint Saved with Older Version Data Integrity RuntimeError: version_key not recognized Loading checkpoint produces unexpected structure Loss doesn't match expected value after loading checkpoint version older pytorch incompatibility data-integrity Loading checkpoint fails with version error Error indicates checkpoint was saved with older version Production upgraded PyTorch but uses old checkpoints PyTorch checkpoint format changed between versions Model state_dict structure changed Optimizer state format updated Pickle protocol changed","anchorText":"Checkpoint Saved with Older Version RuntimeError: version_key not recognized Loading checkpoint produces unexpected structure Loss doesn't match expected value after loading checkpoint version older pytorch incompatibility data-integrity PyTorch checkpoint format changed between versions Model state_dict structure changed Optimizer state format updated Pickle protocol changed ","action":"Reinstall older PyTorch to match checkpoint version","steps":["Reinstall older PyTorch to match checkpoint version","Use torch.load with map_location and weights_only=True","Convert checkpoint: load with old version, re-save with new","Use safetensors format for forward compatibility","Strip version_key from checkpoint dict"]},{"slug":"nccl-hang-detection","title":"NCCL Hang Detection","category":"Communication","text":"NCCL Hang Detection Communication GPU utilization drops to 0% on some ranks NCCL operation never completes No progress for minutes or hours nccl hang detection monitoring distributed heartbeat Training hangs with no error NCCL operation takes longer than expected No error message before hang NCCL watchdog timeout is 30 minutes default Hangs without errors are hard to detect NCCL_DEBUG=INFO adds overhead No native monitoring in NCCL","anchorText":"NCCL Hang Detection GPU utilization drops to 0% on some ranks NCCL operation never completes No progress for minutes or hours nccl hang detection monitoring distributed heartbeat NCCL watchdog timeout is 30 minutes default Hangs without errors are hard to detect NCCL_DEBUG=INFO adds overhead No native monitoring in NCCL ","action":"Implement heartbeat monitoring: each rank sends periodic heartbeat to a coordinator","steps":["Implement heartbeat monitoring: each rank sends periodic heartbeat to a coordinator","Use NCCL_HEARTBEAT_TIMEOUT_SEC for faster hang detection","Monitor per-rank GPU utilization for stuck ranks","Use Denpex to monitor NCCL operations for hangs","Set NCCL_TIMEOUT=600 for faster timeout in production"]},{"slug":"label-noise-training","title":"Label Noise Training Issue","category":"Training Stability","text":"Label Noise Training Issue Training Stability Training accuracy plateaus at noise level Model predicts majority class for ambiguous samples Loss is higher than expected for clean validation set label-noise weak-supervision label-smoothing robust-learning training-stability Model overfits to incorrect labels Validation accuracy much higher than training accuracy Model learns to predict wrong labels confidently Training data contains mislabeled samples Weak supervision introduces label noise Annotation errors in training set Duplicate samples with different labels","anchorText":"Label Noise Training Issue Training accuracy plateaus at noise level Model predicts majority class for ambiguous samples Loss is higher than expected for clean validation set label-noise weak-supervision label-smoothing robust-learning training-stability Training data contains mislabeled samples Weak supervision introduces label noise Annotation errors in training set Duplicate samples with different labels ","action":"Use label smoothing: CrossEntropyLoss(label_smoothing=0.1)","steps":["Use label smoothing: CrossEntropyLoss(label_smoothing=0.1)","Use noise-robust loss: Generalized Cross Entropy","Clean training data with relabeling","Use bootstrapping to correct noisy labels","Use confident learning to find mislabeled samples"]},{"slug":"arrow-ipc-error","title":"Arrow IPC Format Error","category":"Data Pipeline","text":"Arrow IPC Format Error Data Pipeline pyarrow.lib.ArrowInvalid: OSError: Arrow error: Arrow I/O error: not a valid Arrow file arrow ipc pyarrow data-pipeline huggingface Arrow IPC deserialization fails Data loading fails with Arrow error PyArrow version mismatch causes Arrow errors Arrow file is corrupted PyArrow version mismatch between save and load Schema mismatch between writer and reader Incompatible Arrow IPC format version","anchorText":"Arrow IPC Format Error pyarrow.lib.ArrowInvalid: OSError: Arrow error: Arrow I/O error: not a valid Arrow file arrow ipc pyarrow data-pipeline huggingface Arrow file is corrupted PyArrow version mismatch between save and load Schema mismatch between writer and reader Incompatible Arrow IPC format version ","action":"Update PyArrow: pip install -U pyarrow","steps":["Update PyArrow: pip install -U pyarrow","Match PyArrow version between save and load","Verify schema compatibility before reading","Use IPC streaming format for robust deserialization","Check Arrow file integrity: pyarrow.ipc.open_stream()"]},{"slug":"slurm-cgroup-limit","title":"SLURM Cgroup Limit","category":"Infrastructure","text":"SLURM Cgroup Limit Infrastructure slurmstepd: error: cgroup out of memory Job killed with SIGKILL by cgroup Resource usage reports differ from job-level limits slurm cgroup memory infrastructure oom resource-limit Job is killed by cgroup OOM killer Training runs slower than expected due to throttling Job can't access all requested resources Cgroup memory limit set lower than job memory request Cgroup CPU quota limits job to fewer cores Cgroup device access not configured for GPUs Cgroup OOM killer triggers before system OOM killer","anchorText":"SLURM Cgroup Limit slurmstepd: error: cgroup out of memory Job killed with SIGKILL by cgroup Resource usage reports differ from job-level limits slurm cgroup memory infrastructure oom resource-limit Cgroup memory limit set lower than job memory request Cgroup CPU quota limits job to fewer cores Cgroup device access not configured for GPUs Cgroup OOM killer triggers before system OOM killer ","action":"Set cgroup memory limit to match job memory request: --mem","steps":["Set cgroup memory limit to match job memory request: --mem","Set cgroup CPU quota to match job CPU request: --cpus-per-task","Verify cgroup device access: lscgroup | grep nvidia","Use systemd-run or cgexec for cgroup management","Monitor cgroup usage: systemd-cgtop"]},{"slug":"pytorch-caching-allocator-debug","title":"PyTorch Caching Allocator Debug","category":"Memory","text":"PyTorch Caching Allocator Debug Memory Reserved memory is much higher than allocated Caching allocator holds memory for reuse Empty cache doesn't release all memory caching-allocator debugging memory pytorch tools Memory usage is hard to debug Reserved memory doesn't match allocated memory OOM despite low allocated memory Caching allocator reserves memory blocks for reuse Reserved memory includes cached blocks Empty cache releases unused blocks to CUDA Memory fragmentation prevents contiguous allocation","anchorText":"PyTorch Caching Allocator Debug Reserved memory is much higher than allocated Caching allocator holds memory for reuse Empty cache doesn't release all memory caching-allocator debugging memory pytorch tools Caching allocator reserves memory blocks for reuse Reserved memory includes cached blocks Empty cache releases unused blocks to CUDA Memory fragmentation prevents contiguous allocation ","action":"Use torch.cuda.memory._record_memory_history() for detailed trace","steps":["Use torch.cuda.memory._record_memory_history() for detailed trace","Save snapshot: torch.cuda.memory._snapshot()","Use active_bytes vs reserved_bytes for accurate usage","Use expandable_segments=True to reduce fragmentation","Profile with torch.cuda.memory_summary() periodically"]},{"slug":"python-multiprocessing-fork-issue","title":"Python Multiprocessing Fork Issue","category":"Environment","text":"Python Multiprocessing Fork Issue Environment RuntimeError: Cannot re-initialize CUDA in forked subprocess Deadlock after fork with CUDA Multiprocessing hangs at first iteration python multiprocessing fork cuda environment worker Workers deadlock after fork CUDA error in forked worker Multiprocessing pool hangs Fork inherits CUDA context which causes issues Fork after CUDA initialization causes deadlock Lock from parent not released in child Open file descriptors shared incorrectly","anchorText":"Python Multiprocessing Fork Issue RuntimeError: Cannot re-initialize CUDA in forked subprocess Deadlock after fork with CUDA Multiprocessing hangs at first iteration python multiprocessing fork cuda environment worker Fork inherits CUDA context which causes issues Fork after CUDA initialization causes deadlock Lock from parent not released in child Open file descriptors shared incorrectly ","action":"Use spawn start method: torch.multiprocessing.set_start_method('spawn')","steps":["Use spawn start method: torch.multiprocessing.set_start_method('spawn')","Set CUDA device after fork: torch.cuda.set_device()","Use torch.multiprocessing.spawn for CUDA workers","Avoid creating CUDA tensors before fork","Test multiprocessing with simple workers first"]},{"slug":"spectral-norm-clipping","title":"Spectral Norm Clipping","category":"Training Stability","text":"Spectral Norm Clipping Training Stability GAN loss explodes Spectral norm is too restrictive GAN mode collapse despite spectral norm spectral-norm gan training-stability wgan-gp discriminator GAN training diverges despite gradient clipping Spectral norm constraint not working Discriminator overpowers generator Spectral norm on wrong layer type Spectral norm constraint too tight or too loose Spectral norm incompatible with model architecture GAN-specific tuning required for spectral norm","anchorText":"Spectral Norm Clipping GAN loss explodes Spectral norm is too restrictive GAN mode collapse despite spectral norm spectral-norm gan training-stability wgan-gp discriminator Spectral norm on wrong layer type Spectral norm constraint too tight or too loose Spectral norm incompatible with model architecture GAN-specific tuning required for spectral norm ","action":"Apply spectral norm to all conv and linear layers","steps":["Apply spectral norm to all conv and linear layers","Use power iteration method: nn.utils.spectral_norm","Tune spectral norm coefficient: power_iterations","Combine with gradient penalty for WGAN-GP","Monitor discriminator vs generator loss balance"]},{"slug":"tfrecord-tf-data-error","title":"TFRecord / tf.data Error","category":"Data Pipeline","text":"TFRecord / tf.data Error Data Pipeline DataLossError: corrupted TFRecord file InvalidArgumentError: cannot parse TFRecord tf.data pipeline hangs at first batch tfrecord tf-data tensorflow data-pipeline environment TFRecord parsing fails tf.data pipeline hangs Training stalls at data loading TFRecord file is corrupted or truncated tf.data pipeline not properly configured tf.data worker count conflicts with GPU count Schema mismatch between TFRecord and parser","anchorText":"TFRecord / tf.data Error DataLossError: corrupted TFRecord file InvalidArgumentError: cannot parse TFRecord tf.data pipeline hangs at first batch tfrecord tf-data tensorflow data-pipeline environment TFRecord file is corrupted or truncated tf.data pipeline not properly configured tf.data worker count conflicts with GPU count Schema mismatch between TFRecord and parser ","action":"Verify TFRecord integrity: tf.data.TFRecordDataset().take(1)","steps":["Verify TFRecord integrity: tf.data.TFRecordDataset().take(1)","Use num_parallel_calls for parallel data loading","Set prefetch buffer size: dataset.prefetch(tf.data.AUTOTUNE)","Match schema between TFRecord writer and parser","Use tf.debugging.Assert for runtime validation"]},{"slug":"nccl-wrong-rank","title":"NCCL Wrong Rank Configuration","category":"Communication","text":"NCCL Wrong Rank Configuration Communication Different ranks have different data AllReduce produces wrong gradients Validation accuracy fluctuates wildly nccl rank distributed configuration ddp master-addr NCCL collective returns wrong result Training accuracy is poor with NCCL Data parallel training produces inconsistent results MASTER_ADDR or MASTER_PORT not set correctly Rank assignment doesn't match between nodes Environment variables differ across nodes NCCL rank differs from PyTorch rank","anchorText":"NCCL Wrong Rank Configuration Different ranks have different data AllReduce produces wrong gradients Validation accuracy fluctuates wildly nccl rank distributed configuration ddp master-addr MASTER_ADDR or MASTER_PORT not set correctly Rank assignment doesn't match between nodes Environment variables differ across nodes NCCL rank differs from PyTorch rank ","action":"Verify MASTER_ADDR and MASTER_PORT on all nodes","steps":["Verify MASTER_ADDR and MASTER_PORT on all nodes","Check rank assignment with print(os.environ['RANK'])","Use torchrun with --nproc_per_node and --nnodes","Verify NCCL rank matches PyTorch rank","Test with single node first to verify rank assignment"]},{"slug":"training-checkpoint-corruption-on-write","title":"Training Checkpoint Corruption on Write","category":"Reliability","text":"Training Checkpoint Corruption on Write Reliability Checkpoint file is 0 bytes or truncated UnpicklingError or OSError when loading checkpoint Checkpoint was being written when process was killed checkpoint corruption write atomic reliability storage Checkpoint file is corrupted after save Checkpoint size is smaller than expected Loading checkpoint fails after training run Process killed during checkpoint write (SIGKILL, OOM) Storage failure during write Network interruption when writing to network storage Concurrent writes to same checkpoint file","anchorText":"Training Checkpoint Corruption on Write Checkpoint file is 0 bytes or truncated UnpicklingError or OSError when loading checkpoint Checkpoint was being written when process was killed checkpoint corruption write atomic reliability storage Process killed during checkpoint write (SIGKILL, OOM) Storage failure during write Network interruption when writing to network storage Concurrent writes to same checkpoint file ","action":"Use atomic write: save to .tmp, then os.rename() to final name","steps":["Use atomic write: save to .tmp, then os.rename() to final name","Use checkpoint integrity verification: SHA-256 of every shard","Save to local NVMe first, then async copy to persistent storage","Increase checkpoint frequency to minimize lost progress","Use Denpex to detect corrupted checkpoints before loading"]},{"slug":"cpu-ram-exhaustion-during-training","title":"CPU RAM Exhaustion During Training","category":"Memory","text":"CPU RAM Exhaustion During Training Memory dmesg: Out of memory: Killed process free -h shows swap usage at 100% System is unresponsive during training cpu-ram host-memory oom swap performance reliability Training process is killed with OOM System becomes very slow during training Swap usage reaches maximum Dataset loaded entirely into CPU memory Data preprocessing pipeline uses too much RAM CPU offload for large model exceeds host memory Other processes consume host memory","anchorText":"CPU RAM Exhaustion During Training dmesg: Out of memory: Killed process free -h shows swap usage at 100% System is unresponsive during training cpu-ram host-memory oom swap performance reliability Dataset loaded entirely into CPU memory Data preprocessing pipeline uses too much RAM CPU offload for large model exceeds host memory Other processes consume host memory ","action":"Use streaming data loading instead of loading entire dataset","steps":["Use streaming data loading instead of loading entire dataset","Reduce data preprocessing memory usage","Profile CPU memory usage: htop or free -h","Increase system RAM or add swap","Set DataLoader pin_memory=False to reduce host memory pressure"]},{"slug":"weight-decay-misconfiguration","title":"Weight Decay Misconfiguration","category":"Training Stability","text":"Weight Decay Misconfiguration Training Stability Validation loss diverges from training loss Model doesn't converge with weight decay AdamW works differently from Adam+weight_decay weight-decay adamw regularization fine-tuning training-stability Model overfits despite weight decay Model underfits with weight decay Training loss is unstable with weight decay Weight decay applied to all parameters including LayerNorm/bias Adam with L2 regularization vs AdamW confusion Weight decay rate too high for model size Weight decay not decoupled from loss","anchorText":"Weight Decay Misconfiguration Validation loss diverges from training loss Model doesn't converge with weight decay AdamW works differently from Adam+weight_decay weight-decay adamw regularization fine-tuning training-stability Weight decay applied to all parameters including LayerNorm/bias Adam with L2 regularization vs AdamW confusion Weight decay rate too high for model size Weight decay not decoupled from loss ","action":"Apply weight decay only to weight matrices, not biases or norms: param groups","steps":["Apply weight decay only to weight matrices, not biases or norms: param groups","Use AdamW instead of Adam with L2 regularization for proper decoupling","Tune weight decay rate per layer type","Use separate weight decay for embeddings vs transformer blocks","Monitor validation loss to detect over/underfitting"]},{"slug":"imagenet-preprocessing-mismatch","title":"ImageNet Preprocessing Mismatch","category":"Data Pipeline","text":"ImageNet Preprocessing Mismatch Data Pipeline Mean/std values don't match pretrained model expectations Image resize dimensions don't match model input RGB vs BGR channel order issue Normalization range mismatch imagenet preprocessing normalization transfer-learning data-pipeline Pretrained model has poor accuracy on custom data Validation accuracy is much lower than expected Fine-tuned model is worse than zero-shot Using ImageNet stats [0.485, 0.456, 0.406] / [0.229, 0.224, 0.225] for non-ImageNet data Wrong image size for model architecture Channel order differs from model training Model trained on [0,1] but data normalized to [-1,1]","anchorText":"ImageNet Preprocessing Mismatch Mean/std values don't match pretrained model expectations Image resize dimensions don't match model input RGB vs BGR channel order issue Normalization range mismatch imagenet preprocessing normalization transfer-learning data-pipeline Using ImageNet stats [0.485, 0.456, 0.406] / [0.229, 0.224, 0.225] for non-ImageNet data Wrong image size for model architecture Channel order differs from model training Model trained on [0,1] but data normalized to [-1,1] ","action":"Match preprocessing to model's training: use exact same transforms","steps":["Match preprocessing to model's training: use exact same transforms","For HuggingFace models, use AutoImageProcessor","For torchvision models, use models.EfficientNet_B0_Weights.IMAGENET1K_V1.transforms()","Resize to model-specific dimensions: 224, 256, 384, 512","Verify channel order: most models use RGB"]},{"slug":"kubernetes-gpu-pod-pending","title":"Kubernetes GPU Pod Pending","category":"Infrastructure","text":"Kubernetes GPU Pod Pending Infrastructure 0/N nodes are available: insufficient nvidia.com/gpu Pod has unbound PersistentVolumeClaims Pod scheduling failed due to node affinity kubernetes k8s gpu scheduling pending infrastructure Pod stays in Pending state indefinitely kubectl describe pod shows insufficient resources GPU pods are not scheduled GPU resource requests exceed cluster capacity NVIDIA device plugin not installed or not detecting GPUs Node selectors or taints prevent scheduling GPU time-slicing not configured PersistentVolume not bound","anchorText":"Kubernetes GPU Pod Pending 0/N nodes are available: insufficient nvidia.com/gpu Pod has unbound PersistentVolumeClaims Pod scheduling failed due to node affinity kubernetes k8s gpu scheduling pending infrastructure GPU resource requests exceed cluster capacity NVIDIA device plugin not installed or not detecting GPUs Node selectors or taints prevent scheduling GPU time-slicing not configured PersistentVolume not bound ","action":"Check cluster GPU capacity: kubectl describe nodes | grep nvidia.com/gpu","steps":["Check cluster GPU capacity: kubectl describe nodes | grep nvidia.com/gpu","Verify NVIDIA device plugin is running on all nodes","Set correct nodeSelector and tolerations for GPU nodes","Configure GPU time-slicing or MIG for sharing GPUs","Use kubectl describe pod to see specific scheduling reason"]},{"slug":"gradient-accumulation-memory-spike","title":"Gradient Accumulation Memory Spike","category":"Memory","text":"Gradient Accumulation Memory Spike Memory CUDA OOM with gradient_accumulation_steps > 1 Memory grows linearly with accumulation steps OOM after several accumulation steps gradient-accumulation memory ddp no-sync training OOM during gradient accumulation Memory grows with accumulation steps Training crashes mid-accumulation Loss not reduced across accumulation steps causing graph retention Computation graph retained for full accumulation window FP16 scaling issues during accumulation Optimizer state grows with effective batch size","anchorText":"Gradient Accumulation Memory Spike CUDA OOM with gradient_accumulation_steps > 1 Memory grows linearly with accumulation steps OOM after several accumulation steps gradient-accumulation memory ddp no-sync training Loss not reduced across accumulation steps causing graph retention Computation graph retained for full accumulation window FP16 scaling issues during accumulation Optimizer state grows with effective batch size ","action":"Use no_sync() context for non-sync accumulation steps","steps":["Use no_sync() context for non-sync accumulation steps","Detach loss or use scaler.scale(loss / accum_steps)","Clear CUDA cache between accumulation steps if needed","Use gradient checkpointing with accumulation","Profile peak memory during accumulation"]},{"slug":"python-version-mismatch","title":"Python Version Mismatch","category":"Environment","text":"Python Version Mismatch Environment TypeError: 'type' object is not subscriptable (Py 3.9 vs 3.10+) match statement syntax errors on older Python Library requires Python 3.10+ but installed is 3.8 python version environment compatibility type-hints Code works on one machine but fails on another Library imports fail with cryptic errors Type hints or syntax errors appear at runtime Developer uses Python 3.11, production has 3.9 Type union syntax (X | Y) requires 3.10+ Library not available for old Python Docker base image uses different Python than host","anchorText":"Python Version Mismatch TypeError: 'type' object is not subscriptable (Py 3.9 vs 3.10+) match statement syntax errors on older Python Library requires Python 3.10+ but installed is 3.8 python version environment compatibility type-hints Developer uses Python 3.11, production has 3.9 Type union syntax (X | Y) requires 3.10+ Library not available for old Python Docker base image uses different Python than host ","action":"Match Python version across dev, training, and deployment: pyenv, conda","steps":["Match Python version across dev, training, and deployment: pyenv, conda","Use type union syntax compatible with oldest supported Python","Check library Python version requirements before installing","Pin Python version in Docker: FROM python:3.10-slim","Test in clean environment before deploying"]},{"slug":"ema-decay-misconfiguration","title":"EMA Decay Misconfiguration","category":"Training Stability","text":"EMA Decay Misconfiguration Training Stability Validation accuracy is lower for EMA than online model EMA model loss is higher EMA decay not updating properly ema model-averaging stability swa training EMA model is worse than training model EMA model has inconsistent behavior EMA model not improving during training EMA decay rate too high (no averaging) or too low (too slow update) EMA buffer on CPU but model on GPU EMA update during gradient accumulation EMA model not in eval mode for validation","anchorText":"EMA Decay Misconfiguration Validation accuracy is lower for EMA than online model EMA model loss is higher EMA decay not updating properly ema model-averaging stability swa training EMA decay rate too high (no averaging) or too low (too slow update) EMA buffer on CPU but model on GPU EMA update during gradient accumulation EMA model not in eval mode for validation ","action":"Set decay to 0.999-0.9999 for stable EMA","steps":["Set decay to 0.999-0.9999 for stable EMA","Move EMA buffer to same device as model: ema.to(device)","Apply EMA after optimizer step, not during accumulation","Use model_ema.eval() for validation","Use copy_() or torch.optim.swa_utils.AveragedModel"]},{"slug":"webdataset-tar-corruption","title":"WebDataset TAR Corruption","category":"Data Pipeline","text":"WebDataset TAR Corruption Data Pipeline tarfile.ReadError: unexpected end of data Connection reset by peer during download Truncated TAR file at end webdataset tar data-pipeline corruption streaming WebDataset loader fails mid-training TAR file is corrupted or truncated Some samples are missing from WebDataset TAR file download interrupted or corrupted TAR file format not compatible with WebDataset TAR file exceeds maximum size limit Missing required extensions like .tar","anchorText":"WebDataset TAR Corruption tarfile.ReadError: unexpected end of data Connection reset by peer during download Truncated TAR file at end webdataset tar data-pipeline corruption streaming TAR file download interrupted or corrupted TAR file format not compatible with WebDataset TAR file exceeds maximum size limit Missing required extensions like .tar ","action":"Verify TAR integrity after download: tar -tf file.tar | wc -l","steps":["Verify TAR integrity after download: tar -tf file.tar | wc -l","Use MD5 or SHA256 checksums to verify TAR files","Re-download corrupted TAR files with retry logic","Use streaming TAR for large files: stream=False","Use webdataset.ShardWriter to create valid WebDataset TARs"]},{"slug":"nccl-p2p-issue","title":"NCCL P2P Communication Issue","category":"Communication","text":"NCCL P2P Communication Issue Communication NCCL WARN net send/recv setup P2P transfer falls back to slower path GPU direct access not available nccl p2p nvlink communication pipeline-parallel tensor-parallel P2P transfer between GPUs fails GPU-to-GPU communication uses PCIe instead of NVLink Pipeline parallelism is slow GPUs not connected via NVLink P2P not enabled in NVIDIA driver MIG configuration prevents P2P GPU topology not optimal for P2P","anchorText":"NCCL P2P Communication Issue NCCL WARN net send/recv setup P2P transfer falls back to slower path GPU direct access not available nccl p2p nvlink communication pipeline-parallel tensor-parallel GPUs not connected via NVLink P2P not enabled in NVIDIA driver MIG configuration prevents P2P GPU topology not optimal for P2P ","action":"Verify GPU topology: nvidia-smi topo -m","steps":["Verify GPU topology: nvidia-smi topo -m","Enable P2P: nvidia-cuda-mps-control -d (older drivers) or check CUDA_P2P_LEVEL","Use NCCL_P2P_LEVEL=sys or loc for explicit control","Disable P2P if it causes issues: NCCL_P2P_DISABLE=1","Use NVLink for best P2P performance"]},{"slug":"nfs-mount-failure","title":"NFS Mount Failure","category":"Reliability","text":"NFS Mount Failure Reliability NFS: stale file handle Connection reset by peer NFS server not responding nfs mount stale-handle shared-storage reliability Training hangs at data loading NFS file access is slow or times out Stale file handle errors on checkpoint save NFS server is down or unreachable Network interruption between client and NFS server NFS server overloaded with requests Stale file handle after file deleted on server NFS version mismatch between client and server","anchorText":"NFS Mount Failure NFS: stale file handle Connection reset by peer NFS server not responding nfs mount stale-handle shared-storage reliability NFS server is down or unreachable Network interruption between client and NFS server NFS server overloaded with requests Stale file handle after file deleted on server NFS version mismatch between client and server ","action":"Verify NFS mount: mount | grep nfs","steps":["Verify NFS mount: mount | grep nfs","Use mount options: timeo=300, retrans=3, _netdev","Use local NVMe for checkpoints, copy to NFS async","Monitor NFS latency: nfsstat -c","Use Denpex to detect NFS slowness before training stalls"]},{"slug":"memory-leak-in-dataloader","title":"Memory Leak in DataLoader","category":"Memory","text":"Memory Leak in DataLoader Memory htop shows worker processes growing Memory grows linearly with iterations OOM after long training run dataloader memory-leak worker num-workers memory Memory usage grows with each epoch Worker processes use more memory over time OOM after several epochs of training Custom Dataset holds references to data Worker process memory not released between epochs Global state in worker functions Cuda memory not freed in workers","anchorText":"Memory Leak in DataLoader htop shows worker processes growing Memory grows linearly with iterations OOM after long training run dataloader memory-leak worker num-workers memory Custom Dataset holds references to data Worker process memory not released between epochs Global state in worker functions Cuda memory not freed in workers ","action":"Profile worker memory: psutil or memory_profiler","steps":["Profile worker memory: psutil or memory_profiler","Use persistent_workers=False to recycle workers","Avoid global state in worker init_fn","Move large objects to shared memory: shared_memory","Use Denpex memory monitoring to detect leaks early"]},{"slug":"lookahead-optimizer-issues","title":"Lookahead Optimizer Issues","category":"Training Stability","text":"Lookahead Optimizer Issues Training Stability Loss spikes with Lookahead Slow weight tracker doesn't improve over fast weights Lookahead alpha is too aggressive lookahead optimizer slow-weights k-steps alpha training-stability Lookahead slow weights diverge from fast weights Training is unstable with Lookahead Lookahead model performs worse than inner optimizer Lookahead k too small (no slow weight update) or too large Lookahead alpha too high (1.0) means full override Lookahead incompatible with weight decay scheduling Lookahead applied to wrong parameter groups","anchorText":"Lookahead Optimizer Issues Loss spikes with Lookahead Slow weight tracker doesn't improve over fast weights Lookahead alpha is too aggressive lookahead optimizer slow-weights k-steps alpha training-stability Lookahead k too small (no slow weight update) or too large Lookahead alpha too high (1.0) means full override Lookahead incompatible with weight decay scheduling Lookahead applied to wrong parameter groups ","action":"Set k=5-10 for most tasks","steps":["Set k=5-10 for most tasks","Set alpha=0.5 for stable averaging","Use Lookahead only with optimizers that benefit (SGD, Adam)","Combine Lookahead with warmup for stable training","Track both fast and slow weight metrics"]},{"slug":"augmentations-too-aggressive","title":"Augmentations Too Aggressive","category":"Data Pipeline","text":"Augmentations Too Aggressive Data Pipeline Training loss is much higher than expected Augmented images have unrealistic artifacts Label-augmentation mismatch augmentation albumentations randaugment cutmix mixup data-pipeline Validation accuracy much higher than training Model can't learn core features Augmented data doesn't look like natural images Augmentation strength too high for dataset Color jitter too strong Geometric transforms create unrealistic shapes Cutout/CutMix too aggressive Mixup alpha too high","anchorText":"Augmentations Too Aggressive Training loss is much higher than expected Augmented images have unrealistic artifacts Label-augmentation mismatch augmentation albumentations randaugment cutmix mixup data-pipeline Augmentation strength too high for dataset Color jitter too strong Geometric transforms create unrealistic shapes Cutout/CutMix too aggressive Mixup alpha too high ","action":"Reduce augmentation strength progressively during training","steps":["Reduce augmentation strength progressively during training","Visualize augmented samples: vutils.save_image","Use RandAugment with appropriate N and M magnitudes","Tune CutMix/Mixup alpha (0.2-0.4 typical)","Use domain-specific augmentations"]},{"slug":"dcgm-exporter-error","title":"DCGM Exporter Error","category":"Infrastructure","text":"DCGM Exporter Error Infrastructure Failed to initialize NVML DCGM Exporter: connection refused GPU metrics endpoint not available dcgm exporter monitoring prometheus gpu-metrics infrastructure GPU metrics not appearing in Prometheus DCGM exporter pod crashes Grafana shows no GPU data NVIDIA driver version mismatch with DCGM DCGM not running with proper privileges NVML library not accessible from container GPU device files not mounted into DCGM container","anchorText":"DCGM Exporter Error Failed to initialize NVML DCGM Exporter: connection refused GPU metrics endpoint not available dcgm exporter monitoring prometheus gpu-metrics infrastructure NVIDIA driver version mismatch with DCGM DCGM not running with proper privileges NVML library not accessible from container GPU device files not mounted into DCGM container ","action":"Verify NVIDIA driver and DCGM version compatibility","steps":["Verify NVIDIA driver and DCGM version compatibility","Run DCGM with privileged: true and hostPID: true","Mount /dev/nvidia* and /proc/driver/nvidia into DCGM","Check DCGM logs: kubectl logs -n monitoring dcgm-exporter","Test metrics: curl http://localhost:9400/metrics"]},{"slug":"cuda-graph-memory-trap","title":"CUDA Graph Memory Trap","category":"Memory","text":"CUDA Graph Memory Trap Memory Reserved memory grows during graph capture CUDA graph holds memory that prevents reuse Memory snapshot shows graph allocations persist cuda-graph memory-pool capture memory performance Memory grows with each CUDA graph replay CUDA graph capture fails with OOM Memory not released after deleting graph CUDA graph holds memory pool for replay Graph captures include all intermediate tensors Multiple graphs share memory pool Graph deletion doesn't immediately release memory","anchorText":"CUDA Graph Memory Trap Reserved memory grows during graph capture CUDA graph holds memory that prevents reuse Memory snapshot shows graph allocations persist cuda-graph memory-pool capture memory performance CUDA graph holds memory pool for replay Graph captures include all intermediate tensors Multiple graphs share memory pool Graph deletion doesn't immediately release memory ","action":"Use torch.cuda.graph_pool_handle() for explicit pool management","steps":["Use torch.cuda.graph_pool_handle() for explicit pool management","Capture in dedicated stream: with torch.cuda.stream(stream):","Reset memory between graph captures: torch.cuda.empty_cache()","Profile with torch.cuda.memory._snapshot() during graph use","Avoid CUDA graphs if memory is critical"]},{"slug":"openmpi-mpi-issue","title":"OpenMPI / MPI Issue","category":"Environment","text":"OpenMPI / MPI Issue Environment ORTE_ERROR MPI_Init failed Primary process exited before connection mpi openmpi horovod deepspeed environment distributed Distributed training fails to start with MPI mpirun hangs or errors MPI processes don't connect mpirun hostfile not configured SSH keys not set up for passwordless MPI MPI version mismatch between nodes Firewall blocking MPI ports Different MPI installations on different nodes","anchorText":"OpenMPI / MPI Issue ORTE_ERROR MPI_Init failed Primary process exited before connection mpi openmpi horovod deepspeed environment distributed mpirun hostfile not configured SSH keys not set up for passwordless MPI MPI version mismatch between nodes Firewall blocking MPI ports Different MPI installations on different nodes ","action":"Configure hostfile or --host for mpirun","steps":["Configure hostfile or --host for mpirun","Set up passwordless SSH: ssh-keygen && ssh-copy-id","Match MPI versions across all nodes: mpirun --version","Open MPI ports in firewall: typically 1024-65535","Use mca parameters: -mca btl tcp,self"]},{"slug":"ranger-optimizer-issues","title":"Ranger Optimizer Issues","category":"Training Stability","text":"Ranger Optimizer Issues Training Stability Loss spikes with Ranger Ranger slow weight not improving Ranger warmup doesn't help ranger radam lookahead optimizer training-stability Ranger training is unstable Ranger converges slower than Adam Ranger model performs worse than baseline Ranger warmup too short for large batch Lookahead component misconfigured RAdam Rectified term not working as expected Ranger learning rate too high","anchorText":"Ranger Optimizer Issues Loss spikes with Ranger Ranger slow weight not improving Ranger warmup doesn't help ranger radam lookahead optimizer training-stability Ranger warmup too short for large batch Lookahead component misconfigured RAdam Rectified term not working as expected Ranger learning rate too high ","action":"Use Ranger with warmup: 5-10% of total steps","steps":["Use Ranger with warmup: 5-10% of total steps","Set Lookahead k=6 and alpha=0.5 for Ranger","Start with Adam LR then try Ranger","Use Ranger21 for modern improvements","Compare with simple AdamW for sanity check"]},{"slug":"mmap-file-handle-exhaustion","title":"MMap File Handle Exhaustion","category":"Data Pipeline","text":"MMap File Handle Exhaustion Data Pipeline OSError: [Errno 24] Too many open files ulimit -n shows low value mmap can't open file for data loading mmap ulimit file-handles hdf5 data-pipeline DataLoader fails with mmap error Too many open files for mmap OSError: [Errno 24] Too many open files ulimit -n too low for number of files Each worker holds file handles HDF5 file handles not released Long-running training accumulates handles","anchorText":"MMap File Handle Exhaustion OSError: [Errno 24] Too many open files ulimit -n shows low value mmap can't open file for data loading mmap ulimit file-handles hdf5 data-pipeline ulimit -n too low for number of files Each worker holds file handles HDF5 file handles not released Long-running training accumulates handles ","action":"Increase ulimit: ulimit -n 65536","steps":["Increase ulimit: ulimit -n 65536","Set in /etc/security/limits.conf for persistent change","Close file handles explicitly in __del__ or __exit__","Use fewer workers with larger batches","Use Denpex to detect file handle leaks"]},{"slug":"gloo-backend-issue","title":"Gloo Backend Issue","category":"Communication","text":"Gloo Backend Issue Communication Gloo collective timeout RuntimeError: Gloo does not support GPU tensors ProcessGroupGloo initialization fails gloo nccl backend distributed cpu-training communication Gloo collective hangs or fails Gloo is much slower than NCCL Gloo doesn't support GPU collectives Gloo used as fallback for non-GPU training Gloo version mismatch across nodes Gloo doesn't support all NCCL operations Gloo file system not shared for rendezvous","anchorText":"Gloo Backend Issue Gloo collective timeout RuntimeError: Gloo does not support GPU tensors ProcessGroupGloo initialization fails gloo nccl backend distributed cpu-training communication Gloo used as fallback for non-GPU training Gloo version mismatch across nodes Gloo doesn't support all NCCL operations Gloo file system not shared for rendezvous ","action":"Use NCCL backend for GPU training: torch.distributed.init_process_group(backend='nccl')","steps":["Use NCCL backend for GPU training: torch.distributed.init_process_group(backend='nccl')","For CPU training use Gloo with shared file system","Match Gloo versions across all nodes","Use gloo as fallback when NCCL not available","Increase Gloo timeout: dist.init_process_group(timeout=timedelta(minutes=30))"]},{"slug":"spot-instance-preemption","title":"Spot Instance Preemption","category":"Reliability","text":"Spot Instance Preemption Reliability EC2 Spot Instance interruption notice GCP preemtible instance terminated Azure spot VM deallocated spot-instance preemption cloud cost-optimization reliability Training job disappears mid-run Spot instance is reclaimed by cloud No warning before preemption Spot price exceeds bid AWS capacity rebalancing Instance maintenance event Cloud provider needs capacity","anchorText":"Spot Instance Preemption EC2 Spot Instance interruption notice GCP preemtible instance terminated Azure spot VM deallocated spot-instance preemption cloud cost-optimization reliability Spot price exceeds bid AWS capacity rebalancing Instance maintenance event Cloud provider needs capacity ","action":"Use checkpoint every 200-500 steps to minimize lost work","steps":["Use checkpoint every 200-500 steps to minimize lost work","Subscribe to interruption notice: aws ec2 describe-spot-instance-requests","Use mixed spot+on-demand for critical jobs","Implement graceful shutdown handler in training script","Use Denpex to detect preemption and checkpoint automatically"]},{"slug":"kv-cache-memory-growth","title":"KV Cache Memory Growth","category":"Memory","text":"KV Cache Memory Growth Memory Memory is O(L * H * D) per token for KV cache OOM with long context Memory grows linearly with batch * sequence length kv-cache inference long-context transformer memory Memory grows with sequence length OOM at long context lengths KV cache uses more memory than model weights KV cache stores K and V for all layers and all positions Standard MHA uses 2 * num_layers * hidden_size GQA reduces cache size Sliding window attention limits cache growth","anchorText":"KV Cache Memory Growth Memory is O(L * H * D) per token for KV cache OOM with long context Memory grows linearly with batch * sequence length kv-cache inference long-context transformer memory KV cache stores K and V for all layers and all positions Standard MHA uses 2 * num_layers * hidden_size GQA reduces cache size Sliding window attention limits cache growth ","action":"Use Grouped Query Attention: reduce K/V heads","steps":["Use Grouped Query Attention: reduce K/V heads","Use PagedAttention (vLLM) for efficient KV cache","Use Sliding Window Attention (Mistral) for long context","Use FlashAttention to avoid materializing attention","Profile KV cache size: cache_size = 2 * batch * seq_len * num_layers * num_heads * head_dim"]},{"slug":"adafactor-issues","title":"Adafactor Optimizer Issues","category":"Training Stability","text":"Adafactor Optimizer Issues Training Stability Loss spikes with Adafactor Adafactor uses more memory than expected Relative step parameter is wrong adafactor optimizer transformer t5 memory-efficient training-stability Adafactor converges slower than Adam Adafactor loss is unstable Adafactor model performs worse than Adam Adafactor epsilon too high (1e-30) or too low (1e-3) Adafactor relative_step disabled incorrectly Adafactor warmup is missing Adafactor weight decay parameter is different","anchorText":"Adafactor Optimizer Issues Loss spikes with Adafactor Adafactor uses more memory than expected Relative step parameter is wrong adafactor optimizer transformer t5 memory-efficient training-stability Adafactor epsilon too high (1e-30) or too low (1e-3) Adafactor relative_step disabled incorrectly Adafactor warmup is missing Adafactor weight decay parameter is different ","action":"Use Adafactor with default relative_step=True for best results","steps":["Use Adafactor with default relative_step=True for best results","Set scale_parameter=False for fine-tuning","Match learning rate: Adafactor needs different LR than Adam","Add warmup for Adafactor: 1000-10000 steps","Use beta2 tuning if relative_step=False"]},{"slug":"shuffle-buffer-too-small","title":"Shuffle Buffer Too Small","category":"Data Pipeline","text":"Shuffle Buffer Too Small Data Pipeline Shuffle buffer smaller than dataset Samples are seen in similar order across epochs Order matters for model performance shuffle data-pipeline buffer order training-stability Model overfits despite shuffle Validation accuracy is much lower than training Loss is unstable across epochs Shuffle buffer too small to randomize data order WebDataset uses fixed shard order DataLoader default doesn't shuffle iterable datasets Shuffle after bucketing breaks order","anchorText":"Shuffle Buffer Too Small Shuffle buffer smaller than dataset Samples are seen in similar order across epochs Order matters for model performance shuffle data-pipeline buffer order training-stability Shuffle buffer too small to randomize data order WebDataset uses fixed shard order DataLoader default doesn't shuffle iterable datasets Shuffle after bucketing breaks order ","action":"Set shuffle buffer to at least 10% of dataset size","steps":["Set shuffle buffer to at least 10% of dataset size","Use 100% of dataset for full shuffle if memory allows","Shuffle at file/shard level for large datasets","Verify shuffle order with sample inspection","Use PyTorch DataLoader shuffle=True for map-style datasets"]},{"slug":"nvidia-smi-missing","title":"nvidia-smi Missing or Broken","category":"Infrastructure","text":"nvidia-smi Missing or Broken Infrastructure NVIDIA-SMI has failed because it couldn't communicate with the NVIDIA driver nvidia-smi: command not found CUDA driver version is insufficient for CUDA runtime version nvidia-smi driver cuda container gpu-detection infrastructure nvidia-smi: command not found GPU not visible to PyTorch No GPUs detected by training script NVIDIA driver not installed Driver version older than CUDA requires Container started without --gpus flag NVIDIA Container Toolkit not installed","anchorText":"nvidia-smi Missing or Broken NVIDIA-SMI has failed because it couldn't communicate with the NVIDIA driver nvidia-smi: command not found CUDA driver version is insufficient for CUDA runtime version nvidia-smi driver cuda container gpu-detection infrastructure NVIDIA driver not installed Driver version older than CUDA requires Container started without --gpus flag NVIDIA Container Toolkit not installed ","action":"Install NVIDIA driver: apt install nvidia-driver-535 (or appropriate version)","steps":["Install NVIDIA driver: apt install nvidia-driver-535 (or appropriate version)","Match driver to CUDA: CUDA 12.x needs driver >= 525","Use --gpus all for docker: docker run --gpus all image","Install NVIDIA Container Toolkit on host","Verify with nvidia-smi after install"]},{"slug":"inference-memory-leak","title":"Inference Memory Leak","category":"Memory","text":"Inference Memory Leak Memory torch.cuda.memory_allocated() grows over time GPU memory snapshot shows accumulation Inference server crashes after hours inference memory-leak serving llm production memory GPU memory grows during inference Server OOMs after many requests Memory not released between requests Tensors held by reference in request handlers KV cache not freed between requests CUDA graphs retain memory Model in eval mode still uses training memory","anchorText":"Inference Memory Leak torch.cuda.memory_allocated() grows over time GPU memory snapshot shows accumulation Inference server crashes after hours inference memory-leak serving llm production memory Tensors held by reference in request handlers KV cache not freed between requests CUDA graphs retain memory Model in eval mode still uses training memory ","action":"Use torch.no_grad() during inference to avoid autograd memory","steps":["Use torch.no_grad() during inference to avoid autograd memory","Explicitly delete intermediate tensors: del tensor; torch.cuda.empty_cache()","Use request-scoped CUDA streams","Profile with torch.cuda.memory._snapshot()","Use vLLM or TGI for production serving"]},{"slug":"huggingface-hub-error","title":"HuggingFace Hub Error","category":"Environment","text":"HuggingFace Hub Error Environment HTTPError: 401 Client Error: Unauthorized gaierror: [Errno -3] Temporary failure in name resolution 429 Client Error: Too Many Requests huggingface hub model-download dataset-download auth environment Model download fails Dataset download fails HF Hub connection refused HF token not set or expired Network proxy blocking HF Hub Rate limit exceeded for unauthenticated requests HuggingFace Hub server outage Repository is private without proper token","anchorText":"HuggingFace Hub Error HTTPError: 401 Client Error: Unauthorized gaierror: [Errno -3] Temporary failure in name resolution 429 Client Error: Too Many Requests huggingface hub model-download dataset-download auth environment HF token not set or expired Network proxy blocking HF Hub Rate limit exceeded for unauthenticated requests HuggingFace Hub server outage Repository is private without proper token ","action":"Set HF token: huggingface-cli login or export HF_TOKEN=hf_xxx","steps":["Set HF token: huggingface-cli login or export HF_TOKEN=hf_xxx","Check network: curl https://huggingface.co","Use mirror: export HF_ENDPOINT=https://hf-mirror.com","Implement retry with exponential backoff","Cache models locally to avoid re-downloads"]},{"slug":"sam-optimizer-issues","title":"SAM (Sharpness-Aware Minimization) Optimizer Issues","category":"Training Stability","text":"SAM (Sharpness-Aware Minimization) Optimizer Issues Training Stability Loss spikes with SAM SAM model underperforms base optimizer SAM rho too high causes divergence sam sharpness-aware optimizer generalization training-stability SAM training is 2x slower SAM rho is not improving generalization SAM gradient conflict with base optimizer SAM rho too high (0.1+) or too low (0.01) SAM learning rate needs to be 2x higher SAM incompatible with gradient accumulation without modification SAM rho not synchronized across DDP ranks","anchorText":"SAM (Sharpness-Aware Minimization) Optimizer Issues Loss spikes with SAM SAM model underperforms base optimizer SAM rho too high causes divergence sam sharpness-aware optimizer generalization training-stability SAM rho too high (0.1+) or too low (0.01) SAM learning rate needs to be 2x higher SAM incompatible with gradient accumulation without modification SAM rho not synchronized across DDP ranks ","action":"Set rho=0.05-0.5 for typical vision tasks","steps":["Set rho=0.05-0.5 for typical vision tasks","Increase LR by 2x when using SAM (e.g., 0.2 -> 0.4)","For DDP, sync rho across ranks: all_reduce(grad)","Use ASAM (Adaptive SAM) for better robustness","Combine with SWA for best results"]},{"slug":"tokenizer-padding-mismatch","title":"Tokenizer Padding Mismatch","category":"Data Pipeline","text":"Tokenizer Padding Mismatch Data Pipeline Padding side 'right' for training but 'left' for generation Tokenizer pad_token = 0 collides with real token 0 Attention mask is wrong because of padding tokenizer padding attention-mask generation data-pipeline Model performance is asymmetric for left/right context Tokenizer produces different sequences for same text Padding token ID is wrong Decoder models need left-padding for batched generation Pad token ID set to 0 collides with vocabulary Attention mask missing for padded positions Tokenizer not setting pad_token when missing","anchorText":"Tokenizer Padding Mismatch Padding side 'right' for training but 'left' for generation Tokenizer pad_token = 0 collides with real token 0 Attention mask is wrong because of padding tokenizer padding attention-mask generation data-pipeline Decoder models need left-padding for batched generation Pad token ID set to 0 collides with vocabulary Attention mask missing for padded positions Tokenizer not setting pad_token when missing ","action":"Set padding_side='right' for training, 'left' for generation","steps":["Set padding_side='right' for training, 'left' for generation","Set tokenizer.pad_token = tokenizer.eos_token if missing","Verify attention_mask is used in model forward","Pad to multiple of 8 for tensor cores","Use DataCollatorForLanguageModeling or similar"]},{"slug":"tcp-port-exhaustion","title":"TCP Port Exhaustion","category":"Communication","text":"TCP Port Exhaustion Communication OSError: [Errno 99] Cannot assign requested address NCCL WARN connect: Connection refused bind: Address already in use tcp port exhaustion network distributed communication Distributed training fails to bind ports Connection refused errors at scale NCCL fails to establish sockets Ephemeral port range exhausted Too many TIME_WAIT sockets NCCL doesn't reuse connections Container network port range too small","anchorText":"TCP Port Exhaustion OSError: [Errno 99] Cannot assign requested address NCCL WARN connect: Connection refused bind: Address already in use tcp port exhaustion network distributed communication Ephemeral port range exhausted Too many TIME_WAIT sockets NCCL doesn't reuse connections Container network port range too small ","action":"Increase ephemeral port range: sysctl net.ipv4.ip_local_port_range","steps":["Increase ephemeral port range: sysctl net.ipv4.ip_local_port_range","Reduce TIME_WAIT: net.ipv4.tcp_fin_timeout=15","Enable TCP_NODELAY: NCCL_SOCKET_NODELAY=1","Use NCCL with proper socket reuse","Monitor socket count: ss -s"]},{"slug":"graceful-shutdown-missing","title":"Graceful Shutdown Missing","category":"Reliability","text":"Graceful Shutdown Missing Reliability Job killed at checkpoint boundary Metrics not flushed to backend No checkpoint on signal graceful-shutdown sigterm checkpoint reliability spot-instance Training job is killed without saving checkpoint Final metrics not logged SIGTERM causes immediate exit SIGTERM handler not registered in training loop signal.signal(SIGTERM, handler) not called torch.distributed cleanup not in finally block Wandb/TensorBoard not closed properly","anchorText":"Graceful Shutdown Missing Job killed at checkpoint boundary Metrics not flushed to backend No checkpoint on signal graceful-shutdown sigterm checkpoint reliability spot-instance SIGTERM handler not registered in training loop signal.signal(SIGTERM, handler) not called torch.distributed cleanup not in finally block Wandb/TensorBoard not closed properly ","action":"Register SIGTERM handler: signal.signal(signal.SIGTERM, handler)","steps":["Register SIGTERM handler: signal.signal(signal.SIGTERM, handler)","Save checkpoint in handler before exit","Cleanup distributed: dist.destroy_process_group()","Flush metrics: wandb.finish(), writer.close()","Use checkpoint-at-end policy in orchestrator"]},{"slug":"pinned-memory-overuse","title":"Pinned Memory Overuse","category":"Memory","text":"Pinned Memory Overuse Memory free -h shows low available memory Host OOM during DataLoader setup DataLoader with pin_memory is slower than without pinned-memory page-locked dataloader cpu-ram host-memory System RAM is exhausted pin_memory=True causes OOM GPU transfers are slow despite pin_memory=True pin_memory=True reserves host RAM for each worker Each worker uses page-locked memory equal to batch size Many workers multiply pinned memory usage Pinned memory not released when DataLoader destroyed","anchorText":"Pinned Memory Overuse free -h shows low available memory Host OOM during DataLoader setup DataLoader with pin_memory is slower than without pinned-memory page-locked dataloader cpu-ram host-memory pin_memory=True reserves host RAM for each worker Each worker uses page-locked memory equal to batch size Many workers multiply pinned memory usage Pinned memory not released when DataLoader destroyed ","action":"Set pin_memory=False if host RAM is limited","steps":["Set pin_memory=False if host RAM is limited","Reduce num_workers when using pin_memory","Profile: pinned_bytes = num_workers * batch_size * sample_bytes","Use Denpex to monitor pinned memory during training","Use cudaHostRegister for selective pinning"]},{"slug":"lion-optimizer-issues","title":"Lion Optimizer Issues","category":"Training Stability","text":"Lion Optimizer Issues Training Stability Lion loss spikes or diverges Lion training is unstable with default LR Lion needs much lower LR than Adam lion optimizer memory-efficient sign-update training-stability Lion loss is unstable Lion converges much slower than Adam Lion model underperforms Adam Lion learning rate too high (default Adam LR is 3-10x too high) Lion weight decay is interpreted differently Lion momentum (beta1) default 0.95 vs Adam 0.9 Lion update sign() operation can be too aggressive","anchorText":"Lion Optimizer Issues Lion loss spikes or diverges Lion training is unstable with default LR Lion needs much lower LR than Adam lion optimizer memory-efficient sign-update training-stability Lion learning rate too high (default Adam LR is 3-10x too high) Lion weight decay is interpreted differently Lion momentum (beta1) default 0.95 vs Adam 0.9 Lion update sign() operation can be too aggressive ","action":"Use 3-10x lower LR than Adam (e.g., 3e-4 for Lion vs 1e-3 for Adam)","steps":["Use 3-10x lower LR than Adam (e.g., 3e-4 for Lion vs 1e-3 for Adam)","Set weight_decay=0.1-1.0 (Lion handles WD differently)","Use beta1=0.95, beta2=0.98 (Lion defaults)","Profile updates: Lion is 2x faster but needs tuning","Start with Lion paper hyperparameters"]},{"slug":"audio-sample-rate-mismatch","title":"Audio Sample Rate Mismatch","category":"Data Pipeline","text":"Audio Sample Rate Mismatch Data Pipeline Sample rate doesn't match model expectations Audio is too fast/slow Model output is unintelligible audio sample-rate whisper asr tts data-pipeline Whisper transcription is wrong TTS model produces distorted audio Audio model has poor accuracy Pretrained model expects 16kHz but data is 44.1kHz Audio not resampled to model requirements Different sample rates mixed in dataset Mel spectrogram computed at wrong sample rate","anchorText":"Audio Sample Rate Mismatch Sample rate doesn't match model expectations Audio is too fast/slow Model output is unintelligible audio sample-rate whisper asr tts data-pipeline Pretrained model expects 16kHz but data is 44.1kHz Audio not resampled to model requirements Different sample rates mixed in dataset Mel spectrogram computed at wrong sample rate ","action":"Resample audio to model expected rate: torchaudio.transforms.Resample(44100, 16000)","steps":["Resample audio to model expected rate: torchaudio.transforms.Resample(44100, 16000)","Use librosa: librosa.resample(audio, orig_sr=44100, target_sr=16000)","Verify sample rate: torchaudio.info(file)","Match preprocessing to pretrained model spec","Use soundfile or torchaudio for consistent loading"]},{"slug":"gpu-temperature-throttle","title":"GPU Temperature Throttling","category":"Infrastructure","text":"GPU Temperature Throttling Infrastructure GPU temperature exceeds 85C GPU clocks are throttled XID 62 thermal slowdown messages temperature thermal throttling cooling gpu infrastructure Training is slower than expected nvidia-smi shows lower clock than base GPU performance degrades over time Insufficient cooling in server room GPU fans not working at full speed Thermal paste degraded Blocked airflow due to dust Ambient temperature too high","anchorText":"GPU Temperature Throttling GPU temperature exceeds 85C GPU clocks are throttled XID 62 thermal slowdown messages temperature thermal throttling cooling gpu infrastructure Insufficient cooling in server room GPU fans not working at full speed Thermal paste degraded Blocked airflow due to dust Ambient temperature too high ","action":"Monitor GPU temperature: nvidia-smi --query-gpu=temperature.gpu","steps":["Monitor GPU temperature: nvidia-smi --query-gpu=temperature.gpu","Set persistence mode: nvidia-smi -pm 1","Set fan speed to maximum: nvidia-smi -lgc (locked clocks)","Check data center cooling and airflow","Replace thermal paste if GPUs are old"]},{"slug":"activation-distillation-memory","title":"Activation Distillation Memory","category":"Memory","text":"Activation Distillation Memory Memory CUDA OOM with teacher and student in memory Forward pass through teacher OOMs Hidden state matching requires extra memory distillation knowledge-distillation teacher-student memory training OOM during knowledge distillation Memory grows with teacher model size Distillation training crashes with large teacher Both teacher and student in GPU memory Teacher gradients not needed but graph is kept Hidden states for all layers stored for matching Batch size doubled effectively for distillation","anchorText":"Activation Distillation Memory CUDA OOM with teacher and student in memory Forward pass through teacher OOMs Hidden state matching requires extra memory distillation knowledge-distillation teacher-student memory training Both teacher and student in GPU memory Teacher gradients not needed but graph is kept Hidden states for all layers stored for matching Batch size doubled effectively for distillation ","action":"Detach teacher forward: with torch.no_grad(): teacher_out = teacher(x)","steps":["Detach teacher forward: with torch.no_grad(): teacher_out = teacher(x)","Use smaller teacher if memory is tight","Use projection layers to match student hidden size","Process distillation in chunks","Use FP16/BF16 for teacher forward pass"]},{"slug":"ssh-key-issue","title":"SSH Key Authentication Issue","category":"Environment","text":"SSH Key Authentication Issue Environment Permission denied (publickey) Host key verification failed Connection closed by remote host ssh authentication key environment distributed Distributed training fails at SSH step ssh: Permission denied (publickey) mpirun or torchrun can't connect to nodes SSH key not in authorized_keys Wrong SSH key permissions (must be 600) Host key not in known_hosts SSH agent not running or key not loaded Different SSH key for different hosts","anchorText":"SSH Key Authentication Issue Permission denied (publickey) Host key verification failed Connection closed by remote host ssh authentication key environment distributed SSH key not in authorized_keys Wrong SSH key permissions (must be 600) Host key not in known_hosts SSH agent not running or key not loaded Different SSH key for different hosts ","action":"Generate SSH key: ssh-keygen -t ed25519","steps":["Generate SSH key: ssh-keygen -t ed25519","Copy to remote: ssh-copy-id user@host","Set permissions: chmod 600 ~/.ssh/id_ed25519","Add to ssh-agent: eval $(ssh-agent) && ssh-add","Disable strict host checking for cluster: StrictHostKeyChecking=no in ~/.ssh/config"]},{"slug":"amp-bf16-vs-fp16","title":"AMP BF16 vs FP16 Confusion","category":"Training Stability","text":"AMP BF16 vs FP16 Confusion Training Stability FP16 NaN with certain operations BF16 model is less accurate than FP32 Mixed precision doesn't speed up training bf16 fp16 mixed-precision gradscaler training-stability Loss is unstable with FP16 FP16 overflow causes NaN BF16 uses more memory than expected FP16 has limited exponent range causing overflow BF16 has same exponent as FP32 but less precision FP16 needs loss scaling, BF16 doesn't BF16 is preferred for H100/A100","anchorText":"AMP BF16 vs FP16 Confusion FP16 NaN with certain operations BF16 model is less accurate than FP32 Mixed precision doesn't speed up training bf16 fp16 mixed-precision gradscaler training-stability FP16 has limited exponent range causing overflow BF16 has same exponent as FP32 but less precision FP16 needs loss scaling, BF16 doesn't BF16 is preferred for H100/A100 ","action":"Use BF16 on H100, A100, and modern GPUs (Ampere+)","steps":["Use BF16 on H100, A100, and modern GPUs (Ampere+)","Use FP16 with GradScaler on older GPUs (Volta, Turing)","Test both and compare convergence","Use BF16 for transformer training","Use FP16 with GradScaler for CNNs"]},{"slug":"parquet-schema-mismatch","title":"Parquet Schema Mismatch","category":"Data Pipeline","text":"Parquet Schema Mismatch Data Pipeline pyarrow.lib.ArrowTypeError: Schema mismatch Expected int64 but got string Cannot read parquet with PyArrow parquet schema pyarrow data-pipeline dataset Parquet read fails with schema error PyArrow SchemaError Column type doesn't match expected Parquet schema differs between files New columns added in some files Type coercion fails Schema evolution not handled","anchorText":"Parquet Schema Mismatch pyarrow.lib.ArrowTypeError: Schema mismatch Expected int64 but got string Cannot read parquet with PyArrow parquet schema pyarrow data-pipeline dataset Parquet schema differs between files New columns added in some files Type coercion fails Schema evolution not handled ","action":"Verify schema: pyarrow.parquet.read_schema(file)","steps":["Verify schema: pyarrow.parquet.read_schema(file)","Use pyarrow.dataset for schema-flexible reading","Cast types explicitly: df.cast([...])","Use ParquetDataset for heterogeneous schemas","Use Denpex to validate dataset schema before training"]},{"slug":"torchelastic-error","title":"TorchElastic Error","category":"Communication","text":"TorchElastic Error Communication Rendezvous timeout Worker group max size reached RendezvousError: Failed to start rendezvous torchelastic elastic rendezvous etcd communication distributed TorchElastic rendezvous fails Workers can't join elastic job Training crashes when nodes are added/removed etcd/Redis rendezvous backend not reachable MIN_SIZE > available nodes MAX_SIZE less than MIN_SIZE Workers don't have same rendezvous config Network partition during rendezvous","anchorText":"TorchElastic Error Rendezvous timeout Worker group max size reached RendezvousError: Failed to start rendezvous torchelastic elastic rendezvous etcd communication distributed etcd/Redis rendezvous backend not reachable MIN_SIZE > available nodes MAX_SIZE less than MIN_SIZE Workers don't have same rendezvous config Network partition during rendezvous ","action":"Verify etcd/Redis endpoint: etcdctl endpoint health","steps":["Verify etcd/Redis endpoint: etcdctl endpoint health","Set MIN_SIZE <= MAX_SIZE","Match rendezvous backend across workers","Use --rdzv_backend=c10d for local testing","Test with nproc_per_node first"]},{"slug":"disaster-recovery-missing","title":"Disaster Recovery Missing","category":"Reliability","text":"Disaster Recovery Missing Reliability Storage corruption with no backup Data center outage with no failover Lost progress after multi-day training disaster-recovery backup checkpoint reliability storage Entire training run is lost Cannot recover from hardware failure No backup of training artifacts No backup of training data No offsite backup of checkpoints Single point of failure in storage No runbook for hardware failure","anchorText":"Disaster Recovery Missing Storage corruption with no backup Data center outage with no failover Lost progress after multi-day training disaster-recovery backup checkpoint reliability storage No backup of training data No offsite backup of checkpoints Single point of failure in storage No runbook for hardware failure ","action":"Replicate checkpoints to multiple regions: aws s3 sync --cross-region","steps":["Replicate checkpoints to multiple regions: aws s3 sync --cross-region","Use 3-2-1 backup rule: 3 copies, 2 media, 1 offsite","Use S3/Glacier for long-term storage of training artifacts","Set up monitoring for storage health","Use Denpex to detect failures before data loss"]},{"slug":"flash-attention-memory","title":"Flash Attention Memory","category":"Memory","text":"Flash Attention Memory Memory Memory is still high with Flash Attention FlashAttentionError: head_dim must be <= 128 Flash attention is slower than naive attention flash-attention memory-efficient attention long-context memory Flash Attention memory savings aren't realized Flash Attention fails with specific shapes Flash Attention is slower than expected Flash Attention requires head_dim <= 128 or 256 depending on version Flash Attention requires specific dtypes (fp16/bf16) Flash Attention has minimum sequence length Flash Attention doesn't work with custom masks","anchorText":"Flash Attention Memory Memory is still high with Flash Attention FlashAttentionError: head_dim must be <= 128 Flash attention is slower than naive attention flash-attention memory-efficient attention long-context memory Flash Attention requires head_dim <= 128 or 256 depending on version Flash Attention requires specific dtypes (fp16/bf16) Flash Attention has minimum sequence length Flash Attention doesn't work with custom masks ","action":"Verify head_dim compatibility: head_dim in {64, 80, 96, 128, 256}","steps":["Verify head_dim compatibility: head_dim in {64, 80, 96, 128, 256}","Use BF16 or FP16 (not FP32)","Pad sequence to multiple of 8 or 16 for Flash","Use flash-attn library, not built-in","For non-supported dims, use xformers or memory_efficient_attention"]},{"slug":"spectral-normalization-collapse","title":"Spectral Normalization Collapse","category":"Training Stability","text":"Spectral Normalization Collapse Training Stability Generator produces same output for all inputs Spectral norm constraint too tight Discriminator overpowering generator spectral-norm gan mode-collapse wgan training-stability GAN training collapses with spectral norm Discriminator loss goes to zero Generator loss diverges with spectral norm Spectral norm applied to wrong layer type Power iteration not converging Spectral norm coefficient too aggressive Generator and discriminator spectral norm mismatch","anchorText":"Spectral Normalization Collapse Generator produces same output for all inputs Spectral norm constraint too tight Discriminator overpowering generator spectral-norm gan mode-collapse wgan training-stability Spectral norm applied to wrong layer type Power iteration not converging Spectral norm coefficient too aggressive Generator and discriminator spectral norm mismatch ","action":"Apply spectral norm to discriminator conv/linear layers only","steps":["Apply spectral norm to discriminator conv/linear layers only","Use power_iterations=1 for speed, more for accuracy","Tune spectral norm coefficient per architecture","Balance G/D with separate learning rates","Monitor G/D loss ratio (1:1 ideal)"]},{"slug":"streaming-data-error","title":"Streaming Data Error","category":"Data Pipeline","text":"Streaming Data Error Data Pipeline botocore.exceptions.EndpointConnectionError Connection reset by peer during data load StreamingDataset: failed to download shard streaming s3 gcs hf-datasets data-pipeline reliability DataLoader hangs during training Streaming download fails intermittently Stale connection to S3/GCS Network interruption during streaming S3 rate limiting GCS authentication expired Streaming dataset cache not configured Large shard size causes memory issues","anchorText":"Streaming Data Error botocore.exceptions.EndpointConnectionError Connection reset by peer during data load StreamingDataset: failed to download shard streaming s3 gcs hf-datasets data-pipeline reliability Network interruption during streaming S3 rate limiting GCS authentication expired Streaming dataset cache not configured Large shard size causes memory issues ","action":"Implement retry with exponential backoff for streaming","steps":["Implement retry with exponential backoff for streaming","Set HF_DATASETS_CACHE for local caching","Use streaming with caching: streaming=True, cache_dir='/tmp/cache'","Pre-download critical datasets","Use Denpex to detect streaming issues before training stalls"]},{"slug":"cluster-shared-storage-slow","title":"Cluster Shared Storage Slow","category":"Infrastructure","text":"Cluster Shared Storage Slow Infrastructure iostat shows high await on storage nfsstat shows slow operations Lustre stripe count not optimal storage nfs lustre gpfs iops infrastructure Data loading is the bottleneck GPUs are starving for data Training is IO-bound Shared storage bandwidth saturated Too many nodes reading same files NFS single server bottleneck Lustre stripe settings not optimized File count too high for metadata server","anchorText":"Cluster Shared Storage Slow iostat shows high await on storage nfsstat shows slow operations Lustre stripe count not optimal storage nfs lustre gpfs iops infrastructure Shared storage bandwidth saturated Too many nodes reading same files NFS single server bottleneck Lustre stripe settings not optimized File count too high for metadata server ","action":"Use local NVMe for hot data, shared storage for cold","steps":["Use local NVMe for hot data, shared storage for cold","Increase Lustre stripe count for large files","Co-locate data with compute (data locality)","Use Denpex to detect IO bottlenecks","Use parallel file system like WekaFS, GPFS, or BeeGFS"]},{"slug":"transformer-cache-memory","title":"Transformer Cache Memory","category":"Memory","text":"Transformer Cache Memory Memory past_key_values length matches input length Memory grows linearly with generated tokens OOM with 100k+ context kv-cache transformers llm inference memory OOM at inference with long context KV cache uses too much memory transformers.TransformerCache OOM Standard MHA KV cache: 2 * layers * hidden * seq * 2 bytes (FP16) Cache not cleared between requests GQA/MQA not used Sliding window not applied","anchorText":"Transformer Cache Memory past_key_values length matches input length Memory grows linearly with generated tokens OOM with 100k+ context kv-cache transformers llm inference memory Standard MHA KV cache: 2 * layers * hidden * seq * 2 bytes (FP16) Cache not cleared between requests GQA/MQA not used Sliding window not applied ","action":"Use GQA: Llama-2 70B uses 8 KV heads vs 64 attention heads","steps":["Use GQA: Llama-2 70B uses 8 KV heads vs 64 attention heads","Use PagedAttention: vLLM","Clear cache between requests: past_key_values=None","Use sliding window: Mistral","Use cache_implementation='quantized' for int4 cache"]},{"slug":"conda-environment-conflict","title":"Conda Environment Conflict","category":"Environment","text":"Conda Environment Conflict Environment CondaError: ResolvePackageNotFound ImportError: /lib/libstdc++.so.6: version `GLIBCXX' UnsatisfiableError: The following specifications were found to be incompatible conda environment dependency conflict pip environment Conda install fails with conflict Import error after conda install Package version doesn't match Pip-installed packages conflict with conda Conda channel priority not set libstdc++ version mismatch MKL/OpenMP conflict Python version mismatch in env","anchorText":"Conda Environment Conflict CondaError: ResolvePackageNotFound ImportError: /lib/libstdc++.so.6: version `GLIBCXX' UnsatisfiableError: The following specifications were found to be incompatible conda environment dependency conflict pip environment Pip-installed packages conflict with conda Conda channel priority not set libstdc++ version mismatch MKL/OpenMP conflict Python version mismatch in env ","action":"Create new env for each project: conda create -n proj python=3.10","steps":["Create new env for each project: conda create -n proj python=3.10","Use conda-forge channel: conda install -c conda-forge","Avoid pip in conda env: use pip only for packages not in conda","Use mamba for faster solving","Pin important packages: torch=2.1.0"]},{"slug":"swa-training","title":"SWA (Stochastic Weight Averaging) Training","category":"Training Stability","text":"SWA (Stochastic Weight Averaging) Training Training Stability SWA model validation accuracy is lower than online model SWA BN update is missing SWA averaging too early or too late swa weight-averaging generalization bn-update training-stability SWA model is worse than the original model SWA doesn't improve generalization BN statistics are wrong in SWA model SWA applied too early in training BN running statistics not updated for SWA SWA LR not annealed correctly SWA averaging includes bad checkpoints","anchorText":"SWA (Stochastic Weight Averaging) Training SWA model validation accuracy is lower than online model SWA BN update is missing SWA averaging too early or too late swa weight-averaging generalization bn-update training-stability SWA applied too early in training BN running statistics not updated for SWA SWA LR not annealed correctly SWA averaging includes bad checkpoints ","action":"Start SWA at 75-100% of training","steps":["Start SWA at 75-100% of training","Update BN after SWA: torch.optim.swa_utils.update_bn()","Use cyclic or constant LR for SWA phase","Average over multiple checkpoints (5-10)","Use SWA from torch.optim.swa_utils.AveragedModel"]},{"slug":"video-codec-mismatch","title":"Video Codec Mismatch","category":"Data Pipeline","text":"Video Codec Mismatch Data Pipeline torchvision.io.read_video fails PyAV: Codec not supported Video has zero frames after decode video codec decord ffmpeg data-pipeline Video model has poor accuracy Cannot load video file Decoded frames are corrupted Codec not installed in container (ffmpeg missing) Codec not in PyTorch's supported list Variable frame rate videos Corrupted video file","anchorText":"Video Codec Mismatch torchvision.io.read_video fails PyAV: Codec not supported Video has zero frames after decode video codec decord ffmpeg data-pipeline Codec not installed in container (ffmpeg missing) Codec not in PyTorch's supported list Variable frame rate videos Corrupted video file ","action":"Install ffmpeg: apt install ffmpeg","steps":["Install ffmpeg: apt install ffmpeg","Convert to standard codec: ffmpeg -i input.mp4 -c:v libx264 output.mp4","Use decord for fast video loading: pip install decord","Verify video: ffprobe input.mp4","Use torchvision.io.read_video_from_memory for streaming"]},{"slug":"horovod-setup-error","title":"Horovod Setup Error","category":"Communication","text":"Horovod Setup Error Communication horovodrun fails to launch workers hvd.allreduce hangs HorovodSpark RuntimeError horovod mpi nccl elastic communication distributed Horovod allreduce fails hvd.init() fails Horovod timeline shows no communication NCCL or MPI not installed gloo not installed for Horovod Mismatch between hvd.allreduce and tf gradients Number of workers doesn't match GPU count","anchorText":"Horovod Setup Error horovodrun fails to launch workers hvd.allreduce hangs HorovodSpark RuntimeError horovod mpi nccl elastic communication distributed NCCL or MPI not installed gloo not installed for Horovod Mismatch between hvd.allreduce and tf gradients Number of workers doesn't match GPU count ","action":"Install Horovod: HOROVOD_GPU_OPERATIONS=NCCL pip install horovod","steps":["Install Horovod: HOROVOD_GPU_OPERATIONS=NCCL pip install horovod","Verify hvd.init() runs successfully on all workers","Match number of processes to GPU count","Use Horovod timeline for debugging: hvd.init(logging_level=logging.DEBUG)","Use ElasticHorovod for spot instances"]},{"slug":"network-partition","title":"Network Partition","category":"Reliability","text":"Network Partition Reliability NCCL WARN unhandled cuda error NCCL timeout waiting for operation torch.distributed barrier hangs network partition hang distributed reliability Distributed training hangs without error Some workers don't see all-reduce NCCL operations time out Network interface goes down Switch/router failure Network congestion causes timeouts Subnet issues between nodes Firewall rules changing mid-training","anchorText":"Network Partition NCCL WARN unhandled cuda error NCCL timeout waiting for operation torch.distributed barrier hangs network partition hang distributed reliability Network interface goes down Switch/router failure Network congestion causes timeouts Subnet issues between nodes Firewall rules changing mid-training ","action":"Use NCCL with proper timeouts: dist.init_process_group(timeout=timedelta(minutes=30))","steps":["Use NCCL with proper timeouts: dist.init_process_group(timeout=timedelta(minutes=30))","Set NCCL_SOCKET_NTIMEOUT=30 for faster failure detection","Monitor network with ping/iperf during training","Use Denpex to detect network issues before training","Use NCCL_IB_HCA to select specific InfiniBand devices"]},{"slug":"dataset-cache-memory","title":"Dataset Cache Memory","category":"Memory","text":"Dataset Cache Memory Memory GPU memory or RAM grows with cached transformations WebDataset cache directory is large HF datasets cache is on small disk dataset-cache hf-datasets webdataset cache memory Memory grows with cache size Pre-batched dataset uses too much RAM Augmentation cache causes OOM HF datasets cache on small / partition Cache not pruned Transformations applied in __init__ and stored Augmentation pipeline stored all transformed samples","anchorText":"Dataset Cache Memory GPU memory or RAM grows with cached transformations WebDataset cache directory is large HF datasets cache is on small disk dataset-cache hf-datasets webdataset cache memory HF datasets cache on small / partition Cache not pruned Transformations applied in __init__ and stored Augmentation pipeline stored all transformed samples ","action":"Set HF_DATASETS_CACHE to large disk","steps":["Set HF_DATASETS_CACHE to large disk","Prune cache regularly: datasets.Dataset.cleanup_cache_files()","Apply transforms in __getitem__ not __init__","Use streaming for large datasets","Use Denpex to monitor cache size"]},{"slug":"polyak-averaging-issues","title":"Polyak Averaging Issues","category":"Training Stability","text":"Polyak Averaging Issues Training Stability Polyak averaged model underperforms online model Averaging window too small or too large Polyak average not used in evaluation polyak-averaging target-network rl averaging training-stability Polyak average is worse than final model Polyak average doesn't improve training Polyak average is too slow to update Polyak average update frequency too low Polyak average decay rate wrong Averaged model not in eval mode Polyak average applied to wrong parameters (e.g., BN)","anchorText":"Polyak Averaging Issues Polyak averaged model underperforms online model Averaging window too small or too large Polyak average not used in evaluation polyak-averaging target-network rl averaging training-stability Polyak average update frequency too low Polyak average decay rate wrong Averaged model not in eval mode Polyak average applied to wrong parameters (e.g., BN) ","action":"Update Polyak average every step: avg_param = alpha * avg_param + (1 - alpha) * param","steps":["Update Polyak average every step: avg_param = alpha * avg_param + (1 - alpha) * param","Set alpha=0.999 for slow averaging","Use model.eval() for averaged model","Don't average BN statistics","Use target network in RL: target = tau * online + (1-tau) * target"]},{"slug":"tfrecord-corrupted-shard","title":"TFRecord Corrupted Shard","category":"Data Pipeline","text":"TFRecord Corrupted Shard Data Pipeline DataLossError: corrupted record at byte offset Some samples can't be parsed TFRecord file is truncated tfrecord corruption shard tensorflow data-pipeline TFRecord parsing fails on specific shard Training crashes on certain epoch Data loss error in tf.data TFRecord file not finalized (writer crashed) Disk full during TFRecord write Incomplete example at end of file Different TFRecord versions","anchorText":"TFRecord Corrupted Shard DataLossError: corrupted record at byte offset Some samples can't be parsed TFRecord file is truncated tfrecord corruption shard tensorflow data-pipeline TFRecord file not finalized (writer crashed) Disk full during TFRecord write Incomplete example at end of file Different TFRecord versions ","action":"Verify TFRecord integrity: tf.data.TFRecordDataset(file).take(1)","steps":["Verify TFRecord integrity: tf.data.TFRecordDataset(file).take(1)","Use atomic write with temp file then rename","Re-write corrupted shards","Skip corrupted examples with exception handling","Use TFRecord writer with proper close"]},{"slug":"mig-mps-conflict","title":"MIG and MPS Conflict","category":"Infrastructure","text":"MIG and MPS Conflict Infrastructure MPS not allowed with MIG instances CUDA_MPS_ACTIVE_THREAD_PERCENTAGE not working nvidia-cuda-mps-control fails mig mps gpu-sharing multi-tenant infrastructure MIG and MPS conflict CUDA cannot allocate memory MPS server fails to start with MIG MIG and MPS are mutually exclusive on same GPU MIG requires A100/H100, MPS works on most MIG creates hard partitions, MPS shares MIG configured per-GPU, MPS per-GPU","anchorText":"MIG and MPS Conflict MPS not allowed with MIG instances CUDA_MPS_ACTIVE_THREAD_PERCENTAGE not working nvidia-cuda-mps-control fails mig mps gpu-sharing multi-tenant infrastructure MIG and MPS are mutually exclusive on same GPU MIG requires A100/H100, MPS works on most MIG creates hard partitions, MPS shares MIG configured per-GPU, MPS per-GPU ","action":"Use MIG OR MPS, not both: choose based on workload","steps":["Use MIG OR MPS, not both: choose based on workload","MIG for hard isolation, MPS for fine-grained sharing","Verify MIG mode: nvidia-smi -mig 1","Start MPS: nvidia-cuda-mps-control -d","Use Denpex to monitor per-partition resource usage"]},{"slug":"gradient-checkpointing-tradeoff","title":"Gradient Checkpointing Tradeoff","category":"Memory","text":"Gradient Checkpointing Tradeoff Memory Memory still high after enabling checkpointing Training is much slower with checkpointing Gradient checkpointing not in eval mode gradient-checkpointing activation-memory memory-vs-compute training-stability Memory savings not realized with gradient checkpointing Training is 2x slower with checkpointing Checkpointing not actually enabled Gradient checkpointing not applied to all layers Checkpointing enabled but use_reentrant=True (default) might not work in torch.compile Activation memory not actually freed during backward Checkpointing on small layers adds overhead with no memory benefit","anchorText":"Gradient Checkpointing Tradeoff Memory still high after enabling checkpointing Training is much slower with checkpointing Gradient checkpointing not in eval mode gradient-checkpointing activation-memory memory-vs-compute training-stability Gradient checkpointing not applied to all layers Checkpointing enabled but use_reentrant=True (default) might not work in torch.compile Activation memory not actually freed during backward Checkpointing on small layers adds overhead with no memory benefit ","action":"Apply gradient checkpointing to all transformer blocks","steps":["Apply gradient checkpointing to all transformer blocks","Use use_reentrant=False for torch.compile compatibility","Profile memory: torch.cuda.memory_allocated()","Combine with other techniques: Flash Attention, BF16","Use gradient checkpointing selectively on largest blocks"]},{"slug":"pytorch-cuda-mismatch","title":"PyTorch CUDA Mismatch","category":"Environment","text":"PyTorch CUDA Mismatch Environment torch.cuda.is_available() returns False CUDA driver version is insufficient for CUDA runtime version libcudart.so not found pytorch cuda driver version-mismatch environment PyTorch can't detect CUDA RuntimeError: CUDA not available Torch was not built with CUDA enabled PyTorch built for CUDA 11.8 but driver supports CUDA 12.0+ PyTorch built for CUDA 12.x but only CUDA 11.x driver PyTorch CPU-only install in GPU container CUDA toolkit version different from PyTorch bundled CUDA","anchorText":"PyTorch CUDA Mismatch torch.cuda.is_available() returns False CUDA driver version is insufficient for CUDA runtime version libcudart.so not found pytorch cuda driver version-mismatch environment PyTorch built for CUDA 11.8 but driver supports CUDA 12.0+ PyTorch built for CUDA 12.x but only CUDA 11.x driver PyTorch CPU-only install in GPU container CUDA toolkit version different from PyTorch bundled CUDA ","action":"Install PyTorch matching CUDA: pip install torch --index-url https://download.pytorch.org/whl/cu118","steps":["Install PyTorch matching CUDA: pip install torch --index-url https://download.pytorch.org/whl/cu118","Verify with python -c 'import torch; print(torch.version.cuda, torch.cuda.is_available())'","Use CUDA 11.8 for broad driver compatibility","Use official PyTorch Docker images","Match PyTorch CUDA to driver CUDA"]},{"slug":"curriculum-learning-issues","title":"Curriculum Learning Issues","category":"Training Stability","text":"Curriculum Learning Issues Training Stability Final accuracy with curriculum is lower than baseline Easy examples don't transfer to hard Difficulty measure doesn't correlate with model needs curriculum self-paced difficulty training-stability Curriculum learning doesn't improve over baseline Model performs worse with curriculum Curriculum progression is too aggressive Difficulty measure doesn't reflect model learning Curriculum too easy for too long Curriculum jumps to hard examples too fast Easy examples over-trained, hard examples under-trained","anchorText":"Curriculum Learning Issues Final accuracy with curriculum is lower than baseline Easy examples don't transfer to hard Difficulty measure doesn't correlate with model needs curriculum self-paced difficulty training-stability Difficulty measure doesn't reflect model learning Curriculum too easy for too long Curriculum jumps to hard examples too fast Easy examples over-trained, hard examples under-trained ","action":"Define difficulty based on model loss or uncertainty","steps":["Define difficulty based on model loss or uncertainty","Start with top 20% easiest examples","Increase difficulty gradually over training","Mix in some random examples to prevent overfitting to easy","Use self-paced learning: include hard examples gradually"]},{"slug":"lmdb-corruption","title":"LMDB Corruption","category":"Data Pipeline","text":"LMDB Corruption Data Pipeline lmdb.MapFullError: Environment map_size limit reached lmdb.CorruptedError: mdb_page_get mdb_txn_commit: Invalid argument lmdb corruption map-size data-pipeline storage LMDB read fails LMDB: mdb_env_open failed LMDB transaction fails to commit LMDB map_size too small for data LMDB file corruption from interrupted write Concurrent write to same LMDB LMDB not properly closed","anchorText":"LMDB Corruption lmdb.MapFullError: Environment map_size limit reached lmdb.CorruptedError: mdb_page_get mdb_txn_commit: Invalid argument lmdb corruption map-size data-pipeline storage LMDB map_size too small for data LMDB file corruption from interrupted write Concurrent write to same LMDB LMDB not properly closed ","action":"Increase map_size: env = lmdb.open(path, map_size=int(1e12))","steps":["Increase map_size: env = lmdb.open(path, map_size=int(1e12))","Close LMDB env explicitly: env.close()","Use read-only env for workers: env = lmdb.open(path, readonly=True)","Verify LMDB integrity: env.stat()","Use atomic write: write to .tmp, then rename"]},{"slug":"nccl-ib-hca-mismatch","title":"NCCL IB HCA Mismatch","category":"Communication","text":"NCCL IB HCA Mismatch Communication NCCL WARN NET/IB: no usable sockets NCCL falls back to TCP/IP Inter-node bandwidth much lower than intra-node nccl infiniband hca communication distributed ib NCCL allreduce is slow across nodes NCCL uses wrong IB device Inter-node bandwidth is low Wrong IB device selected by NCCL IB subnet manager not configured RoCE vs IB mode mismatch Multiple IB HCAs with different speeds","anchorText":"NCCL IB HCA Mismatch NCCL WARN NET/IB: no usable sockets NCCL falls back to TCP/IP Inter-node bandwidth much lower than intra-node nccl infiniband hca communication distributed ib Wrong IB device selected by NCCL IB subnet manager not configured RoCE vs IB mode mismatch Multiple IB HCAs with different speeds ","action":"Set NCCL_IB_HCA=mlx5_0,mlx5_1 to specify devices","steps":["Set NCCL_IB_HCA=mlx5_0,mlx5_1 to specify devices","Set NCCL_IB_DISABLE=0 to enable IB","Use NCCL_IB_HCA=<best_device> for selection","Verify IB: ibstat, ibv_devinfo","Use NCCL_DEBUG=INFO to debug IB selection"]},{"slug":"hdf5-corruption","title":"HDF5 Corruption","category":"Reliability","text":"HDF5 Corruption Reliability OSError: Unable to open file (Unable to truncate file) h5py.H5Error HDF5 file header corruption hdf5 corruption h5py storage reliability HDF5 read fails OSError: Unable to open file HDF5 file is corrupted HDF5 file not closed properly Concurrent writes to same file HDF5 library version mismatch File truncation from disk full","anchorText":"HDF5 Corruption OSError: Unable to open file (Unable to truncate file) h5py.H5Error HDF5 file header corruption hdf5 corruption h5py storage reliability HDF5 file not closed properly Concurrent writes to same file HDF5 library version mismatch File truncation from disk full ","action":"Close HDF5 files: f.close() or use context manager","steps":["Close HDF5 files: f.close() or use context manager","Use single-writer, multiple-reader pattern","Match h5py versions: pip install h5py==3.x","Use atomic write: write to .tmp, then rename","Verify integrity: h5py.File(path, 'r', swmr=True)"]},{"slug":"torch-compile-memory","title":"torch.compile Memory","category":"Memory","text":"torch.compile Memory Memory torch.compile causes OOM Memory higher with torch.compile than eager mode torch.compile recompiles on shape change torch-compile dynamo compilation memory performance Memory grows with torch.compile First iterations are slow Memory is fragmented after compile torch.compile keeps compiled graphs in memory Recompilation on dynamic shapes Different configs for different model parts Guard evaluation overhead","anchorText":"torch.compile Memory torch.compile causes OOM Memory higher with torch.compile than eager mode torch.compile recompiles on shape change torch-compile dynamo compilation memory performance torch.compile keeps compiled graphs in memory Recompilation on dynamic shapes Different configs for different model parts Guard evaluation overhead ","action":"Set torch._dynamo.config.cache_size_limit=64","steps":["Set torch._dynamo.config.cache_size_limit=64","Use dynamic=False if shapes are static","Use mark_static for known shapes","Profile with torch.compile verbose mode","Use reduce-overhead mode for inference"]},{"slug":"cyclic-lr-scheduler-issues","title":"Cyclic LR Scheduler Issues","category":"Training Stability","text":"Cyclic LR Scheduler Issues Training Stability Loss spikes at peak LR Cyclic LR step size too small or too large Cyclic LR base_lr is too high cyclic-lr scheduler training-stability hyperparameter Cyclic LR training is unstable Cyclic LR doesn't improve over constant LR Cyclic LR model performs worse Step size too small (oscillates too much) or too large (no cycling) Base LR and max LR inverted Mode not appropriate for task Gamma decay too aggressive for exp_range","anchorText":"Cyclic LR Scheduler Issues Loss spikes at peak LR Cyclic LR step size too small or too large Cyclic LR base_lr is too high cyclic-lr scheduler training-stability hyperparameter Step size too small (oscillates too much) or too large (no cycling) Base LR and max LR inverted Mode not appropriate for task Gamma decay too aggressive for exp_range ","action":"Set step_size = 2-8 * epochs_per_step","steps":["Set step_size = 2-8 * epochs_per_step","Set base_lr = max_lr / 10 for stable cycling","Use triangular2 mode for slower decay","Use exp_range with gamma=0.99994 for smooth decay","Combine with warmup for stability"]},{"slug":"rag-embedding-mismatch","title":"RAG Embedding Mismatch","category":"Data Pipeline","text":"RAG Embedding Mismatch Data Pipeline Retrieved documents don't match query Similarity scores are all low or all high Different embedding model used for indexing vs query rag embedding retrieval vector-search data-pipeline RAG retrieval returns irrelevant documents RAG answers are wrong despite correct docs Embedding distance seems random Embedding model changed after indexing Different embedding model for indexing and query Embedding dimensions don't match Mixed language embeddings Embedding normalization not consistent","anchorText":"RAG Embedding Mismatch Retrieved documents don't match query Similarity scores are all low or all high Different embedding model used for indexing vs query rag embedding retrieval vector-search data-pipeline Embedding model changed after indexing Different embedding model for indexing and query Embedding dimensions don't match Mixed language embeddings Embedding normalization not consistent ","action":"Always use same embedding model for indexing and query","steps":["Always use same embedding model for indexing and query","Pin embedding model version in production","Verify embedding dimensions match: model.get_sentence_embedding_dimension()","Normalize embeddings: util.normalize_embeddings","Test retrieval quality before deploying RAG changes"]},{"slug":"gpu-thermal-design-power","title":"GPU TDP / Power Limit","category":"Infrastructure","text":"GPU TDP / Power Limit Infrastructure nvidia-smi shows power limit reached GPU clocks lower than base Power consumption at 100% of limit tdp power-limit gpu-clocks power infrastructure GPU performance is throttled Training is slower than expected Power consumption is higher than expected Power limit set too low for workload Default power limit too conservative GPU throttling due to power not thermal Inefficient GPU utilization","anchorText":"GPU TDP / Power Limit nvidia-smi shows power limit reached GPU clocks lower than base Power consumption at 100% of limit tdp power-limit gpu-clocks power infrastructure Power limit set too low for workload Default power limit too conservative GPU throttling due to power not thermal Inefficient GPU utilization ","action":"Set power limit: nvidia-smi -pl 300 (set in watts)","steps":["Set power limit: nvidia-smi -pl 300 (set in watts)","Use persistence mode: nvidia-smi -pm 1","Lock clocks for consistent performance: nvidia-smi -lgc","Monitor power: nvidia-smi -q -d POWER","Use Denpex to track power usage during training"]},{"slug":"cudnn-benchmark-memory","title":"cuDNN Benchmark Memory","category":"Memory","text":"cuDNN Benchmark Memory Memory First iterations are slow with cudnn.benchmark=True Memory grows during cuDNN algorithm search cuDNN algorithm not found for specific config cudnn benchmark memory performance determinism cuDNN benchmark causes OOM cuDNN not using optimal algorithm cuDNN determinism warnings cuDNN benchmark tries multiple algorithms using memory Algorithm not available for specific tensor layout Variable input sizes trigger re-benchmark cuDNN doesn't support some custom operations","anchorText":"cuDNN Benchmark Memory First iterations are slow with cudnn.benchmark=True Memory grows during cuDNN algorithm search cuDNN algorithm not found for specific config cudnn benchmark memory performance determinism cuDNN benchmark tries multiple algorithms using memory Algorithm not available for specific tensor layout Variable input sizes trigger re-benchmark cuDNN doesn't support some custom operations ","action":"Set torch.backends.cudnn.benchmark=True for fixed input sizes","steps":["Set torch.backends.cudnn.benchmark=True for fixed input sizes","Set cudnn.benchmark=False for variable input sizes","Set cudnn.deterministic=True for reproducibility","Profile with torch.backends.cudnn.benchmark_limit","Use TF32 for Ampere+ GPUs: torch.backends.cuda.matmul.allow_tf32=True"]},{"slug":"libc-version-mismatch","title":"Libc Version Mismatch","category":"Environment","text":"Libc Version Mismatch Environment version `GLIBC_2.34' not found Incompatible C library version Wheel not compatible with musl libc glibc musl alpine compatibility environment Binary fails to run on different OS GLIBC version not found Alpine Linux doesn't work with glibc-built wheel Binary built on newer glibc, deployed on older Alpine uses musl, most wheels built for glibc CUDA libraries link against system glibc Python wheels not built for musl","anchorText":"Libc Version Mismatch version `GLIBC_2.34' not found Incompatible C library version Wheel not compatible with musl libc glibc musl alpine compatibility environment Binary built on newer glibc, deployed on older Alpine uses musl, most wheels built for glibc CUDA libraries link against system glibc Python wheels not built for musl ","action":"Use glibc-based OS (Ubuntu, Debian, CentOS) for compatibility","steps":["Use glibc-based OS (Ubuntu, Debian, CentOS) for compatibility","Use manylinux wheels for cross-platform","Use Alpine only with --no-binary :all: for pure Python","Match base image OS to development environment","Use static linking for critical binaries"]},{"slug":"one-cycle-policy-issue","title":"One-Cycle Policy Issue","category":"Training Stability","text":"One-Cycle Policy Issue Training Stability Loss spikes at peak LR Training is unstable in second half One-cycle model doesn't converge one-cycle super-convergence scheduler momentum training-stability One-cycle policy causes loss spikes One-cycle model underperforms constant LR One-cycle training is unstable at peak Max LR too high for one-cycle Momentum range inverted Total steps miscalculated Anneal strategy doesn't match scheduler","anchorText":"One-Cycle Policy Issue Loss spikes at peak LR Training is unstable in second half One-cycle model doesn't converge one-cycle super-convergence scheduler momentum training-stability Max LR too high for one-cycle Momentum range inverted Total steps miscalculated Anneal strategy doesn't match scheduler ","action":"Set max_lr using LR finder: 3-5x lower than finder recommendation","steps":["Set max_lr using LR finder: 3-5x lower than finder recommendation","Set momentum range: 0.95 -> 0.85 (annealing)","Use total_steps = epochs * steps_per_epoch","Use pct_start=0.3 for most tasks","Use div_factor=25, final_div_factor=1000"]},{"slug":"multi-label-class-imbalance","title":"Multi-Label Class Imbalance","category":"Data Pipeline","text":"Multi-Label Class Imbalance Data Pipeline Per-label F1 varies widely Rare labels are never predicted Validation loss is high multi-label class-imbalance focal-loss bce data-pipeline Model predicts only frequent labels Rare labels have very low recall Loss is dominated by negative samples Each label has different frequency Negative samples dominate in BCE loss Macro F1 differs from micro F1 Hard negative mining not used","anchorText":"Multi-Label Class Imbalance Per-label F1 varies widely Rare labels are never predicted Validation loss is high multi-label class-imbalance focal-loss bce data-pipeline Each label has different frequency Negative samples dominate in BCE loss Macro F1 differs from micro F1 Hard negative mining not used ","action":"Use class weights: pos_weight in BCEWithLogitsLoss","steps":["Use class weights: pos_weight in BCEWithLogitsLoss","Use focal loss for hard examples: FocalLoss","Use OHEM (online hard example mining)","Use macro F1 instead of accuracy","Use multi-label stratified split: MultilabelStratifiedKFold"]},{"slug":"nccl-ipv6-issue","title":"NCCL IPv6 Issue","category":"Communication","text":"NCCL IPv6 Issue Communication NCCL WARN Net: No IPv6 interface found NCCL slow to initialize Connection refused on IPv6 nccl ipv6 ipv4 network communication distributed NCCL hangs at initialization NCCL uses wrong IP family Inter-node communication fails NCCL defaults to IPv6 but only IPv4 available DNS resolves to IPv6 first Cluster doesn't have IPv6 configured Firewall blocks IPv6 traffic","anchorText":"NCCL IPv6 Issue NCCL WARN Net: No IPv6 interface found NCCL slow to initialize Connection refused on IPv6 nccl ipv6 ipv4 network communication distributed NCCL defaults to IPv6 but only IPv4 available DNS resolves to IPv6 first Cluster doesn't have IPv6 configured Firewall blocks IPv6 traffic ","action":"Force IPv4: NCCL_SOCKET_IFNAME=^lo,docker (and disable IPv6)","steps":["Force IPv4: NCCL_SOCKET_IFNAME=^lo,docker (and disable IPv6)","Set NCCL_IB_DISABLE=1 if not using IB","Use GLOO as fallback: NCCL_SOCKET_NODELAY=1","Verify network: ip addr show","Set NCCL_DEBUG=INFO for debugging"]},{"slug":"oom-killed-mid-step","title":"OOM Killed Mid-Step","category":"Reliability","text":"OOM Killed Mid-Step Reliability dmesg shows OOM killer Exit code 137 (SIGKILL) Process vanishes during training step oom oom-killer memory reliability cgroup Training process is killed with OOM Job disappears with exit code 137 No graceful shutdown on OOM Peak memory exceeds available Memory leak grows until OOM Cgroup limit lower than job memory Other processes consume memory","anchorText":"OOM Killed Mid-Step dmesg shows OOM killer Exit code 137 (SIGKILL) Process vanishes during training step oom oom-killer memory reliability cgroup Peak memory exceeds available Memory leak grows until OOM Cgroup limit lower than job memory Other processes consume memory ","action":"Profile peak memory: torch.cuda.max_memory_allocated()","steps":["Profile peak memory: torch.cuda.max_memory_allocated()","Add @torch.cuda.amp.autocast to reduce memory","Use gradient checkpointing","Monitor with nvidia-smi and free -h","Use Denpex to detect memory issues before OOM"]},{"slug":"tensor-parallel-memory","title":"Tensor Parallel Memory","category":"Memory","text":"Tensor Parallel Memory Memory Memory not evenly distributed across GPUs Forward pass OOMs at gather/scatter GPU 0 has more memory than others tensor-parallel megatron model-parallel memory distributed OOM with tensor parallelism Tensor parallel training is slow Memory imbalance across GPUs Embedding layer not split: rank 0 has all embeddings Cross-entropy loss not properly parallelized Attention gather operations are memory-heavy Some operations require full tensor materialization","anchorText":"Tensor Parallel Memory Memory not evenly distributed across GPUs Forward pass OOMs at gather/scatter GPU 0 has more memory than others tensor-parallel megatron model-parallel memory distributed Embedding layer not split: rank 0 has all embeddings Cross-entropy loss not properly parallelized Attention gather operations are memory-heavy Some operations require full tensor materialization ","action":"Use sequence parallel for activations (Megatron)","steps":["Use sequence parallel for activations (Megatron)","Use ZeRO for optimizer state","Profile with torch.distributed.tensor","Balance load with expert parallelism","Use Flash Attention with TP"]},{"slug":"layer-norm-weight-decay","title":"Layer Norm Weight Decay","category":"Training Stability","text":"Layer Norm Weight Decay Training Stability Validation loss is much higher than expected Loss spikes with weight decay Fine-tuning underperforms training from scratch weight-decay layer-norm adamw fine-tuning training-stability Model doesn't converge with weight decay Validation loss is unstable Weight decay hurts transformer training Weight decay applied to LayerNorm gamma and bias Weight decay applied to embedding layer AdamW with default parameter groups includes all params No normalization in weight decay application","anchorText":"Layer Norm Weight Decay Validation loss is much higher than expected Loss spikes with weight decay Fine-tuning underperforms training from scratch weight-decay layer-norm adamw fine-tuning training-stability Weight decay applied to LayerNorm gamma and bias Weight decay applied to embedding layer AdamW with default parameter groups includes all params No normalization in weight decay application ","action":"Apply weight decay only to weight matrices, not biases/LayerNorm","steps":["Apply weight decay only to weight matrices, not biases/LayerNorm","Use parameter groups: no_decay = ['bias', 'LayerNorm.weight']","Use AdamW with proper param groups","Test with no weight decay first","Use weight_decay=0.1 for transformer training"]},{"slug":"contrastive-learning-augmentation","title":"Contrastive Learning Augmentation","category":"Data Pipeline","text":"Contrastive Learning Augmentation Data Pipeline Contrastive loss plateaus at high value Model can't distinguish similar from dissimilar Augmentations too weak: model sees same image contrastive simclr clip self-supervised data-pipeline SimCLR/CLIP training doesn't improve baseline Augmentations too weak don't create useful pairs Augmentations too strong break semantic content Augmentations too weak: positive pairs are identical Augmentations too strong: positive pairs are semantically different No diverse augmentations: model overfits to color Augmentations differ for image and text in CLIP","anchorText":"Contrastive Learning Augmentation Contrastive loss plateaus at high value Model can't distinguish similar from dissimilar Augmentations too weak: model sees same image contrastive simclr clip self-supervised data-pipeline Augmentations too weak: positive pairs are identical Augmentations too strong: positive pairs are semantically different No diverse augmentations: model overfits to color Augmentations differ for image and text in CLIP ","action":"Use SimCLR augmentations: random crop, color jitter, gaussian blur","steps":["Use SimCLR augmentations: random crop, color jitter, gaussian blur","Use moderate strength: ColorJitter(0.8, 0.8, 0.8, 0.2)","Add Gaussian blur for ImageNet","For CLIP, use independent image and text augmentations","Visualize positive pairs to verify they're related"]},{"slug":"slurm-time-limit","title":"SLURM Time Limit","category":"Infrastructure","text":"SLURM Time Limit Infrastructure slurmstepd: Job ... exceeded its time limit Job cancelled with TIMEOUT reason Training reaches 99% then dies slurm time-limit walltime checkpoint infrastructure Training is killed at time limit SLURM job ends unexpectedly No checkpoint saved before time limit Time limit set too short for training duration No checkpoint at end of training No resume logic for SLURM jobs No walltime warning handling","anchorText":"SLURM Time Limit slurmstepd: Job ... exceeded its time limit Job cancelled with TIMEOUT reason Training reaches 99% then dies slurm time-limit walltime checkpoint infrastructure Time limit set too short for training duration No checkpoint at end of training No resume logic for SLURM jobs No walltime warning handling ","action":"Set time limit accurately: --time=24:00:00 for 24h","steps":["Set time limit accurately: --time=24:00:00 for 24h","Save checkpoint before time limit using --signal","Implement signal handler in training script","Use checkpoint at end: --checkpoint-dir=path","Add 10-20% buffer to time estimate"]},{"slug":"checkpoint-loading-memory","title":"Checkpoint Loading Memory","category":"Memory","text":"Checkpoint Loading Memory Memory CUDA OOM at checkpoint.load_state_dict() Memory doubled during checkpoint loading Loading checkpoint on different model architecture checkpoint loading memory safetensors resume OOM during checkpoint load Memory spikes when loading checkpoint Cannot resume from checkpoint Old model in memory while loading new Pickle deserialization creates copies State dict not directly mapped to model Loading on CPU then transferring to GPU uses both","anchorText":"Checkpoint Loading Memory CUDA OOM at checkpoint.load_state_dict() Memory doubled during checkpoint loading Loading checkpoint on different model architecture checkpoint loading memory safetensors resume Old model in memory while loading new Pickle deserialization creates copies State dict not directly mapped to model Loading on CPU then transferring to GPU uses both ","action":"Delete old model before loading: del old_model; torch.cuda.empty_cache()","steps":["Delete old model before loading: del old_model; torch.cuda.empty_cache()","Load on CPU first, then move: model.load_state_dict(torch.load(path, map_location='cpu'))","Use strict=False for partial loading","Use safetensors for memory-mapped loading","Profile memory during load: torch.cuda.memory_summary()"]},{"slug":"docker-image-mismatch","title":"Docker Image Mismatch","category":"Environment","text":"Docker Image Mismatch Environment Import works locally but not in container CUDA version differs between images Different Python versions in images docker container image environment deployment Code works locally but fails in container Container has different library versions Production deployment fails despite passing tests Dev uses Python 3.11, container has 3.9 Container CUDA version different from dev System libraries missing in slim images Working directory differs between environments File paths differ in containers vs host","anchorText":"Docker Image Mismatch Import works locally but not in container CUDA version differs between images Different Python versions in images docker container image environment deployment Dev uses Python 3.11, container has 3.9 Container CUDA version different from dev System libraries missing in slim images Working directory differs between environments File paths differ in containers vs host ","action":"Use same base image for dev and prod: python:3.10-slim","steps":["Use same base image for dev and prod: python:3.10-slim","Pin all dependencies: requirements.txt with hashes","Match CUDA version: nvidia/cuda:12.1.1-cudnn8-runtime","Test in container before deploying: docker run -it image bash","Use multi-stage builds to keep images small"]},{"slug":"fine-tuning-failure","title":"Fine-Tuning Failure","category":"Training Stability","text":"Fine-Tuning Failure Training Stability Validation accuracy is lower than pretrained zero-shot Model loses language/general capabilities Catastrophic forgetting on original task fine-tuning catastrophic-forgetting lora pretrained training-stability Fine-tuned model is worse than pretrained Fine-tuning causes catastrophic forgetting Model loses general capabilities after fine-tuning Learning rate too high (destroys pretrained features) Too many epochs overfit to small dataset LoRA not used: full fine-tuning expensive No replay of original data Wrong task head: classification vs regression mismatch","anchorText":"Fine-Tuning Failure Validation accuracy is lower than pretrained zero-shot Model loses language/general capabilities Catastrophic forgetting on original task fine-tuning catastrophic-forgetting lora pretrained training-stability Learning rate too high (destroys pretrained features) Too many epochs overfit to small dataset LoRA not used: full fine-tuning expensive No replay of original data Wrong task head: classification vs regression mismatch ","action":"Use lower learning rate: 1e-5 to 5e-5 for LLM fine-tuning","steps":["Use lower learning rate: 1e-5 to 5e-5 for LLM fine-tuning","Use LoRA: peft library for parameter-efficient tuning","Use fewer epochs: 1-3 for most fine-tuning","Mix in some pretraining data for replay","Use proper task head for the task"]},{"slug":"image-resize-artifacts","title":"Image Resize Artifacts","category":"Data Pipeline","text":"Image Resize Artifacts Data Pipeline Image is blurry after resize Image is pixelated after resize Resize is asymmetric (different x and y) image-resize bilinear bicubic lanczos data-pipeline Fine-tuned model has lower accuracy than expected Image quality is poor after resize Resize artifacts visible in samples Wrong resize method (BILINEAR for INCEPTION which expects BICUBIC) Resize doesn't preserve aspect ratio Resize too aggressive (1000x1000 -> 224x224) Resize order wrong: resize then crop vs crop then resize","anchorText":"Image Resize Artifacts Image is blurry after resize Image is pixelated after resize Resize is asymmetric (different x and y) image-resize bilinear bicubic lanczos data-pipeline Wrong resize method (BILINEAR for INCEPTION which expects BICUBIC) Resize doesn't preserve aspect ratio Resize too aggressive (1000x1000 -> 224x224) Resize order wrong: resize then crop vs crop then resize ","action":"Match resize method to pretrained model spec","steps":["Match resize method to pretrained model spec","Preserve aspect ratio: resize + pad or resize + center crop","Use LANCZOS for high-quality downsample","Use BICUBIC for ImageNet pretrained models","Visualize resized images to verify quality"]},{"slug":"pipeline-parallel-bubble","title":"Pipeline Parallel Bubble","category":"Communication","text":"Pipeline Parallel Bubble Communication GPU utilization is low during pipeline training Pipeline warmup and cooldown take significant time Pipeline schedule has large bubbles pipeline-parallel bubble 1f1b megatron communication Pipeline parallel efficiency is low Pipeline parallel is slower than data parallel Many GPUs idle during pipeline training Pipeline bubble at start and end of each batch Number of micro-batches too small Pipeline schedule is GPipe (1F1B is better) Number of stages doesn't match GPUs well","anchorText":"Pipeline Parallel Bubble GPU utilization is low during pipeline training Pipeline warmup and cooldown take significant time Pipeline schedule has large bubbles pipeline-parallel bubble 1f1b megatron communication Pipeline bubble at start and end of each batch Number of micro-batches too small Pipeline schedule is GPipe (1F1B is better) Number of stages doesn't match GPUs well ","action":"Use 1F1B schedule: One-Forward-One-Backward","steps":["Use 1F1B schedule: One-Forward-One-Backward","Increase number of micro-batches to fill pipeline","Use interleaved 1F1B for better utilization","Use fewer pipeline stages if possible","Combine with tensor parallel for best efficiency"]},{"slug":"training-stuck-no-progress","title":"Training Stuck No Progress","category":"Reliability","text":"Training Stuck No Progress Reliability Loss is flat for many steps Validation metrics don't improve Training time is way over estimate stuck no-progress deadlock harness reliability Training loss doesn't decrease No progress for many iterations GPU utilization drops to zero Deadlock in distributed training Data loading is the bottleneck Learning rate too small (no visible progress) No gradient flow (detached graph) All-reduce hang in DDP","anchorText":"Training Stuck No Progress Loss is flat for many steps Validation metrics don't improve Training time is way over estimate stuck no-progress deadlock harness reliability Deadlock in distributed training Data loading is the bottleneck Learning rate too small (no visible progress) No gradient flow (detached graph) All-reduce hang in DDP ","action":"Profile with torch.profiler to find bottleneck","steps":["Profile with torch.profiler to find bottleneck","Check GPU utilization: nvidia-smi","Use smaller learning rate to verify any progress","Verify gradient flow: print gradients norm","Set distributed timeout: dist.init_process_group(timeout=...)"]},{"slug":"safetensors-load-error","title":"Safetensors Load Error","category":"Memory","text":"Safetensors Load Error Memory SafetensorsError: Error while deserializing header InvalidHeaderDeserialization File not found or corrupted safetensors checkpoint load corruption memory Safetensors file can't be loaded Error loading safetensors Safetensors corruption Safetensors file is corrupted or truncated Safetensors version mismatch Model architecture doesn't match saved state File is being written while reading Pickle vs safetensors metadata mismatch","anchorText":"Safetensors Load Error SafetensorsError: Error while deserializing header InvalidHeaderDeserialization File not found or corrupted safetensors checkpoint load corruption memory Safetensors file is corrupted or truncated Safetensors version mismatch Model architecture doesn't match saved state File is being written while reading Pickle vs safetensors metadata mismatch ","action":"Verify safetensors integrity: from safetensors import safe_open; safe_open(path)","steps":["Verify safetensors integrity: from safetensors import safe_open; safe_open(path)","Match safetensors version across save and load","Wait for save to complete before reading","Use safetensors.torch.save_model() and load_model()","Check architecture matches: model.load_state_dict(strict=True)"]},{"slug":"loss-curve-anomaly","title":"Loss Curve Anomaly","category":"Training Stability","text":"Loss Curve Anomaly Training Stability Loss spike at specific step Loss plateau despite more training Loss curve is noisy loss-curve anomaly debugging training-stability diagnostics Loss has unexpected spikes Loss plateaus at unexpected level Loss oscillates wildly Bad data batch with wrong labels Learning rate too high Gradient explosion Numerical instability in loss Distributed sync issues","anchorText":"Loss Curve Anomaly Loss spike at specific step Loss plateau despite more training Loss curve is noisy loss-curve anomaly debugging training-stability diagnostics Bad data batch with wrong labels Learning rate too high Gradient explosion Numerical instability in loss Distributed sync issues ","action":"Visualize loss with TensorBoard/Wandb","steps":["Visualize loss with TensorBoard/Wandb","Use log scale for loss","Identify when spikes occur (data, batch)","Reduce learning rate and retry","Use gradient clipping to handle spikes"]},{"slug":"llm-tokenization-truncation","title":"LLM Tokenization Truncation","category":"Data Pipeline","text":"LLM Tokenization Truncation Data Pipeline Token count matches max_length Long documents are silently cut off Information at end of long docs is lost tokenization truncation long-context rag llm data-pipeline LLM summarization misses key points LLM QA fails on long documents Model output seems incomplete max_length truncation removes end of document Sliding window not used No handling of long context Important info at end of long documents","anchorText":"LLM Tokenization Truncation Token count matches max_length Long documents are silently cut off Information at end of long docs is lost tokenization truncation long-context rag llm data-pipeline max_length truncation removes end of document Sliding window not used No handling of long context Important info at end of long documents ","action":"Use sliding window: stride = max_length // 2","steps":["Use sliding window: stride = max_length // 2","Use long-context model: Llama-3 8K, 32K, 128K","Use chunking: split document into chunks","Use FlashAttention for memory efficiency","Use LongRoPE or other long-context techniques"]},{"slug":"efa-driver-issue","title":"AWS EFA Driver Issue","category":"Communication","text":"AWS EFA Driver Issue Communication NCCL WARN NET/IB: No IB devices found EFA device not visible fi_info fails efa aws nccl inter-node communication infrastructure Inter-node bandwidth is low EFA not used by NCCL NCCL falls back to TCP/IP EFA driver not installed EFA-enabled NCCL not installed EFA not enabled in container Security group blocks EFA traffic","anchorText":"AWS EFA Driver Issue NCCL WARN NET/IB: No IB devices found EFA device not visible fi_info fails efa aws nccl inter-node communication infrastructure EFA driver not installed EFA-enabled NCCL not installed EFA not enabled in container Security group blocks EFA traffic ","action":"Install EFA: ./efa_installer.sh -y","steps":["Install EFA: ./efa_installer.sh -y","Install aws-ofi-nccl: ./aws-ofi-nccl/install.sh","Enable EFA in container: --cap-add=IPC_LOCK","Open EFA security group ports","Verify: fi_info -p efa"]},{"slug":"tensor-views-memory-leak","title":"Tensor Views Memory Leak","category":"Memory","text":"Tensor Views Memory Leak Memory GPU memory grows with operations torch.cuda.memory_allocated() shows large tensors Memory not freed after model deletion tensor-views memory-leak reference-counting memory debugging Memory is higher than expected after slicing Deleting a tensor doesn't free memory Memory accumulates with many tensor operations Tensor view holds reference to base tensor Base tensor not garbage collected Python references prevent tensor release Storage vs tensor object confusion","anchorText":"Tensor Views Memory Leak GPU memory grows with operations torch.cuda.memory_allocated() shows large tensors Memory not freed after model deletion tensor-views memory-leak reference-counting memory debugging Tensor view holds reference to base tensor Base tensor not garbage collected Python references prevent tensor release Storage vs tensor object confusion ","action":"Use .clone() if you need a copy: x.clone()","steps":["Use .clone() if you need a copy: x.clone()","Use .contiguous() if you need a new tensor","Detach if grad not needed: x.detach()","Use del to remove references","Use weakref for caching tensors"]},{"slug":"gcc-version-mismatch","title":"GCC Version Mismatch","category":"Environment","text":"GCC Version Mismatch Environment error: command 'gcc' failed with exit status 1 ImportError: /lib/x86_64-linux-gnu/libstdc++.so.6: version `GLIBCXX_3.4.29' not found gcc compiler libstdc++ environment compilation C++ extension fails to compile libstdc++ version not found undefined symbol at runtime GCC version too old for extension GCC version too new for OS libstdc++ missing symbols Container GCC differs from host","anchorText":"GCC Version Mismatch error: command 'gcc' failed with exit status 1 ImportError: /lib/x86_64-linux-gnu/libstdc++.so.6: version `GLIBCXX_3.4.29' not found gcc compiler libstdc++ environment compilation GCC version too old for extension GCC version too new for OS libstdc++ missing symbols Container GCC differs from host ","action":"Match GCC version: gcc --version","steps":["Match GCC version: gcc --version","Install dev tools: apt install build-essential","Use conda's gcc: conda install -c conda-forge gcc","Match libstdc++ to extension requirements","Build extensions in matching environment"]},{"slug":"warmup-missing","title":"Warmup Missing","category":"Training Stability","text":"Warmup Missing Training Stability Loss is very high in first iterations First few epochs have NaN loss Model performance is worse than expected warmup scheduler transformer adamw training-stability Training is unstable at start Loss spikes in first epochs AdamW training diverges without warmup Learning rate too high at start Adam beta2 assumes warmup LayerNorm outputs have high variance at init Embeddings have high gradient magnitudes","anchorText":"Warmup Missing Loss is very high in first iterations First few epochs have NaN loss Model performance is worse than expected warmup scheduler transformer adamw training-stability Learning rate too high at start Adam beta2 assumes warmup LayerNorm outputs have high variance at init Embeddings have high gradient magnitudes ","action":"Add linear warmup: 1000-10000 steps","steps":["Add linear warmup: 1000-10000 steps","Use warmup scheduler: transformers.get_linear_schedule_with_warmup","Set warmup_steps = 0.05-0.1 * total_steps","Use warmup for Adam beta2 stabilization","Combine with cosine decay for best results"]},{"slug":"data-versioning-issue","title":"Data Versioning Issue","category":"Data Pipeline","text":"Data Versioning Issue Data Pipeline Validation accuracy differs from previous run Dataset has new samples Labels changed data-versioning dvc reproducibility data-pipeline reliability Same code gives different results Data drift causes model degradation Cannot reproduce previous training run Data updated without version control Train/test split uses random seed without tracking Data augmentation depends on library version Label versioning not tracked","anchorText":"Data Versioning Issue Validation accuracy differs from previous run Dataset has new samples Labels changed data-versioning dvc reproducibility data-pipeline reliability Data updated without version control Train/test split uses random seed without tracking Data augmentation depends on library version Label versioning not tracked ","action":"Use DVC for data versioning: dvc add data/","steps":["Use DVC for data versioning: dvc add data/","Track dataset hash: hashlib.sha256(file)","Use HF datasets versioning","Lock data with checksum: SHA256 verification","Use Denpex to track data versions with experiments"]},{"slug":"ddp-port-conflict","title":"DDP Port Conflict","category":"Communication","text":"DDP Port Conflict Communication RuntimeError: Address already in use torch.distributed.DistBackendError bind: address already in use ddp port conflict distributed master-port communication DDP training fails to start Address already in use error Distributed init hangs Default port 29500 already in use MASTER_PORT collision Multiple jobs trying to use same port Port not released after job end","anchorText":"DDP Port Conflict RuntimeError: Address already in use torch.distributed.DistBackendError bind: address already in use ddp port conflict distributed master-port communication Default port 29500 already in use MASTER_PORT collision Multiple jobs trying to use same port Port not released after job end ","action":"Set unique MASTER_PORT for each job: export MASTER_PORT=29501","steps":["Set unique MASTER_PORT for each job: export MASTER_PORT=29501","Use port range: export MASTER_PORT=$(shuf -i 29500-49500 -n 1)","Use torchrun with --master_port","Check port usage: netstat -tulpn | grep PORT","Wait for TIME_WAIT to expire or set SO_REUSEADDR"]},{"slug":"silent-data-corruption","title":"Silent Data Corruption","category":"Reliability","text":"Silent Data Corruption Reliability Same code gives different results Bit flips in memory or storage Some samples produce different outputs silent-corruption bit-flip ecc reliability integrity Training loss is wrong but no error Model accuracy is poor without obvious cause Results are non-reproducible across runs GPU memory bit flips (cosmic rays, hardware issues) Storage bit flips in checkpoint CPU memory errors Network bit flips in distributed training","anchorText":"Silent Data Corruption Same code gives different results Bit flips in memory or storage Some samples produce different outputs silent-corruption bit-flip ecc reliability integrity GPU memory bit flips (cosmic rays, hardware issues) Storage bit flips in checkpoint CPU memory errors Network bit flips in distributed training ","action":"Verify data integrity: SHA256 of dataset","steps":["Verify data integrity: SHA256 of dataset","Use ECC memory where available","Checkpoint integrity verification: hash check","Re-run training to detect non-determinism","Use Denpex to detect anomalies in training metrics"]},{"slug":"pipeline-parallel-memory","title":"Pipeline Parallel Memory","category":"Memory","text":"Pipeline Parallel Memory Memory Some GPUs OOM while others have spare memory Pipeline training is memory-bound Memory spikes at start/end of pipeline pipeline-parallel activation-memory stage-balance memory megatron Pipeline parallel OOM Memory imbalance across stages Activation memory grows with pipeline depth Pipeline stage imbalance Activation memory stored for all micro-batches Cross-stage communication buffers First/last stage have different memory profiles","anchorText":"Pipeline Parallel Memory Some GPUs OOM while others have spare memory Pipeline training is memory-bound Memory spikes at start/end of pipeline pipeline-parallel activation-memory stage-balance memory megatron Pipeline stage imbalance Activation memory stored for all micro-batches Cross-stage communication buffers First/last stage have different memory profiles ","action":"Balance pipeline stages by parameter count","steps":["Balance pipeline stages by parameter count","Use gradient checkpointing within stages","Use fewer micro-batches to reduce activation memory","Use 1F1B schedule for less activation memory","Combine with tensor parallel for best results"]},{"slug":"nan-loss","title":"NaN Loss","category":"Training Stability","text":"NaN Loss Training Stability loss.item() returns nan Model outputs are NaN Gradients are NaN Loss is inf nan loss training-stability debugging critical Loss becomes NaN mid-training All model parameters become NaN Training can't recover from NaN Gradient explosion FP16 underflow/overflow Bad data batch with extreme values Division by zero in loss log(0) or sqrt(negative) in loss","anchorText":"NaN Loss loss.item() returns nan Model outputs are NaN Gradients are NaN Loss is inf nan loss training-stability debugging critical Gradient explosion FP16 underflow/overflow Bad data batch with extreme values Division by zero in loss log(0) or sqrt(negative) in loss ","action":"Detect NaN early: assert not torch.isnan(loss)","steps":["Detect NaN early: assert not torch.isnan(loss)","Use GradScaler with inf checks: scaler.step(optimizer)","Reduce learning rate","Use gradient clipping: torch.nn.utils.clip_grad_norm_","Switch to BF16 to avoid underflow"]},{"slug":"dataset-bias","title":"Dataset Bias","category":"Data Pipeline","text":"Dataset Bias Data Pipeline Validation accuracy high, production accuracy low Model focuses on irrelevant features Performance varies across subgroups dataset-bias fairness bias spurious-correlation data-pipeline Model performs well on validation but poorly in production Model relies on background, not object Demographic bias in model predictions Selection bias in data collection Annotation bias from labelers Historical bias in data Sampling bias in splits Spurious correlations in features","anchorText":"Dataset Bias Validation accuracy high, production accuracy low Model focuses on irrelevant features Performance varies across subgroups dataset-bias fairness bias spurious-correlation data-pipeline Selection bias in data collection Annotation bias from labelers Historical bias in data Sampling bias in splits Spurious correlations in features ","action":"Audit dataset for bias: subgroup analysis","steps":["Audit dataset for bias: subgroup analysis","Use diverse data sources","Stratified sampling for splits","Use bias metrics: equalized odds, demographic parity","Use counterfactual data augmentation"]},{"slug":"rdma-configuration-issue","title":"RDMA Configuration Issue","category":"Communication","text":"RDMA Configuration Issue Communication NCCL not using RDMA RDMA device not detected ibstat shows no active ports rdma infiniband mellanox communication infrastructure Inter-node bandwidth is low RDMA not being used Training is slow on multi-node setup RDMA not enabled in BIOS Mellanox OFED driver not installed Subnet manager not running RDMA device not in container PCIe not configured for RDMA","anchorText":"RDMA Configuration Issue NCCL not using RDMA RDMA device not detected ibstat shows no active ports rdma infiniband mellanox communication infrastructure RDMA not enabled in BIOS Mellanox OFED driver not installed Subnet manager not running RDMA device not in container PCIe not configured for RDMA ","action":"Enable RDMA in BIOS settings","steps":["Enable RDMA in BIOS settings","Install Mellanox OFED: mlnxofedinstall","Start subnet manager: opensm","Mount RDMA devices in container: --device=/dev/infiniband","Verify: ibstat and ibv_devinfo"]},{"slug":"deep-speed-oom","title":"DeepSpeed OOM","category":"Memory","text":"DeepSpeed OOM Memory CUDA OOM with DeepSpeed ZeRO-3 still OOMs CPU offload still OOMs deepspeed zero offload memory training DeepSpeed training OOMs ZeRO-3 OOM despite offloading DeepSpeed config doesn't reduce memory ZeRO stage not aggressive enough CPU offload not enabled Activation checkpointing not used Sub-optimal config for model size Offload too much to slow storage","anchorText":"DeepSpeed OOM CUDA OOM with DeepSpeed ZeRO-3 still OOMs CPU offload still OOMs deepspeed zero offload memory training ZeRO stage not aggressive enough CPU offload not enabled Activation checkpointing not used Sub-optimal config for model size Offload too much to slow storage ","action":"Increase ZeRO stage: stage=3 for largest models","steps":["Increase ZeRO stage: stage=3 for largest models","Enable CPU offload: offload_optimizer, offload_param","Use activation checkpointing: activation_checkpointing=True","Tune ZeRO-3 config: reduce_bucket_size, overlap_comm","Use FP16/BF16 to reduce memory"]},{"slug":"ssl-certificate-error","title":"SSL Certificate Error","category":"Environment","text":"SSL Certificate Error Environment SSL: CERTIFICATE_VERIFY_FAILED urllib3.exceptions.MaxRetryError Could not fetch URL ssl certificate https proxy environment HTTPS connection fails Certificate verification fails pip install fails with SSL error Corporate proxy intercepts SSL Self-signed certs in cluster Outdated CA certificates Clock skew causing cert validation to fail","anchorText":"SSL Certificate Error SSL: CERTIFICATE_VERIFY_FAILED urllib3.exceptions.MaxRetryError Could not fetch URL ssl certificate https proxy environment Corporate proxy intercepts SSL Self-signed certs in cluster Outdated CA certificates Clock skew causing cert validation to fail ","action":"Update CA certificates: pip install --upgrade certifi","steps":["Update CA certificates: pip install --upgrade certifi","Set SSL_CERT_FILE: export SSL_CERT_FILE=/path/to/cert","Use corporate proxy with proper cert","Disable verification (NOT recommended): export CURL_CA_BUNDLE=\"\"","Install certs in container: update-ca-certificates"]},{"slug":"gradient-clipping-missing","title":"Gradient Clipping Missing","category":"Training Stability","text":"Gradient Clipping Missing Training Stability Gradient norm grows very large Loss spike followed by NaN Model parameters oscillate wildly gradient-clipping gradient-explosion training-stability transformer rnn Training diverges with large gradients Loss spikes occasionally NaN loss from gradient explosion Gradient norm unbounded No clip_grad_norm_() call Clip value too high to be effective Clip applied to wrong optimizer","anchorText":"Gradient Clipping Missing Gradient norm grows very large Loss spike followed by NaN Model parameters oscillate wildly gradient-clipping gradient-explosion training-stability transformer rnn Gradient norm unbounded No clip_grad_norm_() call Clip value too high to be effective Clip applied to wrong optimizer ","action":"Add gradient clipping: torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)","steps":["Add gradient clipping: torch.nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)","Use max_norm=1.0 for transformers, 5.0 for RNNs","Apply before optimizer step","Use value clipping for specific cases: clip_grad_value_","Combine with warmup for stability"]},{"slug":"validation-set-leakage","title":"Validation Set Leakage","category":"Data Pipeline","text":"Validation Set Leakage Data Pipeline Different validation and test accuracy Hyperparameter tuning leaks into model Preprocessing uses validation statistics data-leakage validation preprocessing reproducibility data-pipeline Validation accuracy much higher than test accuracy Model overfits to validation set Test accuracy is poor despite high validation accuracy Validation samples in training set Preprocessing fit on train+val (e.g., normalization) Augmentation applied to validation set Data augmentation leaks via overlap Test set not held out properly","anchorText":"Validation Set Leakage Different validation and test accuracy Hyperparameter tuning leaks into model Preprocessing uses validation statistics data-leakage validation preprocessing reproducibility data-pipeline Validation samples in training set Preprocessing fit on train+val (e.g., normalization) Augmentation applied to validation set Data augmentation leaks via overlap Test set not held out properly ","action":"Strict separation: train/val/test from start","steps":["Strict separation: train/val/test from start","Fit preprocessing on train only: fit_transform on train, transform on val","Use cross-validation properly: outer test, inner val","Use Denpex to detect data leakage","Never tune on test set"]},{"slug":"nccl-rank-fail","title":"NCCL Rank Fail","category":"Communication","text":"NCCL Rank Fail Communication NCCL WARN unhandled system error NCCL error in: torch.distributed.DistBackendError nccl rank distributed configuration communication Distributed training fails to start Some ranks report NCCL errors Training hangs at init Rank assignment mismatch between nodes SSH/hostfile issues Different CUDA versions across nodes NCCL version mismatch Network connectivity between nodes","anchorText":"NCCL Rank Fail NCCL WARN unhandled system error NCCL error in: torch.distributed.DistBackendError nccl rank distributed configuration communication Rank assignment mismatch between nodes SSH/hostfile issues Different CUDA versions across nodes NCCL version mismatch Network connectivity between nodes ","action":"Verify rank assignment: print(os.environ['RANK'])","steps":["Verify rank assignment: print(os.environ['RANK'])","Test SSH: ssh nodename hostname","Match CUDA versions: nvcc --version","Match NCCL versions across nodes","Test with 2 nodes first"]},{"slug":"dataloader-failure","title":"DataLoader Failure","category":"Reliability","text":"DataLoader Failure Reliability RuntimeError: DataLoader worker (pid X) is killed BrokenPipeError CUDA error in DataLoader worker DataLoader hangs at first iteration dataloader worker failure data-pipeline reliability Training hangs at first batch DataLoader worker crashes DataLoader raises error mid-training num_workers=0 causes GPU starvation Too many workers exhaust memory Worker process dies with OOM Shared memory limit exceeded for shared tensors Data corruption in dataset","anchorText":"DataLoader Failure RuntimeError: DataLoader worker (pid X) is killed BrokenPipeError CUDA error in DataLoader worker DataLoader hangs at first iteration dataloader worker failure data-pipeline reliability num_workers=0 causes GPU starvation Too many workers exhaust memory Worker process dies with OOM Shared memory limit exceeded for shared tensors Data corruption in dataset ","action":"Set num_workers based on CPU cores: os.cpu_count()","steps":["Set num_workers based on CPU cores: os.cpu_count()","Set persistent_workers=True for long training","Use Denpex to monitor DataLoader performance","Limit worker memory with proper num_workers","Use proper error handling in __getitem__"]},{"slug":"fsdp-all-gather-timeout","title":"FSDP All Gather Timeout","category":"Memory","text":"FSDP All Gather Timeout Memory FSDP forward hang FSDP backward hang FSDP collective timeout fsdp all-gather timeout distributed memory FSDP training times out at all-gather FSDP hangs during forward/backward FSDP all-gather is slow Parameter size larger than expected Slow network between nodes FSDP bucket size too large Mixed precision FSDP issues","anchorText":"FSDP All Gather Timeout FSDP forward hang FSDP backward hang FSDP collective timeout fsdp all-gather timeout distributed memory Parameter size larger than expected Slow network between nodes FSDP bucket size too large Mixed precision FSDP issues ","action":"Tune FSDP bucket size: limit_all_gathers=True","steps":["Tune FSDP bucket size: limit_all_gathers=True","Use FSDP backward_prefetch for overlap","Use FSDP forward_prefetch for overlap","Set timeout: dist.init_process_group(timeout=...)","Profile FSDP all-gather: torch.distributed.monitored_barrier"]},{"slug":"mixed-precision-loss-scale","title":"Mixed Precision Loss Scale","category":"Training Stability","text":"Mixed Precision Loss Scale Training Stability FP16 underflow causes loss of gradient GradScaler scale is too small or too large Loss scale never updates mixed-precision fp16 gradscaler loss-scale training-stability FP16 training has NaN loss Loss is zero for many iterations GradScaler is not scaling loss correctly GradScaler scale too small (underflow) or too large (overflow) Inf/nan check fails silently Loss scale not updated for new loss distribution Static loss scale not appropriate for loss profile","anchorText":"Mixed Precision Loss Scale FP16 underflow causes loss of gradient GradScaler scale is too small or too large Loss scale never updates mixed-precision fp16 gradscaler loss-scale training-stability GradScaler scale too small (underflow) or too large (overflow) Inf/nan check fails silently Loss scale not updated for new loss distribution Static loss scale not appropriate for loss profile ","action":"Use dynamic loss scale (default in GradScaler)","steps":["Use dynamic loss scale (default in GradScaler)","Don't manually set scale unless necessary","Use BF16 instead of FP16 to avoid loss scaling","Monitor scale with scaler.get_scale()","Combine with gradient clipping"]},{"slug":"tokenizer-version-mismatch","title":"Tokenizer Version Mismatch","category":"Data Pipeline","text":"Tokenizer Version Mismatch Data Pipeline Tokenizer loads old vocab New tokens not in tokenizer Special tokens differ tokenizer version nlp reproducibility data-pipeline Model produces different outputs in production Token IDs differ between dev and prod Model accuracy drops after deployment Different transformers version between training and inference Tokenizer not saved with model SentencePiece vs BPE mismatch Vocabulary not updated","anchorText":"Tokenizer Version Mismatch Tokenizer loads old vocab New tokens not in tokenizer Special tokens differ tokenizer version nlp reproducibility data-pipeline Different transformers version between training and inference Tokenizer not saved with model SentencePiece vs BPE mismatch Vocabulary not updated ","action":"Save tokenizer with model: tokenizer.save_pretrained(path)","steps":["Save tokenizer with model: tokenizer.save_pretrained(path)","Match transformers version: pip install transformers==4.35.0","Use AutoTokenizer to ensure compatibility","Track tokenizer hash in model card","Test inference in production environment"]},{"slug":"cpu-affinity-misconfiguration","title":"CPU Affinity Misconfiguration","category":"Performance","text":"CPU Affinity Misconfiguration Performance DataLoader slower than expected num_workers doesn't help High context switch rate cpu-affinity numa data-loader performance infrastructure DataLoader workers compete for same cores Training is slower than expected CPU utilization is uneven Workers and main process on same cores NUMA nodes not respected CPU pinning not set Hyperthreading causes contention","anchorText":"CPU Affinity Misconfiguration DataLoader slower than expected num_workers doesn't help High context switch rate cpu-affinity numa data-loader performance infrastructure Workers and main process on same cores NUMA nodes not respected CPU pinning not set Hyperthreading causes contention ","action":"Set CPU affinity for workers: psutil.Process().cpu_affinity()","steps":["Set CPU affinity for workers: psutil.Process().cpu_affinity()","Use numactl for NUMA-aware binding: numactl --cpunodebind=0","Set OMP_NUM_THREADS appropriately","Use taskset for explicit pinning: taskset -c 0-7","Profile with perf top or htop"]},{"slug":"cuda-context-leak","title":"CUDA Context Leak","category":"Memory","text":"CUDA Context Leak Memory nvidia-smi shows growing memory with no process GPU memory not released after process exit Multiprocessing CUDA contexts accumulate cuda-context memory-leak multiprocessing memory cleanup GPU memory grows across processes CUDA context count keeps increasing GPU 0 shows memory in use without processes CUDA context not destroyed on process exit torch.cuda.empty_cache() not called CUDA context per process accumulates Notebook kernel restart not done","anchorText":"CUDA Context Leak nvidia-smi shows growing memory with no process GPU memory not released after process exit Multiprocessing CUDA contexts accumulate cuda-context memory-leak multiprocessing memory cleanup CUDA context not destroyed on process exit torch.cuda.empty_cache() not called CUDA context per process accumulates Notebook kernel restart not done ","action":"Destroy context explicitly: torch.cuda.empty_cache()","steps":["Destroy context explicitly: torch.cuda.empty_cache()","Use multiprocessing with proper cleanup","Use atexit to cleanup: atexit.register(torch.cuda.empty_cache)","Restart notebook kernel periodically","Use CUDA_VISIBLE_DEVICES to limit GPU access"]},{"slug":"ulimit-too-low","title":"Ulimit Too Low","category":"Environment","text":"Ulimit Too Low Environment OSError: [Errno 24] Too many open files Resource temporarily unavailable fork: Resource temporarily unavailable ulimit open-files resource-limit environment multiprocessing Too many open files error Cannot create more processes Data loading fails with file handle error Default ulimit too low for ML workloads ulimit -n shows 1024 (default) Many file handles from HF datasets Process limit too low for multiprocessing","anchorText":"Ulimit Too Low OSError: [Errno 24] Too many open files Resource temporarily unavailable fork: Resource temporarily unavailable ulimit open-files resource-limit environment multiprocessing Default ulimit too low for ML workloads ulimit -n shows 1024 (default) Many file handles from HF datasets Process limit too low for multiprocessing ","action":"Increase ulimit: ulimit -n 65536","steps":["Increase ulimit: ulimit -n 65536","Set persistent: echo '* soft nofile 65536' >> /etc/security/limits.conf","Verify with ulimit -a","Use systemd for service limits: LimitNOFILE=65536","Use Denpex to detect file handle exhaustion"]},{"slug":"lr-too-high","title":"LR Too High","category":"Training Stability","text":"LR Too High Training Stability Loss is NaN after few iterations Loss spike at start of training Model parameters become NaN learning-rate too-high training-stability divergence hyperparameter Loss explodes early in training Training diverges with NaN loss Model outputs are NaN or Inf Learning rate too high for model No warmup before high LR Adam epsilon too small Gradient explosion at start","anchorText":"LR Too High Loss is NaN after few iterations Loss spike at start of training Model parameters become NaN learning-rate too-high training-stability divergence hyperparameter Learning rate too high for model No warmup before high LR Adam epsilon too small Gradient explosion at start ","action":"Reduce learning rate: 10x lower","steps":["Reduce learning rate: 10x lower","Use LR finder to find good LR","Add warmup: 1000-10000 steps","Use gradient clipping","Start with proven LR for architecture"]},{"slug":"image-corrupt-detection","title":"Image Corruption Detection","category":"Data Pipeline","text":"Image Corruption Detection Data Pipeline Cannot identify image file Truncated file read Corrupt JPEG data image-corrupt pil torchvision data-pipeline reliability PIL.UnidentifiedImageError mid-training torchvision.io.read_image fails Training crashes on specific image Image file is corrupted Image format not supported Truncated download Wrong file extension Memory error for huge images","anchorText":"Image Corruption Detection Cannot identify image file Truncated file read Corrupt JPEG data image-corrupt pil torchvision data-pipeline reliability Image file is corrupted Image format not supported Truncated download Wrong file extension Memory error for huge images ","action":"Detect corruption: PIL.Image.verify()","steps":["Detect corruption: PIL.Image.verify()","Use corrupted torchvision: from torchvision.datasets import ImageFolder","Pre-check dataset: walk dataset and verify","Use albumentations error handling","Use streaming with error recovery"]},{"slug":"nccl-version-mismatch","title":"NCCL Version Mismatch","category":"Communication","text":"NCCL Version Mismatch Communication NCCL ABI mismatch NCCL library version not consistent PyTorch NCCL version differs from system nccl version distributed communication incompatibility Distributed training hangs or fails NCCL operations fail silently Different NCCL versions on different nodes Different NCCL versions across nodes Container NCCL older/newer than host NCCL PyTorch bundled NCCL vs system NCCL NCCL ABI breaking changes between versions","anchorText":"NCCL Version Mismatch NCCL ABI mismatch NCCL library version not consistent PyTorch NCCL version differs from system nccl version distributed communication incompatibility Different NCCL versions across nodes Container NCCL older/newer than host NCCL PyTorch bundled NCCL vs system NCCL NCCL ABI breaking changes between versions ","action":"Match NCCL versions: nccl --version","steps":["Match NCCL versions: nccl --version","Use container NCCL consistently: NCCL_LIB_DIR","Pin NCCL version: pip install nvidia-nccl-cu12==2.21.5","Use NCCL_SOCKET_IFNAME for explicit control","Match CUDA versions (NCCL is CUDA-version specific)"]},{"slug":"zombie-process","title":"Zombie Process","category":"Reliability","text":"Zombie Process Reliability Z state in ps output defunct process Process not reaped after exit zombie defunct process multiprocessing reliability Process count grows over time ps shows defunct processes System becomes unresponsive Parent process not calling wait() SIGCHLD handler not installed Process leak from training workers Distributed worker not properly cleaned up","anchorText":"Zombie Process Z state in ps output defunct process Process not reaped after exit zombie defunct process multiprocessing reliability Parent process not calling wait() SIGCHLD handler not installed Process leak from training workers Distributed worker not properly cleaned up ","action":"Always call wait() on child processes","steps":["Always call wait() on child processes","Use Process.join() in Python","Install SIGCHLD handler: signal.signal(SIGCHLD, handler)","Use multiprocessing.Pool with proper cleanup","Use Denpex to detect zombie processes"]},{"slug":"pytorch-cuda-caching-allocator-fragmentation","title":"CUDA Caching Allocator Fragmentation","category":"Memory","text":"CUDA Caching Allocator Fragmentation Memory torch.cuda.memory_reserved() >> memory_allocated() Empty cache doesn't help Snapshot shows many small free blocks cuda-allocator fragmentation memory memory-pool cuda OOM despite enough free memory Reserved memory is high but allocated is low Memory fragmentation visible in snapshot Variable tensor sizes cause fragmentation Allocs and frees of different sizes Long-running training accumulates fragmentation Memory pool cannot coalesce small blocks","anchorText":"CUDA Caching Allocator Fragmentation torch.cuda.memory_reserved() >> memory_allocated() Empty cache doesn't help Snapshot shows many small free blocks cuda-allocator fragmentation memory memory-pool cuda Variable tensor sizes cause fragmentation Allocs and frees of different sizes Long-running training accumulates fragmentation Memory pool cannot coalesce small blocks ","action":"Use expandable_segments: torch.cuda.memory._set_allocator_settings('expandable_segments:True')","steps":["Use expandable_segments: torch.cuda.memory._set_allocator_settings('expandable_segments:True')","Use empty_cache() strategically","Round up allocation sizes","Use static memory pool where possible","Profile with torch.cuda.memory._snapshot()"]},{"slug":"gradient-explosion","title":"Gradient Explosion","category":"Training Stability","text":"Gradient Explosion Training Stability Gradient norm grows very large Loss spike followed by divergence Loss is inf or NaN gradient-explosion gradient-clipping training-stability rnn transformer Loss spikes occasionally NaN loss from large gradients Model parameters become unstable Gradient norm exceeds threshold Recurrent weight matrices amplify gradients No gradient clipping Learning rate too high Unstable loss landscape","anchorText":"Gradient Explosion Gradient norm grows very large Loss spike followed by divergence Loss is inf or NaN gradient-explosion gradient-clipping training-stability rnn transformer Gradient norm exceeds threshold Recurrent weight matrices amplify gradients No gradient clipping Learning rate too high Unstable loss landscape ","action":"Apply gradient clipping: torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)","steps":["Apply gradient clipping: torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)","Use max_norm=1.0 for transformers, 5.0 for RNNs","Add normalization layers (LayerNorm, BatchNorm)","Reduce learning rate","Use weight tying in RNNs"]},{"slug":"determinism-broken","title":"Determinism Broken","category":"Data Pipeline","text":"Determinism Broken Data Pipeline Different loss curves on same data Different model accuracy on same training Distributed training non-deterministic determinism reproducibility seed data-pipeline debugging Same code gives different results Cannot reproduce training run Random seed doesn't fix results CUDA non-deterministic operations DataLoader with multiple workers Distributed training with all-reduce Random operations not seeded Some ops have no deterministic implementation","anchorText":"Determinism Broken Different loss curves on same data Different model accuracy on same training Distributed training non-deterministic determinism reproducibility seed data-pipeline debugging CUDA non-deterministic operations DataLoader with multiple workers Distributed training with all-reduce Random operations not seeded Some ops have no deterministic implementation ","action":"Set torch.use_deterministic_algorithms(True)","steps":["Set torch.use_deterministic_algorithms(True)","Set cudnn.deterministic=True, cudnn.benchmark=False","Seed all RNG: torch, numpy, random, cuda","Use DataLoader worker_init_fn for per-worker seeds","Set CUBLAS_WORKSPACE_CONFIG for deterministic cuBLAS"]},{"slug":"container-time-drift","title":"Container Time Drift","category":"Infrastructure","text":"Container Time Drift Infrastructure Container time differs from host Clock skew in distributed training Schedules based on time fail time-drift ntp container infrastructure scheduling TLS certificates fail in container Logged timestamps are wrong Schedules fire at wrong time Container time not synced NTP not running in container Timezone misconfiguration Container start time drift","anchorText":"Container Time Drift Container time differs from host Clock skew in distributed training Schedules based on time fail time-drift ntp container infrastructure scheduling Container time not synced NTP not running in container Timezone misconfiguration Container start time drift ","action":"Sync container time: mount /etc/localtime","steps":["Sync container time: mount /etc/localtime","Run NTP in container: apt install ntp","Use host time: docker run -v /etc/localtime:/etc/localtime:ro","Use UTC consistently","Set timezone: TZ=UTC"]},{"slug":"optimizer-state-memory","title":"Optimizer State Memory","category":"Memory","text":"Optimizer State Memory Memory Adam optimizer memory is 2x model size AdamW with weight decay adds to state Mixed precision optimizer state with FP32 master weights optimizer adam adamw state memory Memory usage much higher than model size Adam training OOM despite small model Optimizer state uses more memory than expected Adam stores 2x model size (m and v) FP32 master weights for mixed precision Gradient accumulation adds to state Optimizer state not offloaded","anchorText":"Optimizer State Memory Adam optimizer memory is 2x model size AdamW with weight decay adds to state Mixed precision optimizer state with FP32 master weights optimizer adam adamw state memory Adam stores 2x model size (m and v) FP32 master weights for mixed precision Gradient accumulation adds to state Optimizer state not offloaded ","action":"Use Adam 8-bit: bitsandbytes.optim.Adam8bit","steps":["Use Adam 8-bit: bitsandbytes.optim.Adam8bit","Use ZeRO-1/2 to shard optimizer state","Use CPU offload for optimizer state","Use DeepSpeed with offload_optimizer","Use AdaFactor for memory-efficient optimizer"]},{"slug":"env-variable-not-set","title":"Environment Variable Not Set","category":"Environment","text":"Environment Variable Not Set Environment CUDA_VISIBLE_DEVICES not respected HF_TOKEN not set for gated models MASTER_ADDR/MASTER_PORT not set env-variable configuration deployment environment debugging Training uses wrong GPU HF model download fails Distributed training fails to connect Required env var not set in script Container missing env vars Shell doesn't export variable CI/CD doesn't set env vars","anchorText":"Environment Variable Not Set CUDA_VISIBLE_DEVICES not respected HF_TOKEN not set for gated models MASTER_ADDR/MASTER_PORT not set env-variable configuration deployment environment debugging Required env var not set in script Container missing env vars Shell doesn't export variable CI/CD doesn't set env vars ","action":"Set env vars at start: export CUDA_VISIBLE_DEVICES=0,1","steps":["Set env vars at start: export CUDA_VISIBLE_DEVICES=0,1","Use .env files for configuration","Document required env vars","Set in container: ENV VAR=value","Verify env vars: python -c 'import os; print(os.environ.get(\"VAR\"))'"]},{"slug":"loss-not-decreasing","title":"Loss Not Decreasing","category":"Training Stability","text":"Loss Not Decreasing Training Stability Loss is constant across many epochs Loss plateau despite training Loss is higher than expected loss not-decreasing debugging training-stability diagnostics Loss stays flat Loss decreases then plateaus Loss is constant from start Learning rate too low Learning rate too high (jumping around minimum) Bad data (all same labels) Wrong loss function Model architecture broken Gradient flow blocked","anchorText":"Loss Not Decreasing Loss is constant across many epochs Loss plateau despite training Loss is higher than expected loss not-decreasing debugging training-stability diagnostics Learning rate too low Learning rate too high (jumping around minimum) Bad data (all same labels) Wrong loss function Model architecture broken Gradient flow blocked ","action":"Profile with simple test: overfit one batch","steps":["Profile with simple test: overfit one batch","Verify loss decreases: try LR=1e-3 with Adam","Check data: visualize inputs and labels","Check gradient flow: print layer gradients","Try simpler model first"]},{"slug":"multi-task-learning-conflict","title":"Multi-Task Learning Conflict","category":"Data Pipeline","text":"Multi-Task Learning Conflict Data Pipeline Task weights cause imbalance Per-task gradients conflict Multi-task loss is unstable multi-task gradnorm pcgrad training-stability data-pipeline Some tasks improve while others regress Loss is dominated by one task Model performs well on average but poorly on individual tasks Task losses at different scales (BCE vs L1) Gradient conflict between tasks No task weighting or uncertainty weighting Naive loss sum without balancing","anchorText":"Multi-Task Learning Conflict Task weights cause imbalance Per-task gradients conflict Multi-task loss is unstable multi-task gradnorm pcgrad training-stability data-pipeline Task losses at different scales (BCE vs L1) Gradient conflict between tasks No task weighting or uncertainty weighting Naive loss sum without balancing ","action":"Normalize losses to similar scale","steps":["Normalize losses to similar scale","Use GradNorm for adaptive task weighting","Use uncertainty weighting (Kendall et al.)","Use gradient projection: PCGrad","Monitor per-task loss separately"]},{"slug":"ddp-setup-error","title":"DDP Setup Error","category":"Communication","text":"DDP Setup Error Communication RuntimeError: Distributed package doesn't have NCCL built Default process group is not initialized Address already in use Timeout reached ddp setup distributed torchrun communication DDP fails to start torch.distributed.init_process_group fails DDP hangs at first all-reduce init_process_group not called backend not specified MASTER_ADDR/MASTER_PORT not set rank and world_size not set torchrun not used correctly","anchorText":"DDP Setup Error RuntimeError: Distributed package doesn't have NCCL built Default process group is not initialized Address already in use Timeout reached ddp setup distributed torchrun communication init_process_group not called backend not specified MASTER_ADDR/MASTER_PORT not set rank and world_size not set torchrun not used correctly ","action":"Use torchrun: torchrun --nproc_per_node=NUM_GPUS script.py","steps":["Use torchrun: torchrun --nproc_per_node=NUM_GPUS script.py","Set backend: init_process_group(backend='nccl')","Set env vars: RANK, WORLD_SIZE, MASTER_ADDR, MASTER_PORT","Use DistributedDataParallel wrapper","Test with single GPU first"]},{"slug":"wandb-tensorboard-failure","title":"Wandb/TensorBoard Failure","category":"Reliability","text":"Wandb/TensorBoard Failure Reliability wandb: ERROR Unable to log No logs appear in TensorBoard Wandb run is offline wandb tensorboard logging experiment-tracking reliability Metrics not logged Wandb sync fails TensorBoard not updating Wandb API key not set Network issues to wandb server TensorBoard log directory not writable Wandb service outage Disk full for log files","anchorText":"Wandb/TensorBoard Failure wandb: ERROR Unable to log No logs appear in TensorBoard Wandb run is offline wandb tensorboard logging experiment-tracking reliability Wandb API key not set Network issues to wandb server TensorBoard log directory not writable Wandb service outage Disk full for log files ","action":"Set WANDB_API_KEY: export WANDB_API_KEY=xxx","steps":["Set WANDB_API_KEY: export WANDB_API_KEY=xxx","Set WANDB_MODE=offline for offline logging","Verify TensorBoard dir: writer = SummaryWriter('runs/')","Check disk space: df -h","Use Denpex for local experiment tracking"]},{"slug":"transformer-attention-memory","title":"Transformer Attention Memory","category":"Memory","text":"Transformer Attention Memory Memory Sequence length 8K+ causes OOM Memory is O(L^2) for attention matrix Cannot train with long context attention transformer long-context flash-attention memory OOM with long sequences Attention memory dominates GPU usage Memory grows quadratically with sequence length Naive attention materializes L x L attention matrix Memory grows quadratically with sequence length Long sequences exceed GPU memory No efficient attention used","anchorText":"Transformer Attention Memory Sequence length 8K+ causes OOM Memory is O(L^2) for attention matrix Cannot train with long context attention transformer long-context flash-attention memory Naive attention materializes L x L attention matrix Memory grows quadratically with sequence length Long sequences exceed GPU memory No efficient attention used ","action":"Use Flash Attention: pip install flash-attn","steps":["Use Flash Attention: pip install flash-attn","Use xformers memory_efficient_attention","Use chunked attention for very long sequences","Use sliding window attention","Quantize attention to FP8/INT8 for memory savings"]},{"slug":"cosine-annealing-issue","title":"Cosine Annealing Issue","category":"Training Stability","text":"Cosine Annealing Issue Training Stability Final LR is too high or too low Warm restart hurts convergence Cosine schedule period is wrong cosine-annealing scheduler warm-restart training-stability hyperparameter Cosine annealing doesn't improve over constant LR Loss rebounds at warm restart Cosine schedule ends at wrong LR T_max doesn't match total steps min_lr too high (no proper decay) Warm restart frequency wrong No warmup before cosine","anchorText":"Cosine Annealing Issue Final LR is too high or too low Warm restart hurts convergence Cosine schedule period is wrong cosine-annealing scheduler warm-restart training-stability hyperparameter T_max doesn't match total steps min_lr too high (no proper decay) Warm restart frequency wrong No warmup before cosine ","action":"Set T_max = total_steps","steps":["Set T_max = total_steps","Set min_lr = 1e-6 or 1% of max_lr","Use warmup before cosine: 5-10% of training","Use cosine without warm restarts initially","Combine with linear warmup for stability"]},{"slug":"missing-data-augmentation","title":"Missing Data Augmentation","category":"Data Pipeline","text":"Missing Data Augmentation Data Pipeline Training accuracy 99%, validation 70% Loss on train much lower than val Model memorizes training data augmentation overfitting small-data data-pipeline regularization Validation accuracy is much lower than training Model overfits quickly Poor performance on new data No augmentation applied Augmentation applied only to train (correct) but insufficient Too weak augmentation Wrong augmentation for domain","anchorText":"Missing Data Augmentation Training accuracy 99%, validation 70% Loss on train much lower than val Model memorizes training data augmentation overfitting small-data data-pipeline regularization No augmentation applied Augmentation applied only to train (correct) but insufficient Too weak augmentation Wrong augmentation for domain ","action":"Add basic augmentations: RandomCrop, RandomHorizontalFlip","steps":["Add basic augmentations: RandomCrop, RandomHorizontalFlip","Use stronger augmentations for small data: RandAugment, Mixup","Use domain-specific augmentations","Use cutout/cutmix for regularization","Use AugMix for natural images"]},{"slug":"gpu-utilization-low","title":"GPU Utilization Low","category":"Performance","text":"GPU Utilization Low Performance nvidia-smi shows low % Training time is mostly data loading GPU power is low gpu-utilization bottleneck data-loading performance infrastructure GPU utilization is low (10-30%) Training is slower than expected GPUs are starving for data DataLoader is single-threaded (num_workers=0) Augmentation done on CPU Data fetching from slow storage Communication bottleneck CPU-GPU transfer overhead","anchorText":"GPU Utilization Low nvidia-smi shows low % Training time is mostly data loading GPU power is low gpu-utilization bottleneck data-loading performance infrastructure DataLoader is single-threaded (num_workers=0) Augmentation done on CPU Data fetching from slow storage Communication bottleneck CPU-GPU transfer overhead ","action":"Increase DataLoader workers: num_workers=4*NUM_GPUS","steps":["Increase DataLoader workers: num_workers=4*NUM_GPUS","Move augmentation to GPU (kornia)","Use pin_memory=True for faster transfer","Profile with torch.profiler to find bottleneck","Use Denpex to monitor GPU utilization"]},{"slug":"embedding-layer-memory","title":"Embedding Layer Memory","category":"Memory","text":"Embedding Layer Memory Memory Embedding size = vocab_size * hidden_dim Memory is significant for LLM vocab (32K-256K) Embedding gradient is sparse but state is dense embedding vocabulary llm memory sharding Embedding layer is most of model memory OOM with large vocabulary Embedding lookup creates huge tensors Vocab size too large for available memory Embedding matrix not sharded Embedding gradient state is full dense Adam state for embedding is 2x embedding size","anchorText":"Embedding Layer Memory Embedding size = vocab_size * hidden_dim Memory is significant for LLM vocab (32K-256K) Embedding gradient is sparse but state is dense embedding vocabulary llm memory sharding Vocab size too large for available memory Embedding matrix not sharded Embedding gradient state is full dense Adam state for embedding is 2x embedding size ","action":"Use ZeRO-3 to shard embedding across GPUs","steps":["Use ZeRO-3 to shard embedding across GPUs","Use smaller vocabulary","Tie embedding weights: lm_head.weight = embed.weight","Use embedding sharding in tensor parallel","Use BF16/FP16 for embedding to halve memory"]},{"slug":"connection-timeout","title":"Connection Timeout","category":"Environment","text":"Connection Timeout Environment Read timed out Connection reset by peer Operation timed out after 300000 ms timeout connection network environment reliability HF model download times out Wandb sync times out S3 download times out Network latency too high Firewall blocking connection Connection limit reached DNS resolution slow Server overloaded","anchorText":"Connection Timeout Read timed out Connection reset by peer Operation timed out after 300000 ms timeout connection network environment reliability Network latency too high Firewall blocking connection Connection limit reached DNS resolution slow Server overloaded ","action":"Increase timeout: requests.get(url, timeout=300)","steps":["Increase timeout: requests.get(url, timeout=300)","Use retry with backoff: urllib3.util.retry.Retry","Use HF mirror for faster downloads","Run offline and sync later","Use CDN for static content"]},{"slug":"lr-too-low","title":"LR Too Low","category":"Training Stability","text":"LR Too Low Training Stability Loss curve is flat Validation accuracy is low Training takes very long learning-rate too-low slow-convergence training-stability hyperparameter Loss decreases very slowly Loss plateaus at high value Training seems stuck Learning rate too low for model Adam epsilon too high No learning rate warmup Optimizer beta values wrong","anchorText":"LR Too Low Loss curve is flat Validation accuracy is low Training takes very long learning-rate too-low slow-convergence training-stability hyperparameter Learning rate too low for model Adam epsilon too high No learning rate warmup Optimizer beta values wrong ","action":"Increase learning rate: 10x higher","steps":["Increase learning rate: 10x higher","Use LR finder to find good LR","Try cyclic LR or one-cycle policy","Use Adam with default betas","Compare with paper's recommended LR"]},{"slug":"class-imbalance","title":"Class Imbalance","category":"Data Pipeline","text":"Class Imbalance Data Pipeline Confusion matrix shows bias to majority class Loss is dominated by majority class Macro F1 is much lower than micro F1 class-imbalance focal-loss oversampling long-tail data-pipeline Model predicts majority class for all inputs Minority classes have very low recall Accuracy is high but per-class F1 is poor Some classes have very few examples Loss weights not adjusted No oversampling/undersampling Hard examples not weighted Focal loss not used","anchorText":"Class Imbalance Confusion matrix shows bias to majority class Loss is dominated by majority class Macro F1 is much lower than micro F1 class-imbalance focal-loss oversampling long-tail data-pipeline Some classes have very few examples Loss weights not adjusted No oversampling/undersampling Hard examples not weighted Focal loss not used ","action":"Use class weights: CrossEntropyLoss(weight=class_weights)","steps":["Use class weights: CrossEntropyLoss(weight=class_weights)","Use focal loss: FocalLoss(alpha=class_weights, gamma=2)","Oversample minority: WeightedRandomSampler","Undersample majority: RandomUnderSampler","Use SMOTE for synthetic minority samples"]},{"slug":"nccl-bucket-size-mismatch","title":"NCCL Bucket Size Mismatch","category":"Communication","text":"NCCL Bucket Size Mismatch Communication GPU utilization is low during backward DDP overhead is high Distributed training slower than expected ddp bucket-size gradient-reduction performance communication DDP training is slow Gradient reduction is inefficient NCCL bucket size not optimal Default bucket size not optimal Many small parameters = many small reductions Bucket size affects overlap Re-tuning bucket size for model","anchorText":"NCCL Bucket Size Mismatch GPU utilization is low during backward DDP overhead is high Distributed training slower than expected ddp bucket-size gradient-reduction performance communication Default bucket size not optimal Many small parameters = many small reductions Bucket size affects overlap Re-tuning bucket size for model ","action":"Tune bucket size: bucket_cap_mb in DDP","steps":["Tune bucket size: bucket_cap_mb in DDP","Use larger buckets for large embeddings","Use smaller buckets for small models","Profile DDP communication: torch.profiler","Use gradient overlap with computation"]},{"slug":"checkpoint-partial-save","title":"Checkpoint Partial Save","category":"Reliability","text":"Checkpoint Partial Save Reliability Unexpected key(s) in state_dict Missing key(s) in state_dict Checkpoint file size is smaller than expected checkpoint partial-save reliability atomic-write state-dict Checkpoint is missing some keys Loading checkpoint gives Missing key error Model partially recovered after crash Process killed during checkpoint save Storage full mid-write Distributed rank failure during save No atomic write pattern Pickle can't handle complex objects","anchorText":"Checkpoint Partial Save Unexpected key(s) in state_dict Missing key(s) in state_dict Checkpoint file size is smaller than expected checkpoint partial-save reliability atomic-write state-dict Process killed during checkpoint save Storage full mid-write Distributed rank failure during save No atomic write pattern Pickle can't handle complex objects ","action":"Use atomic write: save to .tmp, then rename","steps":["Use atomic write: save to .tmp, then rename","Save with model.state_dict() (not pickle)","Use safetensors for safe saving","Catch exceptions in save code","Use strict=False on load for partial restore"]},{"slug":"cuda-oom","title":"CUDA Out of Memory","category":"Memory","text":"CUDA Out of Memory Memory torch.cuda.OutOfMemoryError CUDA error: out of memory OOM at forward or backward pass OOM at specific step (not always first) cuda oom memory critical training CUDA out of memory error RuntimeError: CUDA OOM Training crashes with OOM at specific batch Model + activations + optimizer state exceed GPU memory Activation memory peaks at specific layers Batch size too large Memory leak Gradient accumulation memory spike","anchorText":"CUDA Out of Memory torch.cuda.OutOfMemoryError CUDA error: out of memory OOM at forward or backward pass OOM at specific step (not always first) cuda oom memory critical training Model + activations + optimizer state exceed GPU memory Activation memory peaks at specific layers Batch size too large Memory leak Gradient accumulation memory spike ","action":"Reduce batch size","steps":["Reduce batch size","Use gradient accumulation","Use mixed precision (BF16/FP16)","Use gradient checkpointing","Use Flash Attention for memory efficiency"]},{"slug":"mixed-precision-overflow","title":"Mixed Precision Overflow","category":"Training Stability","text":"Mixed Precision Overflow Training Stability RuntimeError: loss is inf GradScaler inf check fails Loss scale becomes very small mixed-precision fp16 overflow gradscaler training-stability FP16 loss becomes inf Gradients are inf in FP16 Training crashes with overflow FP16 max value is 65504 Loss/activation values exceed FP16 range No dynamic loss scaling Softmax/log operations can overflow","anchorText":"Mixed Precision Overflow RuntimeError: loss is inf GradScaler inf check fails Loss scale becomes very small mixed-precision fp16 overflow gradscaler training-stability FP16 max value is 65504 Loss/activation values exceed FP16 range No dynamic loss scaling Softmax/log operations can overflow ","action":"Use dynamic loss scaling: torch.cuda.amp.GradScaler","steps":["Use dynamic loss scaling: torch.cuda.amp.GradScaler","Use BF16 on Ampere+ to avoid overflow","Reduce loss magnitude (e.g., divide by accumulation steps)","Clip gradient norm: torch.nn.utils.clip_grad_norm_","Use TF32 for matmul on Ampere+"]},{"slug":"audio-channel-mismatch","title":"Audio Channel Mismatch","category":"Data Pipeline","text":"Audio Channel Mismatch Data Pipeline RuntimeError: channel mismatch Model expects mono but data is stereo Stereo audio trained as mono audio mono stereo channel data-pipeline Audio model fails on stereo input Audio model has poor accuracy on real data Expected 1 channel, got 2 Audio loaded as stereo, model expects mono Different audio sources have different channel counts No channel conversion in pipeline Model architecture doesn't match data channels","anchorText":"Audio Channel Mismatch RuntimeError: channel mismatch Model expects mono but data is stereo Stereo audio trained as mono audio mono stereo channel data-pipeline Audio loaded as stereo, model expects mono Different audio sources have different channel counts No channel conversion in pipeline Model architecture doesn't match data channels ","action":"Convert to mono: torchaudio.transforms.DownmixMono()","steps":["Convert to mono: torchaudio.transforms.DownmixMono()","Use torchaudio.load with normalize=True","Reshape audio: audio.mean(dim=0, keepdim=True)","Verify channel count: waveform.shape[0]","Match model architecture to data channels"]},{"slug":"dns-resolution-failure","title":"DNS Resolution Failure","category":"Infrastructure","text":"DNS Resolution Failure Infrastructure gaierror: [Errno -2] Name or service not known Connection refused DNS server unreachable dns hostname network infrastructure resolution Cannot resolve hostname Connection to master fails HF download fails with DNS error DNS server not configured in container Cluster DNS not reachable DNS cache corrupted IPv6 vs IPv4 DNS issue Corporate DNS blocking external lookups","anchorText":"DNS Resolution Failure gaierror: [Errno -2] Name or service not known Connection refused DNS server unreachable dns hostname network infrastructure resolution DNS server not configured in container Cluster DNS not reachable DNS cache corrupted IPv6 vs IPv4 DNS issue Corporate DNS blocking external lookups ","action":"Set DNS in container: --dns=8.8.8.8","steps":["Set DNS in container: --dns=8.8.8.8","Use /etc/resolv.conf with proper nameservers","Use IP addresses instead of hostnames","Configure corporate DNS forwarding","Test DNS: nslookup huggingface.co"]},{"slug":"activations-checkpoint-compatibility","title":"Activation Checkpointing Compatibility","category":"Memory","text":"Activation Checkpointing Compatibility Memory use_reentrant=True breaks torch.compile Checkpointing not applied to all layers Checkpointing wrong layer type activation-checkpointing torch-compile fsdp ddp memory Activation checkpointing doesn't work with torch.compile Checkpointing causes errors with FSDP Checkpointing fails with DDP use_reentrant=True deprecated with torch.compile Checkpointing on non-compat layers torch.utils.checkpoint vs FSDP checkpoint Activation recomputation in distributed training","anchorText":"Activation Checkpointing Compatibility use_reentrant=True breaks torch.compile Checkpointing not applied to all layers Checkpointing wrong layer type activation-checkpointing torch-compile fsdp ddp memory use_reentrant=True deprecated with torch.compile Checkpointing on non-compat layers torch.utils.checkpoint vs FSDP checkpoint Activation recomputation in distributed training ","action":"Use use_reentrant=False for torch.compile compatibility","steps":["Use use_reentrant=False for torch.compile compatibility","Apply checkpointing only to transformer blocks","Use FSDP's apply_activation_checkpointing","Test with simple model first","Use torch.utils.checkpoint.checkpoint"]},{"slug":"timezone-utc-mismatch","title":"Timezone / UTC Mismatch","category":"Environment","text":"Timezone / UTC Mismatch Environment Time skew between nodes Logs out of order by time Job starts at unexpected time timezone utc time environment logging Logs from different nodes have different times Scheduled job runs at wrong time Time-based metrics are wrong Different timezones on different nodes Container timezone differs from host Cron jobs not in UTC DST changes affect schedules","anchorText":"Timezone / UTC Mismatch Time skew between nodes Logs out of order by time Job starts at unexpected time timezone utc time environment logging Different timezones on different nodes Container timezone differs from host Cron jobs not in UTC DST changes affect schedules ","action":"Use UTC everywhere: export TZ=UTC","steps":["Use UTC everywhere: export TZ=UTC","Set timezone in container: TZ=UTC","Use NTP for time sync","Log timestamps in UTC","Use time.monotonic() for relative timing"]},{"slug":"init-seed-mismatch","title":"Init Seed Mismatch","category":"Training Stability","text":"Init Seed Mismatch Training Stability Different loss curves on re-run Different accuracy on re-run Distributed training non-deterministic seed reproducibility init training-stability determinism Same code gives different results Cannot reproduce previous training Hyperparameter search noisy Random seed not set Seed set but not for all RNG sources CUDA non-deterministic operations DataLoader workers have different seeds Distributed training not synced","anchorText":"Init Seed Mismatch Different loss curves on re-run Different accuracy on re-run Distributed training non-deterministic seed reproducibility init training-stability determinism Random seed not set Seed set but not for all RNG sources CUDA non-deterministic operations DataLoader workers have different seeds Distributed training not synced ","action":"Set all seeds: torch, numpy, random, cuda","steps":["Set all seeds: torch, numpy, random, cuda","Use torch.manual_seed(42) and torch.cuda.manual_seed_all(42)","Set deterministic algorithms: torch.use_deterministic_algorithms(True)","Set worker_init_fn in DataLoader","Use DistributedSampler with same seed"]},{"slug":"text-encoding-mismatch","title":"Text Encoding Mismatch","category":"Data Pipeline","text":"Text Encoding Mismatch Data Pipeline UnicodeDecodeError: 'utf-8' codec can't decode byte Text is mojibake (wrong encoding) Special characters lost text-encoding utf-8 unicode data-pipeline nlp UnicodeDecodeError when loading text Garbled text in training data Tokenizer produces wrong tokens for special chars File saved in different encoding (Latin-1, Windows-1252) CSV has BOM (byte order mark) Web scraping returns wrong encoding Mixed encodings in dataset","anchorText":"Text Encoding Mismatch UnicodeDecodeError: 'utf-8' codec can't decode byte Text is mojibake (wrong encoding) Special characters lost text-encoding utf-8 unicode data-pipeline nlp File saved in different encoding (Latin-1, Windows-1252) CSV has BOM (byte order mark) Web scraping returns wrong encoding Mixed encodings in dataset ","action":"Specify encoding when opening: open(file, encoding='utf-8')","steps":["Specify encoding when opening: open(file, encoding='utf-8')","Use chardet to detect encoding: chardet.detect()","Strip BOM: open(file, encoding='utf-8-sig')","Convert all to UTF-8: iconv -f LATIN1 -t UTF-8","Use pandas with encoding parameter"]},{"slug":"all-reduce-deadlock","title":"All-Reduce Deadlock","category":"Communication","text":"All-Reduce Deadlock Communication Training hangs at backward pass DDP all-reduce hangs FSDP all-gather hangs all-reduce deadlock ddp distributed communication Distributed training hangs at all-reduce All-reduce never completes Deadlock with no error message Different number of parameters across ranks Batch norm sync missing All-reduce in only some ranks Inconsistent model state across ranks Distributed all-reduce timeout not set","anchorText":"All-Reduce Deadlock Training hangs at backward pass DDP all-reduce hangs FSDP all-gather hangs all-reduce deadlock ddp distributed communication Different number of parameters across ranks Batch norm sync missing All-reduce in only some ranks Inconsistent model state across ranks Distributed all-reduce timeout not set ","action":"Verify all ranks have same model: print model.parameters()","steps":["Verify all ranks have same model: print model.parameters()","Use SyncBatchNorm for batch norm: nn.SyncBatchNorm.convert_sync_batchnorm()","Set timeout: dist.init_process_group(timeout=...)","Check model parameter count matches across ranks","Use NCCL_DEBUG=INFO for debugging"]},{"slug":"training-resume-failure","title":"Training Resume Failure","category":"Reliability","text":"Training Resume Failure Reliability RuntimeError: Error(s) in loading state_dict strict=True fails on load Model architecture changed since save resume checkpoint loading state-dict reliability Cannot resume training from checkpoint Missing or unexpected keys in state_dict Optimizer state not loaded correctly strict=True mismatch in load_state_dict Optimizer state has different keys LR scheduler state missing Random state not restored Model architecture changed","anchorText":"Training Resume Failure RuntimeError: Error(s) in loading state_dict strict=True fails on load Model architecture changed since save resume checkpoint loading state-dict reliability strict=True mismatch in load_state_dict Optimizer state has different keys LR scheduler state missing Random state not restored Model architecture changed ","action":"Use strict=False for partial loading: load_state_dict(strict=False)","steps":["Use strict=False for partial loading: load_state_dict(strict=False)","Load optimizer state separately","Restore LR scheduler state","Restore random state for reproducibility","Document checkpoint contents"]},{"slug":"ddp-hang-at-end","title":"DDP Hang at Epoch Boundary","category":"Distributed Training","text":"DDP Hang at Epoch Boundary Distributed Training DDP forward pass hangs after last batch Some ranks report end of epoch while others are processing NCCL timeout at the end of the last batch ddp uneven data distributed epoch hang DDP forward pass hangs at epoch transition Some ranks finish the dataset before others Training stalls between epochs One rank exhausts its dataset partition before others DDP requires all ranks to call the same number of backward passes Drop_last=False causes epoch boundary mismatch","anchorText":"DDP Hang at Epoch Boundary DDP forward pass hangs after last batch Some ranks report end of epoch while others are processing NCCL timeout at the end of the last batch ddp uneven data distributed epoch hang One rank exhausts its dataset partition before others DDP requires all ranks to call the same number of backward passes Drop_last=False causes epoch boundary mismatch ","action":"Set DataLoader(drop_last=True) to drop incomplete batches in DDP","steps":["Set DataLoader(drop_last=True) to drop incomplete batches in DDP","Ensure all dataset shards have the same size","Set find_unused_parameters=True in DDP for variable-length training","Use distributed sampler that ensures even partition"]},{"slug":"ddp-rank-stuck","title":"DDP Rank Stuck During Training","category":"Distributed Training","text":"DDP Rank Stuck During Training Distributed Training One rank takes 10x longer than others GPU utilization varies wildly across ranks Training throughput plateaus despite more compute ddp straggler rank distributed performance stall Training stalls with no error One rank has zero GPU utilization All ranks show same step but no progress Straggler rank has hardware degradation Network between straggler rank and others is slower Straggler rank has contention with other workloads CUDA error on one rank not propagated to others","anchorText":"DDP Rank Stuck During Training One rank takes 10x longer than others GPU utilization varies wildly across ranks Training throughput plateaus despite more compute ddp straggler rank distributed performance stall Straggler rank has hardware degradation Network between straggler rank and others is slower Straggler rank has contention with other workloads CUDA error on one rank not propagated to others ","action":"Identify straggler rank with NCCL_DEBUG=INFO","steps":["Identify straggler rank with NCCL_DEBUG=INFO","Drain straggler node: scontrol update state=drain","Exclude straggler rank from job","Use elastic training to handle stragglers"]},{"slug":"fsdp-flat-param","title":"FSDP Flat Parameter Error","category":"Distributed Training","text":"FSDP Flat Parameter Error Distributed Training FSDP flat_param key not found in state dict FSDP parameter unsharding failed: flat param is not available _key in _flat_param state doesn't match parameter fsdp flat-param sharding checkpoint distributed FSDP wrapping fails with flat parameter issues Full state dict recovery has unexpected keys Resharding after forward pass fails FSDP internal flat parameter management has version mismatch CPU offload and onload cycle corrupts flat param metadata Mixed precision creates flat param inconsistency","anchorText":"FSDP Flat Parameter Error FSDP flat_param key not found in state dict FSDP parameter unsharding failed: flat param is not available _key in _flat_param state doesn't match parameter fsdp flat-param sharding checkpoint distributed FSDP internal flat parameter management has version mismatch CPU offload and onload cycle corrupts flat param metadata Mixed precision creates flat param inconsistency ","action":"Save with full_state_dict=True to avoid flat param issues","steps":["Save with full_state_dict=True to avoid flat param issues","Load with strict=False to handle missing flat param keys","Wrap entire model in single FSDP unit for simplicity","Use HuggingFace AutoModel for FSDP-compatible saving"]},{"slug":"fsdp-mixed-precision","title":"FSDP Mixed Precision Error","category":"Distributed Training","text":"FSDP Mixed Precision Error Distributed Training Loss NaN with FSDP + AMP but not with either alone FSDP mixed precision config mismatch between layers Gradient scaler overflow with FSDP enabled fsdp mixed-precision amp bf16 distributed numerical Loss becomes NaN after FSDP wrapping Mixed precision training diverges with FSDP FSDP and AMP interaction produces numerical issues FSDP re-shards parameters at different precision than forward pass Autocast context interacts incorrectly with FSDP parameter access Gradient accumulation with FSDP causes precision drift","anchorText":"FSDP Mixed Precision Error Loss NaN with FSDP + AMP but not with either alone FSDP mixed precision config mismatch between layers Gradient scaler overflow with FSDP enabled fsdp mixed-precision amp bf16 distributed numerical FSDP re-shards parameters at different precision than forward pass Autocast context interacts incorrectly with FSDP parameter access Gradient accumulation with FSDP causes precision drift ","action":"Set FSDP mixed_precision policy consistently across all layers","steps":["Set FSDP mixed_precision policy consistently across all layers","Use bf16 instead of fp16 for FSDP on H100/A100","Set MixedPrecision(param_dtype=bf16, reduce_dtype=bf16, buffer_dtype=bf16)","Wrap autocast around FSDP forward call explicitly"]},{"slug":"tensor-parallel-error","title":"Tensor Parallel Error","category":"Distributed Training","text":"Tensor Parallel Error Distributed Training Dimension mismatch in tensor parallel split Tensor parallel weight size mismatch AllReduce between tensor parallel ranks failed tensor-parallel tp model-parallelism distributed megatron Multi-GPU inference with tensor parallelism fails Tensor parallel split dimensions don't match model config TP with PP combined fails Model hidden dimensions not divisible by tensor parallel size TP communication group setup failed Weight initialization with TP splitting is inconsistent TP + PP schedule mismatch causes deadlock","anchorText":"Tensor Parallel Error Dimension mismatch in tensor parallel split Tensor parallel weight size mismatch AllReduce between tensor parallel ranks failed tensor-parallel tp model-parallelism distributed megatron Model hidden dimensions not divisible by tensor parallel size TP communication group setup failed Weight initialization with TP splitting is inconsistent TP + PP schedule mismatch causes deadlock ","action":"Set TP size to divide model hidden_dim evenly","steps":["Set TP size to divide model hidden_dim evenly","Verify TP communication world group initialization","Reduce TP degree if dimension mismatch persists","Check PP schedule matches TP configuration"]},{"slug":"deepspeed-init-failed","title":"DeepSpeed Initialization Failed","category":"Distributed Training","text":"DeepSpeed Initialization Failed Distributed Training DeepSpeed configuration validation error DeepSpeed: model not compatible with ZeRO config RuntimeError: DeepSpeed engine initialization failed deepspeed init config zero distributed engine Training fails at DeepSpeed engine init DeepSpeed config rejected Model not compatible with DeepSpeed config Invalid config value for parameter Incompatible config with model architecture DeepSpeed version doesn't support config option Config file has syntax error","anchorText":"DeepSpeed Initialization Failed DeepSpeed configuration validation error DeepSpeed: model not compatible with ZeRO config RuntimeError: DeepSpeed engine initialization failed deepspeed init config zero distributed engine Invalid config value for parameter Incompatible config with model architecture DeepSpeed version doesn't support config option Config file has syntax error ","action":"Validate config with deepspeed --validate_config","steps":["Validate config with deepspeed --validate_config","Check DeepSpeed documentation for supported options","Match config to model architecture","Use DeepSpeed config examples as templates"]},{"slug":"checkpoint-torn-write","title":"Checkpoint Torn Write","category":"Data Integrity","text":"Checkpoint Torn Write Data Integrity Checkpoint file is smaller than expected UnpicklingError when loading checkpoint torch.load fails halfway through checkpoint torn-write save atomic corruption storage Checkpoint file exists but is smaller than expected Resume from checkpoint crashes with pickle error File size differs between saves of same model Job was preempted or killed during checkpoint write Disk full during checkpoint write NFS timeout during checkpoint write Multiple processes writing to same checkpoint file","anchorText":"Checkpoint Torn Write Checkpoint file is smaller than expected UnpicklingError when loading checkpoint torch.load fails halfway through checkpoint torn-write save atomic corruption storage Job was preempted or killed during checkpoint write Disk full during checkpoint write NFS timeout during checkpoint write Multiple processes writing to same checkpoint file ","action":"Write to temporary file first, then rename to checkpoint name","steps":["Write to temporary file first, then rename to checkpoint name","Use atomic writes: save to .tmp, then os.rename()","Check file size before loading: expected vs actual","Keep only valid checkpoints based on size verification"]},{"slug":"dataset-pii-leakage","title":"PII Leakage in Training Data","category":"Data Integrity","text":"PII Leakage in Training Data Data Integrity Model outputs contain names, emails, or phone numbers PII detection tools flag model outputs Privacy audit reveals data leakage pii privacy leakage compliance gdpr ccpa training data Model outputs include personal information PII detected in model outputs Compliance review flags PII leakage Training data contains PII (names, emails, addresses) Model memorizes and reproduces PII Large models with memorization capability","anchorText":"PII Leakage in Training Data Model outputs contain names, emails, or phone numbers PII detection tools flag model outputs Privacy audit reveals data leakage pii privacy leakage compliance gdpr ccpa training data Training data contains PII (names, emails, addresses) Model memorizes and reproduces PII Large models with memorization capability ","action":"Remove PII from training data using regex/NER","steps":["Remove PII from training data using regex/NER","Apply differential privacy during training","Use PII detection on training data","Filter PII from model outputs in production"]},{"slug":"dataset-license-issue","title":"Dataset License Issue","category":"Data Integrity","text":"Dataset License Issue Data Integrity Dataset license: CC-BY-NC, GPL, custom restrictions Cannot redistribute model if trained on this data Legal review of training data required dataset license commercial compliance legal training data Concerns about dataset license for production use License compatibility issues Uncertainty about commercial use rights Dataset license has non-commercial (NC) clause Dataset license is GPL (copyleft) Dataset license unclear or custom Training data includes copyrighted material","anchorText":"Dataset License Issue Dataset license: CC-BY-NC, GPL, custom restrictions Cannot redistribute model if trained on this data Legal review of training data required dataset license commercial compliance legal training data Dataset license has non-commercial (NC) clause Dataset license is GPL (copyleft) Dataset license unclear or custom Training data includes copyrighted material ","action":"Review dataset license before training for production","steps":["Review dataset license before training for production","Use permissively licensed datasets (CC0, Apache 2.0) for commercial","Add license compliance check to training pipeline","Document dataset licenses for model release"]},{"slug":"ray-task-timeout","title":"Ray Task Timeout","category":"Infrastructure","text":"Ray Task Timeout Infrastructure Ray task timeout: task took longer than X seconds TimeoutExpired: task did not complete in time Ray actor becomes unresponsive ray task-timeout infrastructure fault-tolerance distributed timeout Ray task fails with timeout error Some tasks complete faster than others Performance degrades over time Task takes longer than configured timeout Worker is overloaded Network issues between task and result GPU contention between tasks","anchorText":"Ray Task Timeout Ray task timeout: task took longer than X seconds TimeoutExpired: task did not complete in time Ray actor becomes unresponsive ray task-timeout infrastructure fault-tolerance distributed timeout Task takes longer than configured timeout Worker is overloaded Network issues between task and result GPU contention between tasks ","action":"Increase task timeout: ray.get(task, timeout=600)","steps":["Increase task timeout: ray.get(task, timeout=600)","Reduce task size or batch size","Check worker health","Optimize task scheduling"]},{"slug":"ray-dataset-oom","title":"Ray Dataset Out of Memory","category":"Infrastructure","text":"Ray Dataset Out of Memory Infrastructure Ray actor died with OOM during data loading Object store memory exhausted Worker OOM during dataset transformation ray-dataset oom infrastructure object-store memory streaming data Ray Dataset pipeline fails with OOM Workers crash with OOM during data loading Training stalls waiting for data Object store memory exhausted by dataset blocks Worker memory exhausted during transformation Too many blocks cached in object store","anchorText":"Ray Dataset Out of Memory Ray actor died with OOM during data loading Object store memory exhausted Worker OOM during dataset transformation ray-dataset oom infrastructure object-store memory streaming data Object store memory exhausted by dataset blocks Worker memory exhausted during transformation Too many blocks cached in object store ","action":"Increase object_store_memory: ray.init(object_store_memory=20e9)","steps":["Increase object_store_memory: ray.init(object_store_memory=20e9)","Reduce block size for streaming","Use fewer workers","Process data in smaller batches"]},{"slug":"docker-gpu-error","title":"Docker GPU Passthrough Error","category":"Environment","text":"Docker GPU Passthrough Error Environment docker: Error response from daemon: could not select device driver with capabilities: gpu nvidia-smi: command not found inside container torch.cuda.is_available() returns False in container but True on host docker container gpu passthrough nvidia environment Docker container doesn't see GPUs nvidia-smi inside container fails PyTorch reports CUDA not available inside container NVIDIA Container Toolkit not installed --gpus flag not passed to docker run nvidia-docker2 not installed Container runtime not set to nvidia","anchorText":"Docker GPU Passthrough Error docker: Error response from daemon: could not select device driver with capabilities: gpu nvidia-smi: command not found inside container torch.cuda.is_available() returns False in container but True on host docker container gpu passthrough nvidia environment NVIDIA Container Toolkit not installed --gpus flag not passed to docker run nvidia-docker2 not installed Container runtime not set to nvidia ","action":"Install NVIDIA Container Toolkit: sudo apt install nvidia-container-toolkit","steps":["Install NVIDIA Container Toolkit: sudo apt install nvidia-container-toolkit","Use --gpus all flag: docker run --gpus all","Set default runtime: /etc/docker/daemon.json with nvidia runtime","Verify inside container: nvidia-smi"]},{"slug":"docker-permission-error","title":"Docker Permission Error","category":"Environment","text":"Docker Permission Error Environment Permission denied: /dev/nvidia0 docker.sock permission denied EACCES: permission denied docker container permissions gpu nvidia environment Container can't access GPU File permission denied in container Network access denied Container user not in docker group Container user not in video group SELinux blocking device access Filesystem mounted with restrictive permissions","anchorText":"Docker Permission Error Permission denied: /dev/nvidia0 docker.sock permission denied EACCES: permission denied docker container permissions gpu nvidia environment Container user not in docker group Container user not in video group SELinux blocking device access Filesystem mounted with restrictive permissions ","action":"Add user to docker group: sudo usermod -aG docker $USER","steps":["Add user to docker group: sudo usermod -aG docker $USER","Add user to video group for GPU access","Use --privileged flag (security risk)","Set proper file permissions on mounted volumes"]},{"slug":"nfs-stall","title":"NFS / Network Filesystem Stall","category":"Infrastructure","text":"NFS / Network Filesystem Stall Infrastructure File operations hang at NFS mount points ls /mnt/checkpoints hangs indefinitely kernel: NFS: server not responding, still trying Training stalls at checkpoint save or log write nfs network storage stall hang filesystem infrastructure Checkpoint operations hang indefinitely File reads and writes timeout Training appears frozen but process is alive NFS server overloaded with training writes Network congestion between compute and storage nodes NFS namespace conflict from concurrent writers NFS server crash or restart","anchorText":"NFS / Network Filesystem Stall File operations hang at NFS mount points ls /mnt/checkpoints hangs indefinitely kernel: NFS: server not responding, still trying Training stalls at checkpoint save or log write nfs network storage stall hang filesystem infrastructure NFS server overloaded with training writes Network congestion between compute and storage nodes NFS namespace conflict from concurrent writers NFS server crash or restart ","action":"Switch checkpoint write to local NVMe and async copy to NFS","steps":["Switch checkpoint write to local NVMe and async copy to NFS","Use storage-specific tools: s5cmd for S3, gcloud storage for GCS","Check NFS server load and network latency","Mount NFS with nointr and hard options to avoid data corruption"]},{"slug":"cloud-storage-throttling","title":"Cloud Storage Throttling","category":"Infrastructure","text":"Cloud Storage Throttling Infrastructure S3: SlowDown error GCS: 429 rateLimitExceeded Azure: ServerBusy cloud-storage throttling rate-limit infrastructure s3 gcs azure storage Training slows down over time Checkpoint saves take progressively longer Cloud storage API returns 429 (rate limit) Too many concurrent requests to storage Storage service quota exceeded Burst capacity exhausted","anchorText":"Cloud Storage Throttling S3: SlowDown error GCS: 429 rateLimitExceeded Azure: ServerBusy cloud-storage throttling rate-limit infrastructure s3 gcs azure storage Too many concurrent requests to storage Storage service quota exceeded Burst capacity exhausted ","action":"Reduce checkpoint frequency","steps":["Reduce checkpoint frequency","Use batch uploads","Implement exponential backoff retry","Use multiple storage buckets for parallelism"]},{"slug":"slurm-node-oom","title":"SLURM Node Out of Memory","category":"Infrastructure","text":"SLURM Node Out of Memory Infrastructure slurmstepd: error: job X killed by OOM Job exceeded memory limit Dmesg shows OOM killer slurm node oom memory scheduler infrastructure Job is killed before completion Sacct shows OOM in job state Job exits with non-zero status Memory request too low for training workload Memory leak during training Other processes on node consuming memory","anchorText":"SLURM Node Out of Memory slurmstepd: error: job X killed by OOM Job exceeded memory limit Dmesg shows OOM killer slurm node oom memory scheduler infrastructure Memory request too low for training workload Memory leak during training Other processes on node consuming memory ","action":"Request more memory: #SBATCH --mem=256G","steps":["Request more memory: #SBATCH --mem=256G","Use memory profiling to estimate requirements","Check actual memory usage with sacct","Add memory monitoring with Denpex"]},{"slug":"tokenization-error","title":"Tokenization Error","category":"Data Pipeline","text":"Tokenization Error Data Pipeline TokenizersError: tokenizer raised an error ValueError: Tokenizer class mismatch UnicodeDecodeError when tokenizing text KeyError: token not in vocabulary tokenization text encoding nlp huggingface preprocessing data Tokenization fails for specific text samples Tokenizer mismatch with model Unicode errors during tokenization Invalid UTF-8 in text data Token characters not in vocabulary Tokenizer saved with different version Batch with mixed encodings","anchorText":"Tokenization Error TokenizersError: tokenizer raised an error ValueError: Tokenizer class mismatch UnicodeDecodeError when tokenizing text KeyError: token not in vocabulary tokenization text encoding nlp huggingface preprocessing data Invalid UTF-8 in text data Token characters not in vocabulary Tokenizer saved with different version Batch with mixed encodings ","action":"Skip or fix problematic text samples","steps":["Skip or fix problematic text samples","Use tokenizer with handle_invalid_chars=True","Validate text encoding before tokenization","Update tokenizer to handle special characters"]},{"slug":"huggingface-tokenizer-error","title":"HuggingFace Tokenizer Error","category":"Data Pipeline","text":"HuggingFace Tokenizer Error Data Pipeline TokenizersError: Tokenizer raised an error during tokenization ValueError: Tokenizer class XX does not match class YY OSError: Can't load tokenizer for model-name UnicodeDecodeError when tokenizing a specific batch tokenizer huggingface transformers data-pipeline nlp encoding Training crashes with Tokenizer error when loading or processing text data Different datasets fail at different tokenization steps Previously working dataset suddenly fails after transformers version update Tokenizer configuration doesn't match the dataset's actual text length or format Tokenizer saved with different transformers version Corrupted tokenizer files in HuggingFace cache Specific sample has invalid UTF-8 encoding","anchorText":"HuggingFace Tokenizer Error TokenizersError: Tokenizer raised an error during tokenization ValueError: Tokenizer class XX does not match class YY OSError: Can't load tokenizer for model-name UnicodeDecodeError when tokenizing a specific batch tokenizer huggingface transformers data-pipeline nlp encoding Tokenizer configuration doesn't match the dataset's actual text length or format Tokenizer saved with different transformers version Corrupted tokenizer files in HuggingFace cache Specific sample has invalid UTF-8 encoding ","action":"Clear the tokenizer cache: rm -rf ~/.cache/huggingface/tokenizers/ and reload","steps":["Clear the tokenizer cache: rm -rf ~/.cache/huggingface/tokenizers/ and reload","Verify tokenizer loading: AutoTokenizer.from_pretrained(name, use_fast=True)","Set truncation=True and padding=True in tokenizer call","Add exception handling around tokenization","Use Denpex bad tokenization detection"]},{"slug":"augmentation-error","title":"Data Augmentation Error","category":"Data Pipeline","text":"Data Augmentation Error Data Pipeline Image conversion failed: cannot identify image file OpenCV error: assertion failed albumentations: ValueError: image must be 3-dimensional Augmentation produces NaN values augmentation image preprocessing data pipeline albumentations opencv Augmentation fails for specific samples Augmentation produces empty or corrupted images Training crashes with augmentation-related error Corrupted image file in dataset Augmentation library version conflict Unsupported image format Augmentation parameters out of range","anchorText":"Data Augmentation Error Image conversion failed: cannot identify image file OpenCV error: assertion failed albumentations: ValueError: image must be 3-dimensional Augmentation produces NaN values augmentation image preprocessing data pipeline albumentations opencv Corrupted image file in dataset Augmentation library version conflict Unsupported image format Augmentation parameters out of range ","action":"Add try/except in augmentation to skip bad samples","steps":["Add try/except in augmentation to skip bad samples","Validate image format before augmentation","Update augmentation library","Check for NaN values after augmentation"]},{"slug":"nvlink-error","title":"NVLink Error","category":"Hardware","text":"NVLink Error Hardware dmesg: nvlink: link X is down nvidia-smi nvlink --link shows link down NCCL warnings about P2P link issues NCCL falls back to slower PCIe for intra-node communication nvlink gpu hardware p2p communication intra-node nvidia P2P communication between GPUs on same node fails NCCL performance is poor on multi-GPU nodes NVLink link is down for some GPU pairs NVLink hardware fault Loose NVLink connector NVLink firmware mismatch GPU seating issue","anchorText":"NVLink Error dmesg: nvlink: link X is down nvidia-smi nvlink --link shows link down NCCL warnings about P2P link issues NCCL falls back to slower PCIe for intra-node communication nvlink gpu hardware p2p communication intra-node nvidia NVLink hardware fault Loose NVLink connector NVLink firmware mismatch GPU seating issue ","action":"Check NVLink status: nvidia-smi nvlink --link","steps":["Check NVLink status: nvidia-smi nvlink --link","Reseat GPUs and NVLink connectors","Update NVLink firmware","If persistent, drain node and check hardware"]},{"slug":"gpu-thermal-throttle","title":"GPU Thermal Throttling","category":"Hardware","text":"GPU Thermal Throttling Hardware nvidia-smi shows clocks dropped to base nvidia-smi -q shows thermal throttling reason Performance varies with ambient temperature thermal throttle cooling hardware gpu performance temperature Training is slower than expected GPU clock speeds vary during training Performance drops after sustained training GPU cooler failure or dust accumulation Insufficient airflow in server High ambient temperature GPU power limit reached causing thermal throttle","anchorText":"GPU Thermal Throttling nvidia-smi shows clocks dropped to base nvidia-smi -q shows thermal throttling reason Performance varies with ambient temperature thermal throttle cooling hardware gpu performance temperature GPU cooler failure or dust accumulation Insufficient airflow in server High ambient temperature GPU power limit reached causing thermal throttle ","action":"Improve cooling: clean dust, add airflow","steps":["Improve cooling: clean dust, add airflow","Reduce GPU power limit: nvidia-smi -pl 300","Check cooler functionality","Use liquid cooling for high-density deployments"]},{"slug":"gpu-power-cap","title":"GPU Power Cap Reached","category":"Hardware","text":"GPU Power Cap Reached Hardware nvidia-smi shows power limit reached Clock speeds vary with workload Power consumption capped at configured limit power cap limit gpu hardware power-consumption cloud Training is slower than expected GPU clock speeds vary with workload Performance inconsistent across jobs Cloud provider caps GPU power Power infrastructure limit Energy efficiency requirements","anchorText":"GPU Power Cap Reached nvidia-smi shows power limit reached Clock speeds vary with workload Power consumption capped at configured limit power cap limit gpu hardware power-consumption cloud Cloud provider caps GPU power Power infrastructure limit Energy efficiency requirements ","action":"Request higher power cap from cloud provider","steps":["Request higher power cap from cloud provider","Use lower-power GPU type","Optimize model for lower power consumption","Monitor power usage with Denpex"]},{"slug":"cuda-driver-crash","title":"CUDA Driver Crash","category":"Hardware","text":"CUDA Driver Crash Hardware NVRM: Xid error in dmesg nvidia-smi: command not found after crash All CUDA processes killed System may become unstable cuda driver crash gpu hardware nvidia kernel All GPU processes terminate simultaneously nvidia-smi becomes unresponsive System logs show NVIDIA driver error Hardware fault triggers driver crash Driver bug causes kernel panic GPU overheating causes driver reset Overcurrent protection trips","anchorText":"CUDA Driver Crash NVRM: Xid error in dmesg nvidia-smi: command not found after crash All CUDA processes killed System may become unstable cuda driver crash gpu hardware nvidia kernel Hardware fault triggers driver crash Driver bug causes kernel panic GPU overheating causes driver reset Overcurrent protection trips ","action":"Check dmesg for driver error details","steps":["Check dmesg for driver error details","Reboot node to recover","Re-seat GPU and check power cables","Update NVIDIA driver to latest version"]},{"slug":"gpu-overheating","title":"GPU Overheating","category":"Hardware","text":"GPU Overheating Hardware nvidia-smi shows temperature above 85C nvidia-smi -q shows thermal throttling Performance drops after sustained training overheating thermal throttle cooling hardware gpu temperature GPU clock speeds drop during training Performance varies with ambient temperature GPU fans run at maximum speed GPU cooler failure Dust accumulation on heatsink Insufficient airflow in server High ambient temperature GPU power limit causing thermal throttle","anchorText":"GPU Overheating nvidia-smi shows temperature above 85C nvidia-smi -q shows thermal throttling Performance drops after sustained training overheating thermal throttle cooling hardware gpu temperature GPU cooler failure Dust accumulation on heatsink Insufficient airflow in server High ambient temperature GPU power limit causing thermal throttle ","action":"Improve cooling: clean dust, add airflow","steps":["Improve cooling: clean dust, add airflow","Reduce GPU power limit: nvidia-smi -pl 300","Replace thermal paste if old","Use liquid cooling for high-density deployments"]},{"slug":"gpu-not-detected","title":"GPU Not Detected","category":"Hardware","text":"GPU Not Detected Hardware CUDA unavailable RuntimeError: Found no NVIDIA driver on your system nvidia-smi: command not found gpu not-detected driver hardware nvidia detection torch.cuda.is_available() returns False nvidia-smi shows no GPUs GPU disappeared after working previously NVIDIA driver not installed Driver not loaded after kernel update GPU not seated properly GPU hardware failure PCIe slot failure","anchorText":"GPU Not Detected CUDA unavailable RuntimeError: Found no NVIDIA driver on your system nvidia-smi: command not found gpu not-detected driver hardware nvidia detection NVIDIA driver not installed Driver not loaded after kernel update GPU not seated properly GPU hardware failure PCIe slot failure ","action":"Install NVIDIA driver: sudo apt install nvidia-driver-XXX","steps":["Install NVIDIA driver: sudo apt install nvidia-driver-XXX","Reload driver: sudo modprobe nvidia","Re-seat GPU and check PCIe connection","Reboot after driver installation","Check dmesg for GPU detection errors"]},{"slug":"cuda-uncorrectable-ecc","title":"CUDA Uncorrectable ECC Error","category":"Hardware","text":"CUDA Uncorrectable ECC Error Hardware RuntimeError: CUDA error: uncorrectable ECC Xid 48 in dmesg nvidia-smi -q -d ECC shows uncorrectable errors Loss values are NaN or inf ecc uncorrectable memory hardware gpu xid-48 error-correction Training crashes with CUDA ECC error Single-bit errors accumulate over time Model produces corrupted outputs GPU memory cells degrade over time Hardware fault in memory chips ECC error exceeds correctable threshold Manufacturing defect in GPU memory","anchorText":"CUDA Uncorrectable ECC Error RuntimeError: CUDA error: uncorrectable ECC Xid 48 in dmesg nvidia-smi -q -d ECC shows uncorrectable errors Loss values are NaN or inf ecc uncorrectable memory hardware gpu xid-48 error-correction GPU memory cells degrade over time Hardware fault in memory chips ECC error exceeds correctable threshold Manufacturing defect in GPU memory ","action":"Identify faulty GPU: nvidia-smi -q -d ECC","steps":["Identify faulty GPU: nvidia-smi -q -d ECC","Drain faulty GPU: scontrol update state=drain","Request GPU replacement","Verify model integrity after GPU replacement","Use Denpex to auto-detect and drain faulty GPUs"]},{"slug":"python-multiprocessing-fork","title":"Python Multiprocessing Fork Issue","category":"Data Pipeline","text":"Python Multiprocessing Fork Issue Data Pipeline RuntimeError: can't pickle multiprocessing objects Deadlock after multiprocessing.fork() Workers all hang on first iteration python multiprocessing fork worker dataloader pipeline Workers deadlock after fork Workers share unexpected state Fork fails with runtime error Forked workers inherit parent state including CUDA context Fork after CUDA initialization causes deadlock Locks from parent are not released in children Open files in parent are shared incorrectly","anchorText":"Python Multiprocessing Fork Issue RuntimeError: can't pickle multiprocessing objects Deadlock after multiprocessing.fork() Workers all hang on first iteration python multiprocessing fork worker dataloader pipeline Forked workers inherit parent state including CUDA context Fork after CUDA initialization causes deadlock Locks from parent are not released in children Open files in parent are shared incorrectly ","action":"Use DataLoader with num_workers > 0 and proper fork handling","steps":["Use DataLoader with num_workers > 0 and proper fork handling","Set multiprocessing start method to 'spawn' for CUDA safety","Move CUDA initialization to after fork","Use torch.multiprocessing instead of Python's multiprocessing"]},{"slug":"python-oom","title":"Python Out of Memory","category":"Memory","text":"Python Out of Memory Memory MemoryError: out of memory OSError: [Errno 12] Cannot allocate memory Process killed by OOM killer in dmesg python oom memory heap dataloader pipeline Training crashes with MemoryError Python process is killed by OOM killer Swap usage reaches maximum Accumulating large Python objects (lists, dicts) in memory Large dataset cached in Python heap Memory leak in custom code Too many parallel Python processes","anchorText":"Python Out of Memory MemoryError: out of memory OSError: [Errno 12] Cannot allocate memory Process killed by OOM killer in dmesg python oom memory heap dataloader pipeline Accumulating large Python objects (lists, dicts) in memory Large dataset cached in Python heap Memory leak in custom code Too many parallel Python processes ","action":"Use generators and iterators to avoid loading everything into memory","steps":["Use generators and iterators to avoid loading everything into memory","Stream metrics to disk instead of keeping in memory","Profile Python memory with tracemalloc","Reduce dataset caching in Python heap","Monitor Python memory with resource.getrusage"]},{"slug":"python-import-error","title":"Python Import Error","category":"Environment","text":"Python Import Error Environment ModuleNotFoundError: No module named torch ImportError: cannot import name from torch ImportError: DLL load failed Python crashes with import error python import module environment dependency setup Training script fails at import time ModuleNotFoundError ImportError: cannot import name Module not installed Module path not in PYTHONPATH Circular import Conflicting module versions Missing __init__.py","anchorText":"Python Import Error ModuleNotFoundError: No module named torch ImportError: cannot import name from torch ImportError: DLL load failed Python crashes with import error python import module environment dependency setup Module not installed Module path not in PYTHONPATH Circular import Conflicting module versions Missing __init__.py ","action":"Install missing module: pip install module_name","steps":["Install missing module: pip install module_name","Check Python path: sys.path","Fix circular imports by restructuring","Use virtual environments to isolate dependencies","Check for conflicting module versions"]},{"slug":"cuda-illegal-instruction","title":"CUDA Illegal Instruction","category":"Hardware","text":"CUDA Illegal Instruction Hardware CUDA error: an illegal instruction was encountered RuntimeError: CUDA error: illegal instruction GPU compute capability mismatch cuda illegal-instruction gpu hardware compute-capability nvidia Training crashes with CUDA illegal instruction Error points to specific CUDA instruction GPU supports older compute capability than required CUDA binary compiled for newer compute capability than GPU supports GPU driver incompatibility with CUDA toolkit PyTorch compiled for different SM version than GPU Custom CUDA kernels compiled without proper arch flags","anchorText":"CUDA Illegal Instruction CUDA error: an illegal instruction was encountered RuntimeError: CUDA error: illegal instruction GPU compute capability mismatch cuda illegal-instruction gpu hardware compute-capability nvidia CUDA binary compiled for newer compute capability than GPU supports GPU driver incompatibility with CUDA toolkit PyTorch compiled for different SM version than GPU Custom CUDA kernels compiled without proper arch flags ","action":"Match CUDA toolkit version to GPU compute capability","steps":["Match CUDA toolkit version to GPU compute capability","Reinstall PyTorch with correct CUDA version","Recompile custom CUDA kernels for target architecture","Check GPU compute capability: nvidia-smi --query-gpu=compute_cap --format=csv"]},{"slug":"gpu-clock-throttle","title":"GPU Clock Throttle","category":"Hardware","text":"GPU Clock Throttle Hardware nvidia-smi shows clocks dropped nvidia-smi -q shows current clocks lower than max Power consumption at configured limit clock throttle power thermal gpu hardware performance Training performance is lower than expected GPU clock speeds drop during training Power consumption is at limit Power limit reached Thermal limit reached Voltage limit reached HW slowdown (hardware protection)","anchorText":"GPU Clock Throttle nvidia-smi shows clocks dropped nvidia-smi -q shows current clocks lower than max Power consumption at configured limit clock throttle power thermal gpu hardware performance Power limit reached Thermal limit reached Voltage limit reached HW slowdown (hardware protection) ","action":"Check power and thermal limits: nvidia-smi -q -d POWER, TEMPERATURE","steps":["Check power and thermal limits: nvidia-smi -q -d POWER, TEMPERATURE","Reduce GPU power limit if hitting power cap","Improve cooling if thermal throttling","Check for HW slowdown: dmesg | grep Slowdown"]},{"slug":"gpu-memory-bus-error","title":"GPU Memory Bus Error","category":"Hardware","text":"GPU Memory Bus Error Hardware Xid 48 or 64 in dmesg ECC uncorrectable errors Memory bus parity errors Training produces inconsistent results across runs memory-bus hardware gpu ecc xid corruption nvidia Training crashes with memory bus error Random NaN values in training GPU produces corrupted results GPU memory bus hardware fault Memory channels degraded Operating temperature too high Manufacturing defect in GPU memory","anchorText":"GPU Memory Bus Error Xid 48 or 64 in dmesg ECC uncorrectable errors Memory bus parity errors Training produces inconsistent results across runs memory-bus hardware gpu ecc xid corruption nvidia GPU memory bus hardware fault Memory channels degraded Operating temperature too high Manufacturing defect in GPU memory ","action":"Check dmesg for Xid events","steps":["Check dmesg for Xid events","Run nvidia-smi -q -d ECC to check for ECC errors","Drain faulty GPU","Replace GPU if errors persist","Use Denpex to correlate Xid events with training failures"]},{"slug":"gpu-mig-mode","title":"GPU MIG (Multi-Instance GPU) Mode","category":"Hardware","text":"GPU MIG (Multi-Instance GPU) Mode Hardware nvidia-smi shows MIG instances torch.cuda.device_count() returns wrong number NCCL initialization fails with MIG error mig multi-instance gpu nvidia a100 h100 partitioning PyTorch doesn't see all GPU memory NCCL fails to use MIG instances GPU shows multiple smaller devices MIG mode splits GPU into multiple instances PyTorch doesn't natively support MIG NCCL may not work across MIG instances MIG instances have limited memory and compute","anchorText":"GPU MIG (Multi-Instance GPU) Mode nvidia-smi shows MIG instances torch.cuda.device_count() returns wrong number NCCL initialization fails with MIG error mig multi-instance gpu nvidia a100 h100 partitioning MIG mode splits GPU into multiple instances PyTorch doesn't natively support MIG NCCL may not work across MIG instances MIG instances have limited memory and compute ","action":"Disable MIG mode if not needed: nvidia-smi -mig 0","steps":["Disable MIG mode if not needed: nvidia-smi -mig 0","Use CUDA_MIG_ENABLED for PyTorch MIG support","Use NVIDIA MPS for sharing instead of MIG","Check MIG documentation for NCCL support"]},{"slug":"runtime-error","title":"Runtime Error (Generic)","category":"Training Stability","text":"Runtime Error (Generic) Training Stability RuntimeError: generic error message RuntimeError: tensor sizes mismatch RuntimeError: device mismatch runtime-error generic training debugging stability Training crashes with RuntimeError Error message is generic Training was working before Code bug in custom training loop Tensor size mismatch between forward and target Device mismatch (GPU vs CPU tensors) Numerical instability in custom operation","anchorText":"Runtime Error (Generic) RuntimeError: generic error message RuntimeError: tensor sizes mismatch RuntimeError: device mismatch runtime-error generic training debugging stability Code bug in custom training loop Tensor size mismatch between forward and target Device mismatch (GPU vs CPU tensors) Numerical instability in custom operation ","action":"Check full stack trace for the error source","steps":["Check full stack trace for the error source","Verify tensor shapes and devices match","Use torch.autograd.set_detect_anomaly(True) for debugging","Test with simpler model first to isolate the issue"]},{"slug":"type-error","title":"Type Error (Python)","category":"Environment","text":"Type Error (Python) Environment TypeError: unsupported operand type(s) TypeError: object is not callable TypeError: argument must be str not int type-error python typing debugging environment Training crashes with TypeError Error message indicates type mismatch Code worked in older Python version Type mismatch in custom code Library API change in version update Incorrect type annotation Passing wrong type to function","anchorText":"Type Error (Python) TypeError: unsupported operand type(s) TypeError: object is not callable TypeError: argument must be str not int type-error python typing debugging environment Type mismatch in custom code Library API change in version update Incorrect type annotation Passing wrong type to function ","action":"Check type hints and ensure correct types are passed","steps":["Check type hints and ensure correct types are passed","Review error stack trace for type mismatch location","Use mypy or type checkers during development","Pin library versions to avoid API changes"]},{"slug":"value-error","title":"Value Error (Python)","category":"Environment","text":"Value Error (Python) Environment ValueError: invalid value ValueError: setting array element with sequence ValueError: too many values to unpack value-error python validation debugging environment Training crashes with ValueError Error indicates invalid value Code worked in older version Invalid configuration value Data validation failure Shape mismatch in array operations Incorrect argument value","anchorText":"Value Error (Python) ValueError: invalid value ValueError: setting array element with sequence ValueError: too many values to unpack value-error python validation debugging environment Invalid configuration value Data validation failure Shape mismatch in array operations Incorrect argument value ","action":"Check error stack trace for the invalid value","steps":["Check error stack trace for the invalid value","Validate configuration values before use","Use try/except for known validation errors","Check API documentation for valid value ranges"]},{"slug":"key-error","title":"Key Error (Python Dictionary)","category":"Environment","text":"Key Error (Python Dictionary) Environment KeyError: 'key_name' KeyError: 'model.layers.0.weight' KeyError: missing key in state_dict key-error python dictionary debugging environment config Training crashes with KeyError Error message shows the missing key Code worked before with different data Missing key in config file Missing key in state dict when loading checkpoint API response schema changed Dictionary key not initialized","anchorText":"Key Error (Python Dictionary) KeyError: 'key_name' KeyError: 'model.layers.0.weight' KeyError: missing key in state_dict key-error python dictionary debugging environment config Missing key in config file Missing key in state dict when loading checkpoint API response schema changed Dictionary key not initialized ","action":"Use dict.get(key, default) to handle missing keys","steps":["Use dict.get(key, default) to handle missing keys","Validate config schema before use","Check API documentation for all required keys","Use defaultdict for optional keys"]},{"slug":"attribute-error","title":"Attribute Error (Python)","category":"Environment","text":"Attribute Error (Python) Environment AttributeError: object has no attribute AttributeError: type object has no attribute AttributeError: module has no attribute attribute-error python attribute debugging environment api Training crashes with AttributeError Error indicates missing attribute Library API changed Library API change removed/renamed attribute Incorrect object type Typo in attribute name Module import issue","anchorText":"Attribute Error (Python) AttributeError: object has no attribute AttributeError: type object has no attribute AttributeError: module has no attribute attribute-error python attribute debugging environment api Library API change removed/renamed attribute Incorrect object type Typo in attribute name Module import issue ","action":"Check library documentation for current API","steps":["Check library documentation for current API","Verify object type before accessing attribute","Use hasattr() to check attribute exists","Update code to match new library API"]},{"slug":"index-error","title":"Index Error (Python)","category":"Environment","text":"Index Error (Python) Environment IndexError: list index out of range IndexError: tuple index out of range IndexError: string index out of range index-error python indexing debugging environment list Training crashes with IndexError Error indicates index out of range Code works with some data but not others Empty list access Index beyond list length Race condition in async code Off-by-one error in indexing","anchorText":"Index Error (Python) IndexError: list index out of range IndexError: tuple index out of range IndexError: string index out of range index-error python indexing debugging environment list Empty list access Index beyond list length Race condition in async code Off-by-one error in indexing ","action":"Check list length before accessing index: if len(lst) > idx","steps":["Check list length before accessing index: if len(lst) > idx","Use try/except for index access","Use enumerate for safe iteration","Validate data length before indexing"]},{"slug":"torch-compile-error","title":"PyTorch torch.compile Error","category":"Training Stability","text":"PyTorch torch.compile Error Training Stability RuntimeError: Triton compilation failed torch._dynamo.exc.BackendCompilerFailed InductorError: failed to compile Graph break in compiled model torch-compile dynamo inductor triton compilation optimization Training crashes during or after first forward pass Error related to Triton or Inductor Compiled model produces wrong results torch.compile doesn't support certain Python patterns Custom CUDA kernels need TorchInductor-compatible wrappers Dynamic shapes cause graph breaks Hardware (GPU) doesn't support compiled operations","anchorText":"PyTorch torch.compile Error RuntimeError: Triton compilation failed torch._dynamo.exc.BackendCompilerFailed InductorError: failed to compile Graph break in compiled model torch-compile dynamo inductor triton compilation optimization torch.compile doesn't support certain Python patterns Custom CUDA kernels need TorchInductor-compatible wrappers Dynamic shapes cause graph breaks Hardware (GPU) doesn't support compiled operations ","action":"Disable torch.compile and use eager mode","steps":["Disable torch.compile and use eager mode","Fix unsupported patterns in model code","Use torch._dynamo.config.suppress_errors() for debugging","Update to latest PyTorch for better compile support","Use torch.compile(mode='reduce-overhead') as fallback"]},{"slug":"torch-jit-compile-error","title":"PyTorch JIT Compile Error","category":"Training Stability","text":"PyTorch JIT Compile Error Training Stability torch.jit.frontend.UnsupportedNodeError RuntimeError: Script compilation failed RuntimeError: type hints mismatch torch-jit compilation script trace optimization type-hints torch.jit.script fails to compile torch.jit.trace fails to compile JIT-compiled model produces wrong results JIT doesn't support certain Python features Type hints missing or wrong Library not JIT-compatible Complex Python control flow not supported","anchorText":"PyTorch JIT Compile Error torch.jit.frontend.UnsupportedNodeError RuntimeError: Script compilation failed RuntimeError: type hints mismatch torch-jit compilation script trace optimization type-hints JIT doesn't support certain Python features Type hints missing or wrong Library not JIT-compatible Complex Python control flow not supported ","action":"Use torch.jit.script with type hints","steps":["Use torch.jit.script with type hints","Avoid unsupported Python features in JIT-compiled code","Use @torch.jit.ignore for unsupported functions","Test with simple model first","Use eager mode for development, JIT for production"]},{"slug":"torch-hub-error","title":"PyTorch Hub Error","category":"Environment","text":"PyTorch Hub Error Environment URLError: <urlopen error HTTPError: 404 Client Error RuntimeError: Error downloading model torch-hub pretrained model-loading environment download torch.hub.load fails Model download fails Hub connection error No internet connection to PyTorch Hub Hub server unavailable Model name typo Cache corrupted Firewall blocking PyTorch Hub","anchorText":"PyTorch Hub Error URLError: <urlopen error HTTPError: 404 Client Error RuntimeError: Error downloading model torch-hub pretrained model-loading environment download No internet connection to PyTorch Hub Hub server unavailable Model name typo Cache corrupted Firewall blocking PyTorch Hub ","action":"Check internet connection to pytorch.org","steps":["Check internet connection to pytorch.org","Verify model name on PyTorch Hub","Clear hub cache: rm -rf ~/.cache/torch/hub/","Use local model files instead of hub","Set TORCH_HOME to a writable directory"]},{"slug":"torch-multiprocessing-error","title":"PyTorch Multiprocessing Error","category":"Data Pipeline","text":"PyTorch Multiprocessing Error Data Pipeline RuntimeError: DataLoader worker (pid X) exited unexpectedly RuntimeError: Cannot re-init Dataloader multiprocessing.AuthenticationError torch-multiprocessing dataloader worker pickle fork spawn pipeline DataLoader workers fail to start Workers crash with multiprocessing error Shared tensors not visible across workers Worker process crashes during init Pickle error when passing objects to workers CUDA tensors cannot be shared across processes Fork vs spawn multiprocessing issues","anchorText":"PyTorch Multiprocessing Error RuntimeError: DataLoader worker (pid X) exited unexpectedly RuntimeError: Cannot re-init Dataloader multiprocessing.AuthenticationError torch-multiprocessing dataloader worker pickle fork spawn pipeline Worker process crashes during init Pickle error when passing objects to workers CUDA tensors cannot be shared across processes Fork vs spawn multiprocessing issues ","action":"Set num_workers=0 as diagnostic","steps":["Set num_workers=0 as diagnostic","Use spawn instead of fork for CUDA safety","Avoid passing CUDA tensors to workers","Ensure picklable objects are passed to workers","Use torch.multiprocessing for CUDA-safe workers"]},{"slug":"torch-distributed-error","title":"PyTorch Distributed Error","category":"Distributed Training","text":"PyTorch Distributed Error Distributed Training RuntimeError: Distributed package doesn't have NCCL built in torch.distributed.DistBackendError RuntimeError: process group not initialized TimeoutError: timed out initializing process group torch-distributed ddp fsdp process-group init distributed torch.distributed fails to initialize DDP or FSDP setup error Distributed training fails to start PyTorch not built with NCCL support Incorrect MASTER_ADDR or MASTER_PORT Network configuration issues Process group not destroyed between runs","anchorText":"PyTorch Distributed Error RuntimeError: Distributed package doesn't have NCCL built in torch.distributed.DistBackendError RuntimeError: process group not initialized TimeoutError: timed out initializing process group torch-distributed ddp fsdp process-group init distributed PyTorch not built with NCCL support Incorrect MASTER_ADDR or MASTER_PORT Network configuration issues Process group not destroyed between runs ","action":"Reinstall PyTorch with NCCL: pip install torch --index-url https://download.pytorch.org/whl/cu118","steps":["Reinstall PyTorch with NCCL: pip install torch --index-url https://download.pytorch.org/whl/cu118","Set MASTER_ADDR and MASTER_PORT correctly","Use torchrun instead of manual launch","Call torch.distributed.destroy_process_group() at exit"]},{"slug":"torch-cuda-error","title":"PyTorch CUDA Error (Generic)","category":"Environment","text":"PyTorch CUDA Error (Generic) Environment RuntimeError: CUDA error: generic RuntimeError: CUDA error: unknown error RuntimeError: CUDA out of memory torch-cuda gpu cuda generic error debugging Training fails with generic CUDA error Error doesn't provide specific information CUDA operations fail unexpectedly Various CUDA issues Hardware problems Driver issues Memory issues Compatibility issues","anchorText":"PyTorch CUDA Error (Generic) RuntimeError: CUDA error: generic RuntimeError: CUDA error: unknown error RuntimeError: CUDA out of memory torch-cuda gpu cuda generic error debugging Various CUDA issues Hardware problems Driver issues Memory issues Compatibility issues ","action":"Check error message for specific issue","steps":["Check error message for specific issue","Verify CUDA toolkit and driver compatibility","Update NVIDIA driver","Check GPU hardware health","Use Denpex to diagnose specific CUDA errors"]},{"slug":"adamw-weight-decay","title":"AdamW Weight Decay Misconfiguration","category":"Training Stability","text":"AdamW Weight Decay Misconfiguration Training Stability Loss: nan, val_loss: high Loss decreases but val_loss increases Model doesn't converge adamw weight-decay optimizer regularization training-stability overfitting Loss doesn't improve despite good setup Model overfits or underfits Validation metrics plateau Weight decay applied to all parameters including biases and norms Weight decay too high or too low Weight decay not applied to correct parameter groups Learning rate and weight decay not balanced","anchorText":"AdamW Weight Decay Misconfiguration Loss: nan, val_loss: high Loss decreases but val_loss increases Model doesn't converge adamw weight-decay optimizer regularization training-stability overfitting Weight decay applied to all parameters including biases and norms Weight decay too high or too low Weight decay not applied to correct parameter groups Learning rate and weight decay not balanced ","action":"Apply weight decay only to weight parameters, not biases or norms","steps":["Apply weight decay only to weight parameters, not biases or norms","Use parameter groups to separate decay and no-decay params","Tune weight decay with learning rate sweep","Start with weight_decay=0.01 for AdamW","Monitor train/val loss gap for overfitting"]},{"slug":"adam-epsilon-issue","title":"Adam Epsilon Hyperparameter Issue","category":"Training Stability","text":"Adam Epsilon Hyperparameter Issue Training Stability Loss explodes to NaN Loss plateaus at high value Training is unstable adam epsilon optimizer training-stability fp16 bf16 Training diverges or plateaus Loss values are NaN or extremely large Model doesn't converge with default Adam Epsilon too small causes division by zero in Adam update Epsilon too large prevents effective updates Epsilon interacts with gradient clipping Default epsilon (1e-8) may be too small for fp16","anchorText":"Adam Epsilon Hyperparameter Issue Loss explodes to NaN Loss plateaus at high value Training is unstable adam epsilon optimizer training-stability fp16 bf16 Epsilon too small causes division by zero in Adam update Epsilon too large prevents effective updates Epsilon interacts with gradient clipping Default epsilon (1e-8) may be too small for fp16 ","action":"Use default epsilon (1e-8) for fp32 training","steps":["Use default epsilon (1e-8) for fp32 training","Use epsilon=1e-6 or 1e-4 for fp16/bf16 training","Tune epsilon if training is unstable","Monitor gradient norms during training","Use AdamW for better default behavior"]},{"slug":"sgd-momentum-issue","title":"SGD Momentum Configuration Issue","category":"Training Stability","text":"SGD Momentum Configuration Issue Training Stability Loss increases after warmup Loss oscillates with large amplitude Training time exceeds expectations sgd momentum optimizer training-stability nesterov convergence Loss oscillates wildly Loss diverges after some training Training is very slow to converge Momentum too high causes oscillation Momentum too low causes slow convergence Momentum without Nesterov slows training Learning rate and momentum not balanced","anchorText":"SGD Momentum Configuration Issue Loss increases after warmup Loss oscillates with large amplitude Training time exceeds expectations sgd momentum optimizer training-stability nesterov convergence Momentum too high causes oscillation Momentum too low causes slow convergence Momentum without Nesterov slows training Learning rate and momentum not balanced ","action":"Reduce learning rate (momentum and LR are coupled)","steps":["Reduce learning rate (momentum and LR are coupled)","Use Nesterov momentum for faster convergence","Tune momentum with learning rate sweep","Start with momentum=0.9 and tune from there","Monitor loss for oscillation patterns"]},{"slug":"num-workers-zero","title":"Dataloader num_workers=0 Too Slow","category":"Data Pipeline","text":"Dataloader num_workers=0 Too Slow Data Pipeline Low GPU-SM utilization High CPU data loading time GPU waiting for data dataloader num-workers performance pipeline gpu-utilization GPU utilization is low (e.g., 30-50%) Training is slower than expected DataLoader iteration is the bottleneck Data loading is single-threaded GPU is faster than data loading pipeline CPU bottleneck in data augmentation num_workers=0 means no parallel data loading","anchorText":"Dataloader num_workers=0 Too Slow Low GPU-SM utilization High CPU data loading time GPU waiting for data dataloader num-workers performance pipeline gpu-utilization Data loading is single-threaded GPU is faster than data loading pipeline CPU bottleneck in data augmentation num_workers=0 means no parallel data loading ","action":"Set num_workers=4-8 for production training","steps":["Set num_workers=4-8 for production training","Use prefetch_factor=2 to preload data","Profile data loading with PyTorch Profiler","Use persistent_workers=True to avoid worker init overhead","Consider using NVIDIA DALI for faster data loading"]},{"slug":"pin-memory-issue","title":"Pinned Memory Transfer Issue","category":"Data Pipeline","text":"Pinned Memory Transfer Issue Data Pipeline RuntimeError: cudaHostAlloc failed OSError: cannot allocate pinned memory Slow GPU-CPU data transfer pin-memory host-memory dataloader pipeline performance Data loading is slow OOM errors in pinned memory Pinned memory allocation fails Insufficient host memory for pinned allocation Too many CUDA contexts requesting pinned memory Pinned memory pool exhausted by other processes Large tensors with pin_memory=True exceed available memory","anchorText":"Pinned Memory Transfer Issue RuntimeError: cudaHostAlloc failed OSError: cannot allocate pinned memory Slow GPU-CPU data transfer pin-memory host-memory dataloader pipeline performance Insufficient host memory for pinned allocation Too many CUDA contexts requesting pinned memory Pinned memory pool exhausted by other processes Large tensors with pin_memory=True exceed available memory ","action":"Set pin_memory=False as diagnostic","steps":["Set pin_memory=False as diagnostic","Reduce num_workers to decrease pinned memory usage","Increase system RAM","Release other CUDA contexts","Use pin_memory=True only with sufficient host memory"]},{"slug":"checkpoint-version-incompatible","title":"Checkpoint Version Incompatible","category":"Data Integrity","text":"Checkpoint Version Incompatible Data Integrity RuntimeError: version mismatch RuntimeError: invalid version key Loading checkpoint produces state_dict mismatch checkpoint version pytorch incompatibility data-integrity Loading checkpoint fails with version error Error indicates version mismatch Checkpoint was saved with different PyTorch version PyTorch version changed between save and load Checkpoint format changed in new PyTorch version Optimized data structures changed Model architecture changed between versions","anchorText":"Checkpoint Version Incompatible RuntimeError: version mismatch RuntimeError: invalid version key Loading checkpoint produces state_dict mismatch checkpoint version pytorch incompatibility data-integrity PyTorch version changed between save and load Checkpoint format changed in new PyTorch version Optimized data structures changed Model architecture changed between versions ","action":"Reinstall PyTorch to match checkpoint version","steps":["Reinstall PyTorch to match checkpoint version","Convert checkpoint to current version: load with weights_only=True and map_location","Strip version keys from checkpoint","Use torch.save with pickle protocol=2 for compatibility"]},{"slug":"optimizer-state-dict-mismatch","title":"Optimizer State Dict Mismatch","category":"Data Integrity","text":"Optimizer State Dict Mismatch Data Integrity Error(s) in loading optimizer state_dict: Unexpected key(s) KeyError: missing optimizer state for parameters optimizer.state_dict() shapes don't match model optimizer checkpoint state-dict resume training mismatch Resume from checkpoint fails with optimizer state mismatch Optimizer states don't match current model Training degrades after resume because optimizer state was loaded wrong Optimizer changed (AdamW to SGD) but checkpoint has old states Model architecture changed Checkpoint from architecture A loaded with architecture B Optimizer hyperparameters differ between save and load","anchorText":"Optimizer State Dict Mismatch Error(s) in loading optimizer state_dict: Unexpected key(s) KeyError: missing optimizer state for parameters optimizer.state_dict() shapes don't match model optimizer checkpoint state-dict resume training mismatch Optimizer changed (AdamW to SGD) but checkpoint has old states Model architecture changed Checkpoint from architecture A loaded with architecture B Optimizer hyperparameters differ between save and load ","action":"Load optimizer state_dict with strict=False and filter mismatched keys","steps":["Load optimizer state_dict with strict=False and filter mismatched keys","Reinitialize optimizer: torch.optim.AdamW(model.parameters())","Use a learning rate warmup after optimizer reinit","Use Denpex to identify mismatched parameter names"]},{"slug":"ema-checkpoint-issue","title":"EMA Checkpoint Issue","category":"Training Stability","text":"EMA Checkpoint Issue Training Stability EMA model weights are all zeros or NaN EMA model produces different results than expected EMA model diverges from training model ema exponential-moving-average checkpoint model-averaging training-stability EMA model not loaded correctly EMA weights not applied at inference EMA tracking fails during training EMA decay rate misconfigured EMA model not saved with checkpoint EMA implementation bug EMA applied to wrong parameters","anchorText":"EMA Checkpoint Issue EMA model weights are all zeros or NaN EMA model produces different results than expected EMA model diverges from training model ema exponential-moving-average checkpoint model-averaging training-stability EMA decay rate misconfigured EMA model not saved with checkpoint EMA implementation bug EMA applied to wrong parameters ","action":"Use standard EMA decay (0.999 or 0.9999)","steps":["Use standard EMA decay (0.999 or 0.9999)","Save EMA model state with checkpoint","Use timm or PyTorch's torch.optim.swa_utils for EMA","Verify EMA weights are being updated","Test EMA model separately before training"]},{"slug":"nccl-error-2","title":"NCCL Error 2: Internal Error","category":"Communication","text":"NCCL Error 2: Internal Error Communication NCCL error 2: internal error at /workspace/nccl/src/misc/argcheck.cpp NCCL internal assertion failed NCCL WARN NCCL error 2: internal error nccl internal error-2 assertion communication distributed Training terminates with NCCL error 2 NCCL internal assertion failure Training crashes with opaque error code NCCL internal state corruption GPU memory corruption affecting NCCL buffers Network communication causing internal inconsistency NCCL version mismatch with CUDA driver","anchorText":"NCCL Error 2: Internal Error NCCL error 2: internal error at /workspace/nccl/src/misc/argcheck.cpp NCCL internal assertion failed NCCL WARN NCCL error 2: internal error nccl internal error-2 assertion communication distributed NCCL internal state corruption GPU memory corruption affecting NCCL buffers Network communication causing internal inconsistency NCCL version mismatch with CUDA driver ","action":"Check dmesg on the failing rank for Xid events","steps":["Check dmesg on the failing rank for Xid events","Verify NCCL version compatibility: python -c \"import torch; print(torch.cuda.nccl.version())\"","Run CUDA memtest on the failing GPU","Reset the GPU: nvidia-smi --gpu-reset","Update NCCL and CUDA driver to compatible versions"]},{"slug":"checkpoint-version-newer","title":"Checkpoint Saved with Newer Version","category":"Data Integrity","text":"Checkpoint Saved with Newer Version Data Integrity RuntimeError: invalid version_key RuntimeError: unexpected EOF, expected serialized values Checkpoint loads partially with random weights checkpoint version pytorch incompatibility data-integrity newer Loading checkpoint fails with version error Error indicates checkpoint was saved with newer version Production runs older PyTorch than dev Checkpoint format changed between PyTorch versions Internal serialization changed Optimizer state format updated Model class structure changed","anchorText":"Checkpoint Saved with Newer Version RuntimeError: invalid version_key RuntimeError: unexpected EOF, expected serialized values Checkpoint loads partially with random weights checkpoint version pytorch incompatibility data-integrity newer Checkpoint format changed between PyTorch versions Internal serialization changed Optimizer state format updated Model class structure changed ","action":"Match PyTorch version between save and load","steps":["Match PyTorch version between save and load","Reinstall production PyTorch to match dev version","Convert checkpoint: load with map_location, re-save with target version","Use HuggingFace safetensors for forward compatibility","Strip version_key from checkpoint dict"]},{"slug":"nccl-init-timeout","title":"NCCL Initialization Timeout","category":"Communication","text":"NCCL Initialization Timeout Communication TimeoutError: NCCL init timeout NCCL WARN connection timed out RuntimeError: timed out initializing process group nccl init timeout distributed multi-node firewall torch.distributed.init_process_group hangs NCCL init times out before all ranks connect Distributed training fails to start Network firewalls blocking NCCL ports Slow network between compute nodes NCCL trying to bind to wrong interface DNS resolution failure between nodes","anchorText":"NCCL Initialization Timeout TimeoutError: NCCL init timeout NCCL WARN connection timed out RuntimeError: timed out initializing process group nccl init timeout distributed multi-node firewall Network firewalls blocking NCCL ports Slow network between compute nodes NCCL trying to bind to wrong interface DNS resolution failure between nodes ","action":"Check firewall rules: sudo iptables -L (allow 29500-29599)","steps":["Check firewall rules: sudo iptables -L (allow 29500-29599)","Verify network connectivity: ping and nc -zv between nodes","Set NCCL_SOCKET_IFNAME to the correct interface","Increase NCCL timeout: export NCCL_TIMEOUT=7200","Use NCCL_DEBUG=INFO to diagnose init issues"]},{"slug":"nccl-bad-rdma","title":"NCCL Bad RDMA Performance","category":"Communication","text":"NCCL Bad RDMA Performance Communication NCCL performance is 10x slower than expected GPUDirect RDMA is not available NCCL falls back to slower transport nccl rdma gpudirect performance network infiniband Multi-node training is much slower than expected NCCL performance is well below interconnect bandwidth Communication time dominates training time GPUDirect RDMA is disabled or unavailable NIC is not connected to GPU PCIe switch Outdated NIC driver NCCL not configured for RDMA transport","anchorText":"NCCL Bad RDMA Performance NCCL performance is 10x slower than expected GPUDirect RDMA is not available NCCL falls back to slower transport nccl rdma gpudirect performance network infiniband GPUDirect RDMA is disabled or unavailable NIC is not connected to GPU PCIe switch Outdated NIC driver NCCL not configured for RDMA transport ","action":"Enable GPUDirect RDMA: sudo modprobe nvidia_peermem","steps":["Enable GPUDirect RDMA: sudo modprobe nvidia_peermem","Update NIC driver: ibstat for IB or ethtool for RoCE","Verify GPU-NIC topology: nvidia-smi topo -m","Set NCCL_IB_HCA to the correct device","Check NCCL_NET_GDR_LEVEL is appropriate"]},{"slug":"cuda-shared-memory-limit","title":"CUDA Shared Memory Limit Exceeded","category":"Memory","text":"CUDA Shared Memory Limit Exceeded Memory RuntimeError: too much shared memory requested cudaErrorInvalidValue: invalid kernel argument cuda shared-memory kernel memory gpu limit CUDA kernel fails to launch with shared memory error Reduce shared memory usage to fit within limits Shared memory request exceeds per-block limit Kernel compiled with too many threads per block Reduction operations with large input size","anchorText":"CUDA Shared Memory Limit Exceeded RuntimeError: too much shared memory requested cudaErrorInvalidValue: invalid kernel argument cuda shared-memory kernel memory gpu limit Shared memory request exceeds per-block limit Kernel compiled with too many threads per block Reduction operations with large input size ","action":"Reduce shared memory per block: reduce BLOCK_SIZE","steps":["Reduce shared memory per block: reduce BLOCK_SIZE","Use fewer threads per block","Split computation into multiple kernel launches","Use registers instead of shared memory where possible","Use dynamic shared memory with proper allocation"]},{"slug":"batch-norm-ddp-sync","title":"BatchNorm DDP Synchronization Issue","category":"Data Pipeline","text":"BatchNorm DDP Synchronization Issue Data Pipeline BatchNorm running mean/var differs between ranks DDP model produces different outputs on different ranks BN momentum is not synchronized batchnorm sync ddp distributed running-stats model-quality Model performance is worse with DDP than single GPU BN running stats differ across ranks Validation accuracy fluctuates randomly SyncBN not enabled: BN running stats computed independently per rank BN momentum differs across ranks BN tracking stats accumulated differently due to different data order","anchorText":"BatchNorm DDP Synchronization Issue BatchNorm running mean/var differs between ranks DDP model produces different outputs on different ranks BN momentum is not synchronized batchnorm sync ddp distributed running-stats model-quality SyncBN not enabled: BN running stats computed independently per rank BN momentum differs across ranks BN tracking stats accumulated differently due to different data order ","action":"Convert to SyncBatchNorm: model = torch.nn.SyncBatchNorm.convert_sync_batchnorm(model)","steps":["Convert to SyncBatchNorm: model = torch.nn.SyncBatchNorm.convert_sync_batchnorm(model)","Enable DDP with find_unused_parameters=True if needed","Verify BN running stats match across ranks","Use channel-last memory format for better performance with BN"]},{"slug":"cloud-quota-exceeded","title":"Cloud GPU Quota Exceeded","category":"Infrastructure","text":"Cloud GPU Quota Exceeded Infrastructure AWS: InstanceLimitExceeded: You have requested more instances than your current EC2 instance type limit GCP: QUOTA_EXCEEDED: GPUs per region Azure: OperationNotAllowed: QuotaExceeded cloud quota gpu infrastructure aws gcp azure limit Cannot launch new training instance Cloud provider shows quota exceeded error Job stays in pending state indefinitely GPU quota limit reached for the account Concurrent instance limit exceeded Region-specific GPU quota exhausted Reserved instance quota not yet active","anchorText":"Cloud GPU Quota Exceeded AWS: InstanceLimitExceeded: You have requested more instances than your current EC2 instance type limit GCP: QUOTA_EXCEEDED: GPUs per region Azure: OperationNotAllowed: QuotaExceeded cloud quota gpu infrastructure aws gcp azure limit GPU quota limit reached for the account Concurrent instance limit exceeded Region-specific GPU quota exhausted Reserved instance quota not yet active ","action":"Request GPU quota increase from cloud provider","steps":["Request GPU quota increase from cloud provider","Use different region with available quota","Switch to smaller GPU type (fewer quota needed)","Use spot/preemptible instances for burst capacity","Delete unused instances to free up quota"]},{"slug":"gpu-ecc-error-rate-high","title":"High ECC Error Rate Detection","category":"Hardware","text":"High ECC Error Rate Detection Hardware dmesg: nvidia-nvme: Corrected ECC errors detected on GPU X nvidia-smi -q -d ECC shows non-zero volatile counts Correctable error count growing over time ecc correctable-error memory gpu hardware degradation monitoring Training errors increase over time GPU produces inconsistent results Memory errors detected in dmesg GPU memory cells degrading over time Manufacturing defect in GPU memory Operating temperature too high Memory bus degradation","anchorText":"High ECC Error Rate Detection dmesg: nvidia-nvme: Corrected ECC errors detected on GPU X nvidia-smi -q -d ECC shows non-zero volatile counts Correctable error count growing over time ecc correctable-error memory gpu hardware degradation monitoring GPU memory cells degrading over time Manufacturing defect in GPU memory Operating temperature too high Memory bus degradation ","action":"Monitor ECC error rates: nvidia-smi -q -d ECC","steps":["Monitor ECC error rates: nvidia-smi -q -d ECC","Schedule GPU replacement when correctable errors grow","Improve cooling if error rate correlates with temperature","Move workloads off GPU when error rate exceeds threshold","Use Denpex to monitor ECC error trends over time"]},{"slug":"model-state-dict-corrupt","title":"Model State Dict Corruption","category":"Data Integrity","text":"Model State Dict Corruption Data Integrity Model output is random or constant Loss is inf or NaN after loading State dict keys don't match expected parameters model-state-dict corruption checkpoint data-integrity pretrained Model produces garbage outputs after loading checkpoint Loss jumps to unexpected value Validation accuracy is much worse after resume Checkpoint file is corrupted (bit rot, partial write) Model architecture changed between save and load State dict saved with different precision (fp32 vs fp16) Pretrained model has been modified externally","anchorText":"Model State Dict Corruption Model output is random or constant Loss is inf or NaN after loading State dict keys don't match expected parameters model-state-dict corruption checkpoint data-integrity pretrained Checkpoint file is corrupted (bit rot, partial write) Model architecture changed between save and load State dict saved with different precision (fp32 vs fp16) Pretrained model has been modified externally ","action":"Verify checkpoint integrity: compute SHA-256 of all shards","steps":["Verify checkpoint integrity: compute SHA-256 of all shards","Compare parameter norms before and after loading","Run a validation pass to confirm model quality","Re-download from source if pretrained weights are corrupted","Use safetensors format which validates checksums"]},{"slug":"nccl-watchdog-disable","title":"NCCL Watchdog Configuration","category":"Distributed Training","text":"NCCL Watchdog Configuration Distributed Training NCCL WARN Operation completed before timeout Watchdog timeout not firing when expected Training hangs for full HANG_TIME before failure nccl watchdog timeout distributed monitoring failure-detection Training hangs for too long before NCCL timeout Watchdog fires too quickly on transient slowness NCCL errors don't appear when expected Default HANG_TIME is 1800 seconds (30 minutes) Watchdog settings not tuned for workload Network latency affects watchdog effectiveness Node failures vs slow nodes hard to distinguish","anchorText":"NCCL Watchdog Configuration NCCL WARN Operation completed before timeout Watchdog timeout not firing when expected Training hangs for full HANG_TIME before failure nccl watchdog timeout distributed monitoring failure-detection Default HANG_TIME is 1800 seconds (30 minutes) Watchdog settings not tuned for workload Network latency affects watchdog effectiveness Node failures vs slow nodes hard to distinguish ","action":"Set NCCL_TIMEOUT=600 for faster failure detection on critical workloads","steps":["Set NCCL_TIMEOUT=600 for faster failure detection on critical workloads","Set NCCL_HEARTBEAT_TIMEOUT_SEC=300 for heartbeat monitoring","Enable NCCL_ASYNC_ERROR_HANDLING=1 for graceful recovery","Use NCCL_DEBUG_SUBSYS=ALL for detailed diagnosis","Monitor watchdog timeouts with Denpex for anomaly detection"]},{"slug":"nccl-async-error","title":"NCCL Asynchronous Error Handling","category":"Communication","text":"NCCL Asynchronous Error Handling Communication NCCL WARN asynchronous operation failed NCCL collective returns but data is corrupt Training continues with corrupted gradients nccl async error-handling distributed reliability timeout Training continues after NCCL error should have stopped it Deadlock in async error handling NCCL errors not properly propagated to Python NCCL_ASYNC_ERROR_HANDLING not properly configured Custom error handling conflicts with NCCL internal handling Async collective results not properly checked Error handling code has race conditions","anchorText":"NCCL Asynchronous Error Handling NCCL WARN asynchronous operation failed NCCL collective returns but data is corrupt Training continues with corrupted gradients nccl async error-handling distributed reliability timeout NCCL_ASYNC_ERROR_HANDLING not properly configured Custom error handling conflicts with NCCL internal handling Async collective results not properly checked Error handling code has race conditions ","action":"Set NCCL_ASYNC_ERROR_HANDLING=1 for graceful error handling","steps":["Set NCCL_ASYNC_ERROR_HANDLING=1 for graceful error handling","Always check return value of async NCCL operations","Use torch.distributed.barrier() after critical collectives","Implement proper timeout handling in async code","Test error handling with intentional failures in dev"]},{"slug":"cuda-caching-allocator","title":"CUDA Caching Allocator Issue","category":"Memory","text":"CUDA Caching Allocator Issue Memory nvidia-smi shows memory not released torch.cuda.memory_allocated() vs nvidia-smi shows different values Caching allocator holds onto memory blocks cuda caching-allocator memory pytorch oom release GPU memory not released after deleting tensors Memory usage grows over training OOM despite torch.cuda.empty_cache() not helping Caching allocator keeps memory blocks for reuse Blocks are not released back to GPU driver Reserved memory grows over time Memory fragmentation prevents release","anchorText":"CUDA Caching Allocator Issue nvidia-smi shows memory not released torch.cuda.memory_allocated() vs nvidia-smi shows different values Caching allocator holds onto memory blocks cuda caching-allocator memory pytorch oom release Caching allocator keeps memory blocks for reuse Blocks are not released back to GPU driver Reserved memory grows over time Memory fragmentation prevents release ","action":"Call torch.cuda.empty_cache() to release cached memory","steps":["Call torch.cuda.empty_cache() to release cached memory","Use expandables_segments=True for dynamic memory","Use PYTORCH_CUDA_ALLOC_CONF=max_split_size_mb to control block size","Monitor memory with torch.cuda.memory_summary()","Use Denpex to track memory patterns over time"]},{"slug":"webdataset-error","title":"WebDataset Error","category":"Data Pipeline","text":"WebDataset Error Data Pipeline RuntimeError: WebDataset shard not found tarfile.ReadError: invalid header ConnectionError: failed to fetch shard webdataset streaming data shards tar pipeline WebDataset fails to load Shards return errors Training stalls at shard boundary Shard URL is wrong or file is missing tar file is corrupted Network connection failed during download Shard format is inconsistent","anchorText":"WebDataset Error RuntimeError: WebDataset shard not found tarfile.ReadError: invalid header ConnectionError: failed to fetch shard webdataset streaming data shards tar pipeline Shard URL is wrong or file is missing tar file is corrupted Network connection failed during download Shard format is inconsistent ","action":"Verify shard URLs are accessible: aws s3 ls or gsutil ls","steps":["Verify shard URLs are accessible: aws s3 ls or gsutil ls","Check shard integrity: tar -tf shard.tar | head","Use valid_shards=True to skip invalid shards in WebDataset","Implement retry logic with exponential backoff for downloads","Pre-download shards to local storage for reliability"]},{"slug":"lr-warmup-decay","title":"LR Warmup-Decay Schedule Issue","category":"Training Stability","text":"LR Warmup-Decay Schedule Issue Training Stability Loss jumps at step 1000 (warmup end) Loss jumps at step 10000 (decay start) Unstable training in middle epochs warmup decay learning-rate scheduler training-stability transition Loss spikes at warmup end Loss spikes at decay start Training is unstable at schedule transitions Learning rate changes too quickly at warmup end Decay rate is too aggressive Warmup and decay phases overlap incorrectly Schedule milestones don't match actual training steps","anchorText":"LR Warmup-Decay Schedule Issue Loss jumps at step 1000 (warmup end) Loss jumps at step 10000 (decay start) Unstable training in middle epochs warmup decay learning-rate scheduler training-stability transition Learning rate changes too quickly at warmup end Decay rate is too aggressive Warmup and decay phases overlap incorrectly Schedule milestones don't match actual training steps ","action":"Use gradual warmup over 5-10% of total steps","steps":["Use gradual warmup over 5-10% of total steps","Match decay schedule to actual training steps","Use cosine decay instead of step decay for smoother transitions","Test schedule with small number of steps before full training","Monitor learning rate and loss to catch transition issues"]},{"slug":"gpu-scheduling-delay","title":"GPU Scheduling Delay","category":"Infrastructure","text":"GPU Scheduling Delay Infrastructure SLURM: pending for hours due to unavailable resources Kubernetes: pod pending due to insufficient GPU Cloud: waiting for GPU capacity gpu scheduling queue infrastructure slurm kubernetes cloud Jobs stay in queue for hours GPU resources show as available but not allocated Scheduling takes longer than expected All GPUs in use by other jobs Node failures leave GPUs in drain state GPU quota exceeded GPU allocation policy too restrictive","anchorText":"GPU Scheduling Delay SLURM: pending for hours due to unavailable resources Kubernetes: pod pending due to insufficient GPU Cloud: waiting for GPU capacity gpu scheduling queue infrastructure slurm kubernetes cloud All GPUs in use by other jobs Node failures leave GPUs in drain state GPU quota exceeded GPU allocation policy too restrictive ","action":"Request specific GPU resources to avoid unnecessary waiting","steps":["Request specific GPU resources to avoid unnecessary waiting","Use lower-priority GPU types for non-critical work","Implement backfill scheduling for short jobs","Clean up zombie processes to free GPUs","Use Denpex pre-flight checks to validate resource availability"]},{"slug":"cuda-uvm-error","title":"CUDA Unified Memory Error","category":"Memory","text":"CUDA Unified Memory Error Memory CUDA error: page fault or illegal memory access CUDA out of memory with UVM Performance degradation with UVM cuda uvm unified-memory page-fault memory performance Page fault errors during training Training is slower when using UVM CUDA error related to managed memory Page fault when GPU accesses CPU memory Memory transfer overhead between CPU and GPU UVM oversubscription causes thrashing UVM page size not optimal for workload","anchorText":"CUDA Unified Memory Error CUDA error: page fault or illegal memory access CUDA out of memory with UVM Performance degradation with UVM cuda uvm unified-memory page-fault memory performance Page fault when GPU accesses CPU memory Memory transfer overhead between CPU and GPU UVM oversubscription causes thrashing UVM page size not optimal for workload ","action":"Enable UVM with cudaMallocManaged for automatic oversubscription","steps":["Enable UVM with cudaMallocManaged for automatic oversubscription","Pre-touch memory to avoid page faults during training","Use stream-ordered allocations for better performance","Profile UVM page faults to optimize memory access patterns","Consider model parallelism instead of UVM for very large models"]},{"slug":"torchvision-error","title":"Torchvision Error","category":"Environment","text":"Torchvision Error Environment RuntimeError: torchvision version mismatch ValueError: PIL image not loaded ImportError: cannot import from torchvision torchvision vision pretrained environment compatibility torchvision fails to load pretrained model Image transforms produce wrong output CUDA error in vision model Torchvision version doesn't match PyTorch Pretrained model URL is broken Image format not supported CUDA backend not available in torchvision","anchorText":"Torchvision Error RuntimeError: torchvision version mismatch ValueError: PIL image not loaded ImportError: cannot import from torchvision torchvision vision pretrained environment compatibility Torchvision version doesn't match PyTorch Pretrained model URL is broken Image format not supported CUDA backend not available in torchvision ","action":"Match torchvision version to PyTorch version","steps":["Match torchvision version to PyTorch version","Use try/except for model loading with fallbacks","Verify image format before transforms","Update torchvision: pip install torchvision --upgrade","Use Denpex to check torchvision compatibility"]},{"slug":"tokenizer-pad-truncate","title":"Tokenizer Padding and Truncation Error","category":"Data Pipeline","text":"Tokenizer Padding and Truncation Error Data Pipeline ValueError: sequence length exceeds maximum RuntimeError: cannot pad to max length Batch has sequences longer than model supports tokenizer padding truncation nlp data-pipeline sequence Tokenizer errors with variable-length sequences Sequence length exceeds model max length OOM due to long sequences in batch Sequence in batch exceeds model max length Padding token not set correctly Truncation strategy not appropriate for data Different sequences in batch have very different lengths","anchorText":"Tokenizer Padding and Truncation Error ValueError: sequence length exceeds maximum RuntimeError: cannot pad to max length Batch has sequences longer than model supports tokenizer padding truncation nlp data-pipeline sequence Sequence in batch exceeds model max length Padding token not set correctly Truncation strategy not appropriate for data Different sequences in batch have very different lengths ","action":"Enable truncation: tokenizer(text, truncation=True, max_length=512)","steps":["Enable truncation: tokenizer(text, truncation=True, max_length=512)","Set padding to longest: padding='max_length' or 'longest'","Use dynamic padding: DataCollatorWithPadding","Filter out sequences longer than max length in preprocessing","Use sorted batching to reduce padding overhead"]},{"slug":"signal-handler-missing","title":"Missing Signal Handler","category":"Reliability","text":"Missing Signal Handler Reliability SIGTERM/SIGINT received but training continues CUDA OOM after interrupted training Checkpoints corrupted when process killed signal handler graceful-shutdown reliability checkpoint interruption Training doesn't save checkpoint on Ctrl+C GPU memory not released on interrupt Process killed mid-training without cleanup Signal handler not registered Cleanup logic in main thread only CUDA context not released on signal Multiple processes not handling signals","anchorText":"Missing Signal Handler SIGTERM/SIGINT received but training continues CUDA OOM after interrupted training Checkpoints corrupted when process killed signal handler graceful-shutdown reliability checkpoint interruption Signal handler not registered Cleanup logic in main thread only CUDA context not released on signal Multiple processes not handling signals ","action":"Register signal handlers: signal.signal(signal.SIGTERM, handler)","steps":["Register signal handlers: signal.signal(signal.SIGTERM, handler)","Save checkpoint in handler before exit","Use torch.distributed.destroy_process_group() in handler","Release GPU memory: torch.cuda.empty_cache() in handler","Test signal handling with kill -SIGTERM <pid> in dev"]},{"slug":"gpu-memory-bandwidth-limit","title":"GPU Memory Bandwidth Limit","category":"Hardware","text":"GPU Memory Bandwidth Limit Hardware Low GPU SM efficiency despite high compute usage nvidia-smi shows high GPU utilization but slow training Memory throughput is at peak gpu memory-bandwidth performance memory-bound compute-bound profiling Training is slower than theoretical maximum GPU compute utilization is high but training is slow Profiling shows memory bandwidth is bottleneck Model is memory-bound not compute-bound GPU memory bandwidth is the bottleneck Compute units idle waiting for memory Small batch size doesn't fully utilize compute","anchorText":"GPU Memory Bandwidth Limit Low GPU SM efficiency despite high compute usage nvidia-smi shows high GPU utilization but slow training Memory throughput is at peak gpu memory-bandwidth performance memory-bound compute-bound profiling Model is memory-bound not compute-bound GPU memory bandwidth is the bottleneck Compute units idle waiting for memory Small batch size doesn't fully utilize compute ","action":"Increase batch size to improve compute-to-memory ratio","steps":["Increase batch size to improve compute-to-memory ratio","Use gradient accumulation to simulate larger batches","Use activation checkpointing to reduce memory pressure","Profile with PyTorch Profiler to identify memory-bound operations","Use mixed precision to reduce memory bandwidth requirements"]},{"slug":"weight-init-wrong","title":"Wrong Weight Initialization","category":"Training Stability","text":"Wrong Weight Initialization Training Stability loss: nan in first step loss stays constant for thousands of steps Model produces constant outputs for all inputs weight-initialization kaiming xavier training-stability dead-neurons vanishing-gradient Loss is NaN from the first step Loss decreases very slowly or not at all Many neurons output 0 (dead ReLU) Weights initialized too small causing vanishing gradients Weights initialized too large causing exploding gradients Activation function not suited to initialization scheme No weight initialization specified in model code","anchorText":"Wrong Weight Initialization loss: nan in first step loss stays constant for thousands of steps Model produces constant outputs for all inputs weight-initialization kaiming xavier training-stability dead-neurons vanishing-gradient Weights initialized too small causing vanishing gradients Weights initialized too large causing exploding gradients Activation function not suited to initialization scheme No weight initialization specified in model code ","action":"Use PyTorch default initialization for the layer type","steps":["Use PyTorch default initialization for the layer type","Use kaiming_uniform_ for ReLU networks","Use xavier_uniform_ for tanh/sigmoid networks","Use torch.nn.init.normal_ with mean=0, std=0.02 for transformers","Check activation function compatibility: ReLU needs He init, tanh needs Xavier init"]},{"slug":"nccl-gpu-direct-roce","title":"NCCL GPUDirect over RoCE","category":"Communication","text":"NCCL GPUDirect over RoCE Communication NCCL WARN RoCE is not enabled GPUDirect RDMA is not available on RoCE NCCL performance is 10x slower than expected roce rdma nccl gpudirect ethernet distributed Multi-node training is slow on Ethernet fabric GPUDirect RDMA not working on RoCE NCCL performance is poor on RoCE RoCE not enabled on the NIC GPUDirect RDMA not configured for RoCE PFC and ECN not configured on switches NCCL not configured for RoCE transport","anchorText":"NCCL GPUDirect over RoCE NCCL WARN RoCE is not enabled GPUDirect RDMA is not available on RoCE NCCL performance is 10x slower than expected roce rdma nccl gpudirect ethernet distributed RoCE not enabled on the NIC GPUDirect RDMA not configured for RoCE PFC and ECN not configured on switches NCCL not configured for RoCE transport ","action":"Enable RoCE on the NIC: sudo cma_roce_mode -d mlx5_0 -m 2","steps":["Enable RoCE on the NIC: sudo cma_roce_mode -d mlx5_0 -m 2","Configure PFC: mlnx_qos -i eth0 --pfc 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1","Use NCCL_IB_HCA to specify the RoCE-capable NIC","Enable GPUDirect RDMA: sudo modprobe nvidia_peermem","Test with NCCL_DEBUG_SUBSYS=TRANSPORT,ALL"]},{"slug":"activation-checkpoint-misuse","title":"Activation Checkpoint Misuse","category":"Memory","text":"Activation Checkpoint Misuse Memory OOM error even with activation checkpointing Training is 2-3x slower with checkpointing RuntimeError: checkpoint failed activation-checkpointing gradient-checkpointing memory optimization oom OOM despite enabling activation checkpointing Training is much slower with checkpointing Gradient computation fails with checkpointing Checkpointing only some layers not enough Checkpointing all layers too slow use_reentrant=False not specified when required Checkpointing incompatible with model architecture","anchorText":"Activation Checkpoint Misuse OOM error even with activation checkpointing Training is 2-3x slower with checkpointing RuntimeError: checkpoint failed activation-checkpointing gradient-checkpointing memory optimization oom Checkpointing only some layers not enough Checkpointing all layers too slow use_reentrant=False not specified when required Checkpointing incompatible with model architecture ","action":"Checkpoint every N layers (start with N=1, increase if OOM)","steps":["Checkpoint every N layers (start with N=1, increase if OOM)","Set use_reentrant=False for newer PyTorch versions","Profile memory to find optimal checkpoint interval","Test with small input first to verify correctness","Use gradient checkpointing only for memory-bound layers"]},{"slug":"ema-decay-too-high","title":"EMA Decay Too High","category":"Training Stability","text":"EMA Decay Too High Training Stability EMA model parameters are very different from training model EMA model doesn't improve generalization EMA model is identical to initial model ema exponential-moving-average model-averaging training-stability generalization EMA model is too slow to track training model EMA model is too noisy EMA model diverges from training model EMA decay too high (e.g., 0.9999) means EMA updates too slowly EMA decay too low (e.g., 0.9) means EMA tracks noisy training EMA momentum not tuned for training duration EMA not applied to all parameters","anchorText":"EMA Decay Too High EMA model parameters are very different from training model EMA model doesn't improve generalization EMA model is identical to initial model ema exponential-moving-average model-averaging training-stability generalization EMA decay too high (e.g., 0.9999) means EMA updates too slowly EMA decay too low (e.g., 0.9) means EMA tracks noisy training EMA momentum not tuned for training duration EMA not applied to all parameters ","action":"Use standard EMA decay 0.999 for most training","steps":["Use standard EMA decay 0.999 for most training","Use 0.9999 for very long training (millions of steps)","Use 0.99 for short training or fine-tuning","Tune EMA decay with validation performance","Save EMA model separately and evaluate independently"]},{"slug":"transformers-version-mismatch","title":"Transformers Version Mismatch","category":"Environment","text":"Transformers Version Mismatch Environment Model output differs between training and inference Transformers deprecation warnings Models fail to load with newer transformers version transformers version-mismatch inference environment compatibility Model produces different outputs in different environments Pretrained model loads but performs differently New transformers version breaks old code Transformers version changed between training and inference Breaking changes in transformers API Tokenizer behavior changed in new version Model config format updated","anchorText":"Transformers Version Mismatch Model output differs between training and inference Transformers deprecation warnings Models fail to load with newer transformers version transformers version-mismatch inference environment compatibility Transformers version changed between training and inference Breaking changes in transformers API Tokenizer behavior changed in new version Model config format updated ","action":"Pin transformers version: pip install transformers==4.36.0","steps":["Pin transformers version: pip install transformers==4.36.0","Test model in same environment as training","Use HuggingFace model versioning","Check transformers release notes for breaking changes","Use Denpex to detect transformers version mismatches between training and inference"]},{"slug":"gradient-accumulation-misuse","title":"Gradient Accumulation Misuse","category":"Training Stability","text":"Gradient Accumulation Misuse Training Stability Loss is inconsistent across accumulation steps Effective batch size doesn't match expected OOM during gradient accumulation gradient-accumulation batch-size memory training-stability optimization Loss decreases but model doesn't improve Loss is noisier than expected OOM with gradient accumulation Gradient accumulation steps not matching effective batch size Loss scaling not adjusted for accumulation steps Normalization not applied correctly Gradient accumulation incompatible with certain optimizers","anchorText":"Gradient Accumulation Misuse Loss is inconsistent across accumulation steps Effective batch size doesn't match expected OOM during gradient accumulation gradient-accumulation batch-size memory training-stability optimization Gradient accumulation steps not matching effective batch size Loss scaling not adjusted for accumulation steps Normalization not applied correctly Gradient accumulation incompatible with certain optimizers ","action":"Match effective batch size: batch_size * accumulation_steps","steps":["Match effective batch size: batch_size * accumulation_steps","Scale loss by accumulation_steps","Normalize gradients before accumulation","Test with small accumulation steps first","Verify gradient accumulation with single-step comparison"]},{"slug":"audio-data-loading","title":"Audio Data Loading Error","category":"Data Pipeline","text":"Audio Data Loading Error Data Pipeline torchaudio.load() raises error librosa.load() returns empty SoundFile: file does not exist or is not a valid audio file audio torchaudio librosa data-pipeline speech audio-data Audio loading fails with format error Training stalls on audio file Audio model produces distorted output Audio file is corrupted or truncated Audio format not supported by library Sample rate mismatch between dataset and model Missing audio decoding libraries (ffmpeg, sox)","anchorText":"Audio Data Loading Error torchaudio.load() raises error librosa.load() returns empty SoundFile: file does not exist or is not a valid audio file audio torchaudio librosa data-pipeline speech audio-data Audio file is corrupted or truncated Audio format not supported by library Sample rate mismatch between dataset and model Missing audio decoding libraries (ffmpeg, sox) ","action":"Install audio libraries: pip install torchaudio soundfile librosa","steps":["Install audio libraries: pip install torchaudio soundfile librosa","Install system audio: sudo apt install ffmpeg sox","Check audio file integrity: ffprobe file.wav","Resample audio to model expected sample rate","Use Denpex to detect corrupted audio files in dataset"]},{"slug":"cgroup-memory-limit","title":"Cgroup Memory Limit","category":"Reliability","text":"Cgroup Memory Limit Reliability dmesg: memory cgroup out of memory: Killed process Process killed with SIGKILL (exit code 137) kubelet: Out of memory killing container cgroup memory container docker kubernetes oom kill Training crashes with OOM kill Process is killed without error Container exits with 137 (SIGKILL) Container memory limit too low for training workload Memory leak in training process Other processes in container consuming memory GPU memory pinned memory reducing available system memory","anchorText":"Cgroup Memory Limit dmesg: memory cgroup out of memory: Killed process Process killed with SIGKILL (exit code 137) kubelet: Out of memory killing container cgroup memory container docker kubernetes oom kill Container memory limit too low for training workload Memory leak in training process Other processes in container consuming memory GPU memory pinned memory reducing available system memory ","action":"Increase container memory limit","steps":["Increase container memory limit","Set memory limit slightly above training requirement","Profile memory usage before setting limit","Use resource limits that match actual usage","Use Denpex memory monitoring to detect leaks before OOM"]},{"slug":"nccl-p2p-bandwidth","title":"NCCL P2P Bandwidth Issue","category":"Communication","text":"NCCL P2P Bandwidth Issue Communication NCCL performance is poor within a node GPU-to-GPU transfer is slow NCCL warning about P2P fallback nccl p2p bandwidth intra-node nvlink performance Intra-node GPU communication is slow NCCL falls back to slower transport P2P communication not available between some GPU pairs P2P disabled between GPU pairs NVLink not available between some GPUs PCIe topology doesn't support P2P GPU driver doesn't enable P2P","anchorText":"NCCL P2P Bandwidth Issue NCCL performance is poor within a node GPU-to-GPU transfer is slow NCCL warning about P2P fallback nccl p2p bandwidth intra-node nvlink performance P2P disabled between GPU pairs NVLink not available between some GPUs PCIe topology doesn't support P2P GPU driver doesn't enable P2P ","action":"Enable P2P: export NCCL_P2P_LEVEL=SYS","steps":["Enable P2P: export NCCL_P2P_LEVEL=SYS","Check P2P availability: python -c 'import torch; print(torch.cuda.can_device_access_peer(0,1))'","Use NVLink for P2P when available","Configure NCCL to prefer P2P over shared memory","Profile P2P bandwidth with NCCL_DEBUG=INFO"]},{"slug":"nvls-multicast-slot-exhaustion","title":"NVLS Multicast Slot Exhaustion (NVSwitch)","category":"Communication","text":"NVLS Multicast Slot Exhaustion (NVSwitch) Communication Failed to bind NVLink SHARP (NVLS) Multicast memory of size ... : CUDA error 2 'out of memory' Fabric Manager: all the NVSwitch multicast resources/slot ids are used ncclUnhandledCudaError on communicator init nccl nvls nvlink-sharp nvswitch multicast h100 h200 Training crashes when many sub-communicators are created (e.g. per-pair FSDP groups) Crash is fatal even though a non-NVLS path exists Started after upgrading NCCL NCCL aggressively allocates an NVLS multicast group per sub-communicator The ~128 hardware multicast slots on the NVSwitch fill up cuMulticastBindMem failure is treated as fatal with no graceful fallback","anchorText":"NVLS Multicast Slot Exhaustion (NVSwitch) Failed to bind NVLink SHARP (NVLS) Multicast memory of size ... : CUDA error 2 'out of memory' Fabric Manager: all the NVSwitch multicast resources/slot ids are used ncclUnhandledCudaError on communicator init nccl nvls nvlink-sharp nvswitch multicast h100 h200 NCCL aggressively allocates an NVLS multicast group per sub-communicator The ~128 hardware multicast slots on the NVSwitch fill up cuMulticastBindMem failure is treated as fatal with no graceful fallback # Workaround: disable NVLS multicast and use standard NVLink transport\nexport NCCL_NVLS_ENABLE=0\n# Confirm the cause in Fabric Manager logs\njournalctl -u nvidia-fabricmanager | grep -i 'multicast'","action":"Disable NVLS: export NCCL_NVLS_ENABLE=0","steps":["Disable NVLS: export NCCL_NVLS_ENABLE=0","Downgrade to NCCL 2.28.x until a graceful-fallback fix lands","Reduce the number of distinct sub-communicators where possible","Check Fabric Manager logs for slot exhaustion to confirm"]},{"slug":"nccl-roce-qp-timeout-checkpoint","title":"RoCE QP Timeout During Distributed Checkpoint Save","category":"Communication","text":"RoCE QP Timeout During Distributed Checkpoint Save Communication ibv_modify_qp failed with error Connection timed out errno 110 RuntimeError: NCCL Error 2: unhandled system error Hang then timeout during torch.distributed.checkpoint save nccl roce infiniband checkpoint qp-timeout scale Training runs fine but dcp.save() fails at scale (e.g. 8 nodes / 64 GPUs) Smaller node counts (2/4/6) work Failure is in the checkpoint gather, not the training step The gather/star communication pattern of checkpoint save stresses the fabric differently than ring all-reduce RoCE v2 congestion or PFC/ECN misconfiguration at scale Per-rank metadata collection overloads rank 0 / saturates QPs","anchorText":"RoCE QP Timeout During Distributed Checkpoint Save ibv_modify_qp failed with error Connection timed out errno 110 RuntimeError: NCCL Error 2: unhandled system error Hang then timeout during torch.distributed.checkpoint save nccl roce infiniband checkpoint qp-timeout scale The gather/star communication pattern of checkpoint save stresses the fabric differently than ring all-reduce RoCE v2 congestion or PFC/ECN misconfiguration at scale Per-rank metadata collection overloads rank 0 / saturates QPs # Raise RoCE timeout and confirm GID index, then retry the save\nexport NCCL_IB_TIMEOUT=22\nexport NCCL_IB_GID_INDEX=3\n# Confirm it is fabric-related (slow but should not time out):\nexport NCCL_IB_DISABLE=1","action":"Tune RoCE flow control (PFC + ECN/DCQCN) end-to-end","steps":["Tune RoCE flow control (PFC + ECN/DCQCN) end-to-end","Stagger or shard checkpoint writes to reduce simultaneous QP setup","Verify NCCL_IB_GID_INDEX is correct on every node","Raise IB/RoCE timeout (NCCL_IB_TIMEOUT) and retry count","Test with NCCL_IB_DISABLE=1 to confirm it is fabric-related"]},{"slug":"nccl-cuda-failure-999","title":"ncclUnhandledCudaError: Cuda failure 999 'unknown error'","category":"Communication","text":"ncclUnhandledCudaError: Cuda failure 999 'unknown error' Communication ncclUnhandledCudaError: Call to CUDA function failed Last error: Cuda failure 999 'unknown error' Crash inside _verify_param_shape_across_processes nccl cuda error-999 ddp initialization driver-mismatch DDP/NCCL init crashes on one or more ranks Error is generic with no specific cause Often during parameter-shape verification at DDP construction A GPU is in an unrecoverable state (prior Xid, ECC, or hung context) CUDA driver / runtime / NCCL version mismatch across ranks GPU memory pressure or a leaked context before init","anchorText":"ncclUnhandledCudaError: Cuda failure 999 'unknown error' ncclUnhandledCudaError: Call to CUDA function failed Last error: Cuda failure 999 'unknown error' Crash inside _verify_param_shape_across_processes nccl cuda error-999 ddp initialization driver-mismatch A GPU is in an unrecoverable state (prior Xid, ECC, or hung context) CUDA driver / runtime / NCCL version mismatch across ranks GPU memory pressure or a leaked context before init ","action":"Run with NCCL_DEBUG=INFO to capture the failing rank and call","steps":["Run with NCCL_DEBUG=INFO to capture the failing rank and call","Verify every GPU is healthy: nvidia-smi -q and check dmesg for Xid","Align NCCL, torch, and CUDA driver versions across all ranks","Reset GPUs (nvidia-smi --gpu-reset) or reboot a node showing a faulted GPU","Reduce memory pressure / clear stale processes before launch"]},{"slug":"nvlink-xid31-mmu-fault","title":"NVLink/NVLS Failure with Xid 31 MMU Fault Cascade","category":"Hardware","text":"NVLink/NVLS Failure with Xid 31 MMU Fault Cascade Hardware transport/nvls.cc:244 NCCL WARN Cuda failure 1 'invalid argument' Xid (PCI:...): 31 ... MMU Fault: ENGINE GRAPHICS ... FAULT_PDE ACCESS_TYPE_VIRT_WRITE an illegal memory access was encountered = 700 nvlink nvls xid-31 mmu-fault fabric-manager hopper NCCL/NVLS operations fail across the whole node Multiple frameworks (NCCL tests, HPL, TF) crash with related errors Failures appear after a Fabric Manager restart Forceful Fabric Manager restarts leave NVLink/NVSwitch state inconsistent MMU page-table (Xid 31) faults follow from the coherency loss NVLS multicast mappings become invalid mid-operation","anchorText":"NVLink/NVLS Failure with Xid 31 MMU Fault Cascade transport/nvls.cc:244 NCCL WARN Cuda failure 1 'invalid argument' Xid (PCI:...): 31 ... MMU Fault: ENGINE GRAPHICS ... FAULT_PDE ACCESS_TYPE_VIRT_WRITE an illegal memory access was encountered = 700 nvlink nvls xid-31 mmu-fault fabric-manager hopper Forceful Fabric Manager restarts leave NVLink/NVSwitch state inconsistent MMU page-table (Xid 31) faults follow from the coherency loss NVLS multicast mappings become invalid mid-operation # Clean recovery sequence after a fabric-manager-induced cascade\nsudo systemctl stop nvidia-fabricmanager\nsudo nvidia-smi --gpu-reset\nsudo systemctl start nvidia-fabricmanager","action":"Stop Fabric Manager, reset the GPUs, then restart Fabric Manager cleanly","steps":["Stop Fabric Manager, reset the GPUs, then restart Fabric Manager cleanly","Reboot the node if the reset does not clear the faults","As a workaround, disable NVLS: export NCCL_NVLS_ENABLE=0","Avoid restarting Fabric Manager while jobs are running"]},{"slug":"nccl-broadcast-coalesced-invalid-arg","title":"broadcast_coalesced Fails with Cuda failure 1 'invalid argument'","category":"Communication","text":"broadcast_coalesced Fails with Cuda failure 1 'invalid argument' Communication RuntimeError: NCCL Error 1: unhandled cuda error NCCL WARN Cuda failure 1 'invalid argument' Socket recv failed while polling for opId nccl dataparallel broadcast-coalesced invalid-argument ddp Crash during model replication at the end of an epoch Happens inside torch._C._broadcast_coalesced Uses nn.DataParallel rather than DDP nn.DataParallel re-replicates the model and re-inits NCCL state per step SharedInit fails with an invalid CUDA argument under the launcher's device/visibility setup Often a CUDA_VISIBLE_DEVICES / context mismatch inside the wrapper","anchorText":"broadcast_coalesced Fails with Cuda failure 1 'invalid argument' RuntimeError: NCCL Error 1: unhandled cuda error NCCL WARN Cuda failure 1 'invalid argument' Socket recv failed while polling for opId nccl dataparallel broadcast-coalesced invalid-argument ddp nn.DataParallel re-replicates the model and re-inits NCCL state per step SharedInit fails with an invalid CUDA argument under the launcher's device/visibility setup Often a CUDA_VISIBLE_DEVICES / context mismatch inside the wrapper ","action":"Migrate from nn.DataParallel to DistributedDataParallel (DDP)","steps":["Migrate from nn.DataParallel to DistributedDataParallel (DDP)","Run with NCCL_DEBUG=INFO to capture the failing SharedInit","Ensure CUDA_VISIBLE_DEVICES and device_ids are consistent in the launcher","Pin a known-good NCCL version"]},{"slug":"fsdp-nccl-system-error-ib","title":"FSDP ncclSystemError on InfiniBand (works on TCP)","category":"Communication","text":"FSDP ncclSystemError on InfiniBand (works on TCP) Communication ncclSystemError: System call (e.g. socket, malloc) or external library call failed misc/socket.cc returns error code 2 (ENOENT) on node 2 bootstrap Abort COMPLETE across all 8 GPUs on the failing node nccl fsdp infiniband system-error rdma multi-node Multi-node FSDP hangs/aborts at the first all-gather One node reports the failure, the other looks fine Works when IB is disabled (TCP fallback) Inter-node socket bootstrap fails on one node before the IB path is used NCCL RDMA-SHARP plugin symbol/version mismatch Container network isolation blocks the bootstrap socket","anchorText":"FSDP ncclSystemError on InfiniBand (works on TCP) ncclSystemError: System call (e.g. socket, malloc) or external library call failed misc/socket.cc returns error code 2 (ENOENT) on node 2 bootstrap Abort COMPLETE across all 8 GPUs on the failing node nccl fsdp infiniband system-error rdma multi-node Inter-node socket bootstrap fails on one node before the IB path is used NCCL RDMA-SHARP plugin symbol/version mismatch Container network isolation blocks the bootstrap socket # Pin the bootstrap interface and verify the IB fabric between nodes\nexport NCCL_SOCKET_IFNAME=eth0\nibstat; ib_write_bw -d mlx5_0 # run server/client across the two nodes\n# Confirm IB-specific by forcing TCP:\nexport NCCL_IB_DISABLE=1","action":"Validate IB fabric end-to-end (ibstat, ib_write_bw between the two nodes)","steps":["Validate IB fabric end-to-end (ibstat, ib_write_bw between the two nodes)","Align the NCCL RDMA-SHARP plugin version across nodes/containers","Set NCCL_SOCKET_IFNAME to the correct bootstrap interface","Confirm it is IB-specific by testing with NCCL_IB_DISABLE=1 (TCP)"]},{"slug":"nccl-hang-culprit-rank","title":"Finding the Culprit Rank in an NCCL Hang","category":"Distributed Training","text":"Finding the Culprit Rank in an NCCL Hang Distributed Training GPU utilization 100% but power draw low (~70W) on the stuck rank All ranks blocked in the same collective (AllReduce/ReduceScatter) No NCCL error emitted, watchdog eventually times out nccl hang straggler flight-recorder debugging distributed The whole job hangs with no error Every rank shows the same NCCL wait, so the stack is unhelpful Need to find which specific GPU/rank diverged One rank stopped progressing (data stall, slow I/O, faulted GPU) Synchronous collective semantics block all peers waiting on it The faulted rank is in a different state than the rest","anchorText":"Finding the Culprit Rank in an NCCL Hang GPU utilization 100% but power draw low (~70W) on the stuck rank All ranks blocked in the same collective (AllReduce/ReduceScatter) No NCCL error emitted, watchdog eventually times out nccl hang straggler flight-recorder debugging distributed One rank stopped progressing (data stall, slow I/O, faulted GPU) Synchronous collective semantics block all peers waiting on it The faulted rank is in a different state than the rest # Capture per-rank flight-recorder traces to find the last op each rank ran\nexport TORCH_NCCL_TRACE_BUFFER_SIZE=2000\nexport NCCL_DEBUG=INFO\n# On a hang, compare opCounts across ranks; the lagging rank is the culprit.","action":"Look for the rank with high SM utilization but anomalously low power — that is the spinning/stuck one","steps":["Look for the rank with high SM utilization but anomalously low power — that is the spinning/stuck one","Export per-rank heartbeat / NCCL opCount and find the rank that stopped advancing","Use NCCL_DEBUG=INFO + flight recorder (TORCH_NCCL_TRACE_BUFFER_SIZE) to find the last completed op per rank","Isolate which communicator/group is deadlocked, then inspect that rank's host (dmesg, iostat)"]},{"slug":"nccl-ras-memory-corruption","title":"NCCL RAS Query Crashes Job (Memory Corruption, 2.27.3)","category":"Communication","text":"NCCL RAS Query Crashes Job (Memory Corruption, 2.27.3) Communication recvValueWithTimeout failed on SocketImpl ... Connection was likely closed (TCPStore) RAS VERBOSE STATUS returns corrupted communicator data Job terminates shortly after a RAS query nccl ras memory-corruption tcpstore 2.27.3 monitoring Polling NCCL RAS status crashes a running job RAS reports impossible rank numbers / missing processes High reproduction probability at large scale Memory corruption within the NCCL RAS subsystem on 2.27.3 Corrupted state propagates to TCPStore, dropping connections Triggered by the RAS status query path itself","anchorText":"NCCL RAS Query Crashes Job (Memory Corruption, 2.27.3) recvValueWithTimeout failed on SocketImpl ... Connection was likely closed (TCPStore) RAS VERBOSE STATUS returns corrupted communicator data Job terminates shortly after a RAS query nccl ras memory-corruption tcpstore 2.27.3 monitoring Memory corruption within the NCCL RAS subsystem on 2.27.3 Corrupted state propagates to TCPStore, dropping connections Triggered by the RAS status query path itself ","action":"Do not query NCCL RAS during active training on 2.27.3","steps":["Do not query NCCL RAS during active training on 2.27.3","Upgrade NCCL to a release where the RAS corruption is fixed","Use external health signals (DCGM, dmesg Xid) instead of RAS while on 2.27.3"]},{"slug":"nccl-nvls-invalid-argument","title":"NVLS Cuda failure 1 'invalid argument'","category":"Communication","text":"NVLS Cuda failure 1 'invalid argument' Communication transport/nvls.cc:598 NCCL WARN Cuda failure 1 'invalid argument' transport/nvls.cc NCCL WARN Cuda failure 1 invalid argument NVLS setup warnings repeated across ranks nccl nvls invalid-argument h100 nvswitch nvlink-sharp Repeated NVLS warnings during NCCL setup Collectives degrade or fail on NVSwitch nodes Hard to diagnose from the warning alone NVLS multicast setup receives an invalid argument (driver/CUDA/NCCL mismatch or unsupported config) Closely related to NVSwitch multicast resource handling Often a stack-version or fabric-state issue rather than user code","anchorText":"NVLS Cuda failure 1 'invalid argument' transport/nvls.cc:598 NCCL WARN Cuda failure 1 'invalid argument' transport/nvls.cc NCCL WARN Cuda failure 1 invalid argument NVLS setup warnings repeated across ranks nccl nvls invalid-argument h100 nvswitch nvlink-sharp NVLS multicast setup receives an invalid argument (driver/CUDA/NCCL mismatch or unsupported config) Closely related to NVSwitch multicast resource handling Often a stack-version or fabric-state issue rather than user code # Disable NVLS to confirm it is the source, then collect debug logs\nexport NCCL_NVLS_ENABLE=0\nexport NCCL_DEBUG=INFO","action":"Disable NVLS to confirm and work around: export NCCL_NVLS_ENABLE=0","steps":["Disable NVLS to confirm and work around: export NCCL_NVLS_ENABLE=0","Collect full NCCL_DEBUG=INFO around the NVLS setup","Align CUDA driver, CUDA runtime, and NCCL versions","Check Fabric Manager / NVSwitch health on the node"]},{"slug":"nccl-net-id-0-not-found","title":"NCCL 'Could not find NET with id 0' (NIC Fusion)","category":"Communication","text":"NCCL 'Could not find NET with id 0' (NIC Fusion) Communication ncclInternalError: Internal check failed. Last error: Could not find NET with id 0 Could not find a path for pattern 1, falling back to simple order Intermittent init failures on partial-node jobs nccl nic-fusion net-id kubernetes infiniband 2.26.2 Intermittent (~25%) failures at torch.distributed.barrier()/init Fails on partial-node Kubernetes GPU allocations Started on NCCL 2.26.2 NIC fusion (2.26.2) renames/remaps NET IDs When graph path-finding fails it falls back to hardcoded NET id 0 But fusion may have removed NET/0, so the lookup fails","anchorText":"NCCL 'Could not find NET with id 0' (NIC Fusion) ncclInternalError: Internal check failed. Last error: Could not find NET with id 0 Could not find a path for pattern 1, falling back to simple order Intermittent init failures on partial-node jobs nccl nic-fusion net-id kubernetes infiniband 2.26.2 NIC fusion (2.26.2) renames/remaps NET IDs When graph path-finding fails it falls back to hardcoded NET id 0 But fusion may have removed NET/0, so the lookup fails # Workaround on NCCL 2.26.2: limit NIC fusion so NET/0 survives\nexport NCCL_NET_MERGE_LEVEL=LOC\nexport NCCL_DEBUG=INFO # confirm 'Could not find NET with id 0' disappears","action":"Upgrade to NCCL 2.27.x where NIC fusion is less aggressive and keeps NET/0","steps":["Upgrade to NCCL 2.27.x where NIC fusion is less aggressive and keeps NET/0","Workaround: export NCCL_NET_MERGE_LEVEL=LOC to limit NIC fusion","Pin full-node allocations where feasible to stabilize topology"]},{"slug":"nccl-roce-gid-read-failed","title":"NCCL RoCE GID Read Failed (Invalid argument)","category":"Communication","text":"NCCL RoCE GID Read Failed (Invalid argument) Communication WARN NET/IB: read failed in ncclIbRoceGetVersionNum: Invalid argument GID table has leading zero entries (indices 0-3) RoCE connection abort during init nccl roce gid-index macvlan kubernetes rdma NCCL init aborts on Kubernetes pods with RDMA Happens with macvlan sub-interfaces or virtual NICs Connection abort during communicator setup GID iteration starts from index 0 macvlan/virtual interfaces have all-zero GIDs at indices 0-3, valid entries start at 4 Reading gid_attrs on a zero-GID returns EINVAL","anchorText":"NCCL RoCE GID Read Failed (Invalid argument) WARN NET/IB: read failed in ncclIbRoceGetVersionNum: Invalid argument GID table has leading zero entries (indices 0-3) RoCE connection abort during init nccl roce gid-index macvlan kubernetes rdma GID iteration starts from index 0 macvlan/virtual interfaces have all-zero GIDs at indices 0-3, valid entries start at 4 Reading gid_attrs on a zero-GID returns EINVAL # Find a valid GID index and pin it\nshow_gids # pick a RoCE v2 entry with a real address\nexport NCCL_IB_GID_INDEX=3\nexport NCCL_IB_ROCE_VERSION_NUM=2","action":"Upgrade to NCCL 2.26.2+ which skips zero-GID entries","steps":["Upgrade to NCCL 2.26.2+ which skips zero-GID entries","Set NCCL_IB_GID_INDEX explicitly to a valid index (e.g. 3 or 4)","Optionally pin NCCL_IB_ROCE_VERSION_NUM=2","Confirm valid GID indices with show_gids"]},{"slug":"cuda-fabric-handle-import-101","title":"cuMemImportFromShareableHandle Fails (CUDA error 101)","category":"Hardware","text":"cuMemImportFromShareableHandle Fails (CUDA error 101) Hardware transport/p2p.cc NCCL WARN Cuda failure 101 'invalid device ordinal' cuMemImportFromShareableHandle with CU_MEM_HANDLE_TYPE_FABRIC fails P2P ncclSend/ncclRecv across containers fails cuda fabric mnnvl gh200 error-101 driver-bug P2P send/recv between containers on the same GH200 node fails Both containers see the GPU as device 0 Error is in fabric handle import CUDA UMD bug in driver 570.00 Fabric handle export/import across container/PID namespaces breaks when both see the GPU as device 0 Not an NCCL bug","anchorText":"cuMemImportFromShareableHandle Fails (CUDA error 101) transport/p2p.cc NCCL WARN Cuda failure 101 'invalid device ordinal' cuMemImportFromShareableHandle with CU_MEM_HANDLE_TYPE_FABRIC fails P2P ncclSend/ncclRecv across containers fails cuda fabric mnnvl gh200 error-101 driver-bug CUDA UMD bug in driver 570.00 Fabric handle export/import across container/PID namespaces breaks when both see the GPU as device 0 Not an NCCL bug ","action":"Upgrade the CUDA driver beyond 570.00","steps":["Upgrade the CUDA driver beyond 570.00","Ensure consistent device enumeration across containers","As a stopgap, avoid cross-container P2P on the affected driver"]},{"slug":"nccl-commsplit-segfault-nonblocking","title":"ncclCommSplit Segfault with Non-Blocking Init","category":"Communication","text":"ncclCommSplit Segfault with Non-Blocking Init Communication SIGSEGV in ncclGroupCommJoin at include/group.h GDB shows *pp = 0x1 (dangling pointer) in group traversal Crash after ncclCommSplit + ncclAllGather nccl commsplit segfault non-blocking threads 2.26.2 Segfault during AllGather after ncclCommSplit Only with non-blocking communicator init + threads Setting blocking mode avoids it ncclCommEnsureReady prematurely completed a groupJob This happened before ncclCommGetAsyncError was called Leaving a dangling pointer in the group chain","anchorText":"ncclCommSplit Segfault with Non-Blocking Init SIGSEGV in ncclGroupCommJoin at include/group.h GDB shows *pp = 0x1 (dangling pointer) in group traversal Crash after ncclCommSplit + ncclAllGather nccl commsplit segfault non-blocking threads 2.26.2 ncclCommEnsureReady prematurely completed a groupJob This happened before ncclCommGetAsyncError was called Leaving a dangling pointer in the group chain # Workaround on affected NCCL versions: force blocking init\nexport NCCL_COMM_BLOCKING=1","action":"Upgrade to NCCL 2.26.2 (group-job completion moved to ncclCommGetAsyncError)","steps":["Upgrade to NCCL 2.26.2 (group-job completion moved to ncclCommGetAsyncError)","Workaround: set NCCL_COMM_BLOCKING=1 to force blocking init","Avoid mixing threads with non-blocking init on affected versions"]},{"slug":"nccl-ras-init-segfault","title":"NCCL RAS Race Segfault During Initialization","category":"Communication","text":"NCCL RAS Race Segfault During Initialization Communication SIGSEGV at address 0x10 during NCCL init Backtrace in rasCollCommsInit() Crash when a RAS command runs during bootstrap nccl ras segfault initialization race-condition 2.27.3 Segfault during NCCL init when RAS is queried concurrently Only rank 0 crashes Happens with the RAS feature (NCCL 2.26+) RAS report generation traversed peerInfo before bootstrapAllGather populated it Race between RAS access and peer-info initialization Unguarded read of uninitialized peer data","anchorText":"NCCL RAS Race Segfault During Initialization SIGSEGV at address 0x10 during NCCL init Backtrace in rasCollCommsInit() Crash when a RAS command runs during bootstrap nccl ras segfault initialization race-condition 2.27.3 RAS report generation traversed peerInfo before bootstrapAllGather populated it Race between RAS access and peer-info initialization Unguarded read of uninitialized peer data ","action":"Upgrade to NCCL 2.27.3 (adds a peerInfoValid atomic checked before RAS access)","steps":["Upgrade to NCCL 2.27.3 (adds a peerInfoValid atomic checked before RAS access)","Do not issue RAS queries during communicator init","Delay health polling until after init completes"]},{"slug":"nccl-onerank-reduce-sdc","title":"Single-Rank Reduce-Scatter Silent Data Corruption","category":"Data Integrity","text":"Single-Rank Reduce-Scatter Silent Data Corruption Data Integrity Trailing elements equal 0.0 in reduce_scatter output dist.reduce_scatter_tensor with ReduceOp.AVG on world_size=1 Silent: no exception, no NCCL warning nccl sdc reduce-scatter silent-corruption data-integrity 2.29.7 Reduce output has trailing zeros that should be real values Only with non-power-of-2 / non-aligned tensor sizes No error or warning is emitted Element distribution used integer floor division nElts/bn before alignment The last block was assigned zero elements and skipped Result: trailing elements never written","anchorText":"Single-Rank Reduce-Scatter Silent Data Corruption Trailing elements equal 0.0 in reduce_scatter output dist.reduce_scatter_tensor with ReduceOp.AVG on world_size=1 Silent: no exception, no NCCL warning nccl sdc reduce-scatter silent-corruption data-integrity 2.29.7 Element distribution used integer floor division nElts/bn before alignment The last block was assigned zero elements and skipped Result: trailing elements never written # Detection: verify no unexpected trailing zeros after reduce_scatter\nimport torch, torch.distributed as dist\nout = torch.empty(shard, device='cuda')\ndist.reduce_scatter_tensor(out, inp, op=dist.ReduceOp.AVG)\nassert out[-8:].abs().sum() > 0, 'possible NCCL one-rank reduce SDC'","action":"Upgrade to NCCL 2.29.7 (uses ceiling division divUp(nElts, bn))","steps":["Upgrade to NCCL 2.29.7 (uses ceiling division divUp(nElts, bn))","Until upgraded, pad tensors to aligned sizes for world_size=1 reductions","Add a checksum/parity validation on critical reduce outputs"]},{"slug":"deepspeed-param-in-flight-save","title":"DeepSpeed ZeRO-3: Cannot partition a param in flight (save)","category":"Reliability","text":"DeepSpeed ZeRO-3: Cannot partition a param in flight (save) Reliability AssertionError: Cannot partition a param in flight Crash in _zero3_consolidated_fp16_state_dict() save_16bit_model in the stack trace deepspeed zero-3 checkpoint inflight-param save Checkpoint save crashes under ZeRO-3 Happens in save_fp16_model / save_16bit_model Save interval not aligned to grad accumulation save_16bit_model() lacked the save_checkpoint_prologue() call that save_checkpoint() has Prologue partitions all parameters before saving Without it, some params remain INFLIGHT and the partition assert fires","anchorText":"DeepSpeed ZeRO-3: Cannot partition a param in flight (save) AssertionError: Cannot partition a param in flight Crash in _zero3_consolidated_fp16_state_dict() save_16bit_model in the stack trace deepspeed zero-3 checkpoint inflight-param save save_16bit_model() lacked the save_checkpoint_prologue() call that save_checkpoint() has Prologue partitions all parameters before saving Without it, some params remain INFLIGHT and the partition assert fires ","action":"Upgrade DeepSpeed to a version including PR #1741 (prologue added to save_16bit_model)","steps":["Upgrade DeepSpeed to a version including PR #1741 (prologue added to save_16bit_model)","Align save interval to a multiple of gradient_accumulation_steps","Use save_checkpoint() (which runs the prologue) instead of save_16bit_model where possible"]},{"slug":"deepspeed-inflight-params-backward","title":"DeepSpeed ZeRO-3: still have inflight params (backward)","category":"Reliability","text":"DeepSpeed ZeRO-3: still have inflight params (backward) Reliability RuntimeError: still have inflight params Stack trace through partitioned_param_coordinator.py reset_step() ZeRO-3 with dynamic forward graphs deepspeed zero-3 inflight-param backward rlhf engine.backward() raises about inflight params Models with dynamic/conditional forward graphs Common in RLHF training reset_step() finds params still INFLIGHT from the forward pass Different modules execute on different steps with dynamic graphs The fetch queue has unresolved operations from a previous forward path","anchorText":"DeepSpeed ZeRO-3: still have inflight params (backward) RuntimeError: still have inflight params Stack trace through partitioned_param_coordinator.py reset_step() ZeRO-3 with dynamic forward graphs deepspeed zero-3 inflight-param backward rlhf reset_step() finds params still INFLIGHT from the forward pass Different modules execute on different steps with dynamic graphs The fetch queue has unresolved operations from a previous forward path ","action":"Upgrade to a DeepSpeed version that cleans up inflight params when the fetch queue has stale ops","steps":["Upgrade to a DeepSpeed version that cleans up inflight params when the fetch queue has stale ops","Ensure all parameters used in forward are actually exercised each step","For RLHF, keep the active module set consistent or flush coordinator state between phases"]},{"slug":"deepspeed-nvme-checkpoint-race","title":"DeepSpeed ZeRO-3 NVMe Offload Checkpoint Race (FileExistsError)","category":"Reliability","text":"DeepSpeed ZeRO-3 NVMe Offload Checkpoint Race (FileExistsError) Reliability FileExistsError: [Errno 17] File exists Error at shutil.copytree() in the consolidation step ZeRO-3 + NVMe offload only deepspeed zero-3 nvme-offload checkpoint race-condition save_checkpoint crashes with ZeRO-3 + NVMe offload Only happens multi-GPU Requires gather-16bit-weights on save All ranks create/copy into the same offloaded_tensors/ directory copytree on an existing dir raises FileExistsError No per-rank isolation of the consolidation path","anchorText":"DeepSpeed ZeRO-3 NVMe Offload Checkpoint Race (FileExistsError) FileExistsError: [Errno 17] File exists Error at shutil.copytree() in the consolidation step ZeRO-3 + NVMe offload only deepspeed zero-3 nvme-offload checkpoint race-condition All ranks create/copy into the same offloaded_tensors/ directory copytree on an existing dir raises FileExistsError No per-rank isolation of the consolidation path ","action":"Upgrade to a DeepSpeed version using per-rank subdirectories (offloaded_tensors/rank<N>/)","steps":["Upgrade to a DeepSpeed version using per-rank subdirectories (offloaded_tensors/rank<N>/)","Serialize the save so only rank 0 consolidates","Clear stale offloaded_tensors/ between runs"]},{"slug":"deepspeed-host-ram-oom-offload","title":"DeepSpeed ZeRO Offload Host-RAM OOM (SIGKILL -9)","category":"Memory","text":"DeepSpeed ZeRO Offload Host-RAM OOM (SIGKILL -9) Memory exits with return code = -9 (SIGKILL) free -h shows CPU RAM dropping toward 0 pin_memory: true in the offload config deepspeed zero-offload host-ram oom sigkill pin-memory Process killed with return code -9 and no traceback CPU RAM fills up over seconds then the job dies Only with ZeRO CPU offload enabled Each rank materializes its params + optimizer states in CPU memory pin_memory: true pre-allocates large pinned buffers Total host memory demand exceeds available RAM","anchorText":"DeepSpeed ZeRO Offload Host-RAM OOM (SIGKILL -9) exits with return code = -9 (SIGKILL) free -h shows CPU RAM dropping toward 0 pin_memory: true in the offload config deepspeed zero-offload host-ram oom sigkill pin-memory Each rank materializes its params + optimizer states in CPU memory pin_memory: true pre-allocates large pinned buffers Total host memory demand exceeds available RAM // ds_config.json — reduce host-RAM pressure from offload\n{\n \"zero_optimization\": {\n \"stage\": 3,\n \"offload_optimizer\": { \"device\": \"cpu\", \"pin_memory\": false },\n \"offload_param\": { \"device\": \"cpu\", \"pin_memory\": false },\n \"stage3_max_live_parameters\": 6e8\n }\n}","action":"Set pin_memory: false in both offload entries","steps":["Set pin_memory: false in both offload entries","Disable CPU offload if host RAM is insufficient (fit on GPU/more nodes)","Reduce stage3_max_live_parameters and buffer sizes","Scale to more nodes to spread the host-memory footprint"]},{"slug":"deepspeed-loss-scale-minimum","title":"DeepSpeed fp16: Current loss scale already at minimum","category":"Training Stability","text":"DeepSpeed fp16: Current loss scale already at minimum Training Stability Exception: Current loss scale already at minimum - cannot decrease scale anymore overflow counter increments every step Loss scale monotonically drops from 2^16 to 1 deepspeed fp16 loss-scale overflow bf16 mixed-precision Loss scale decreases every step until training exits Common fine-tuning bf16-pretrained models (e.g. LLaMA2) in fp16 On V100-class hardware without bf16 Continuous gradient overflow in fp16 (narrow exponent range) Loss-scale hysteresis cannot recover under constant overflow Model weights/activations exceed fp16 dynamic range","anchorText":"DeepSpeed fp16: Current loss scale already at minimum Exception: Current loss scale already at minimum - cannot decrease scale anymore overflow counter increments every step Loss scale monotonically drops from 2^16 to 1 deepspeed fp16 loss-scale overflow bf16 mixed-precision Continuous gradient overflow in fp16 (narrow exponent range) Loss-scale hysteresis cannot recover under constant overflow Model weights/activations exceed fp16 dynamic range ","action":"Use bf16 instead of fp16 (requires A100/A30/H100)","steps":["Use bf16 instead of fp16 (requires A100/A30/H100)","Reduce loss_scale_window (e.g. 100) so it adapts faster","Lower learning rate / add gradient clipping to curb overflow","Reduce block_size / sequence length if activations overflow"]},{"slug":"deepspeed-zero3-weight-not-2d","title":"DeepSpeed ZeRO-3: 'weight' must be 2-D at F.embedding","category":"Reliability","text":"DeepSpeed ZeRO-3: 'weight' must be 2-D at F.embedding Reliability RuntimeError: 'weight' must be 2-D Crash at F.embedding(weight, input) Occurs during model.generate() after ZeRO-3 deepspeed zero-3 embedding inference weight-2d Inference/generation after ZeRO-3 training fails Error at the embedding lookup Direct .weight access on a ZeRO-3 partitioned param ZeRO-3 partitions parameters, flattening the embedding weight to 1-D Direct .weight access during generate() sees the partitioned 1-D tensor Parameter is not gathered before use","anchorText":"DeepSpeed ZeRO-3: 'weight' must be 2-D at F.embedding RuntimeError: 'weight' must be 2-D Crash at F.embedding(weight, input) Occurs during model.generate() after ZeRO-3 deepspeed zero-3 embedding inference weight-2d ZeRO-3 partitions parameters, flattening the embedding weight to 1-D Direct .weight access during generate() sees the partitioned 1-D tensor Parameter is not gathered before use ","action":"Upgrade to DeepSpeed >= 0.9.5","steps":["Upgrade to DeepSpeed >= 0.9.5","Save with trainer.save_model() rather than model.save_pretrained()","Gather params before direct access (GatheredParameters) for inference"]},{"slug":"deepspeed-zero3-prefetch-slow","title":"DeepSpeed ZeRO-3 Slow (Synchronous Param Prefetch)","category":"Performance","text":"DeepSpeed ZeRO-3 Slow (Synchronous Param Prefetch) Performance Total pre-fetch time dominates the step profile (e.g. 4.4s of 6.2s) _prefetch_params_non_blocking accounts for ~72% of step time ZeRO-3 ~2x slower than ZeRO-2 deepspeed zero-3 performance prefetch throughput ZeRO-3 is much slower than ZeRO-2 for the same model Profiling shows most step time in param prefetch Communication does not overlap compute The non-blocking prefetch effectively transfers synchronously Param gather does not overlap with computation Default bucket/live-param settings too small for the model","anchorText":"DeepSpeed ZeRO-3 Slow (Synchronous Param Prefetch) Total pre-fetch time dominates the step profile (e.g. 4.4s of 6.2s) _prefetch_params_non_blocking accounts for ~72% of step time ZeRO-3 ~2x slower than ZeRO-2 deepspeed zero-3 performance prefetch throughput The non-blocking prefetch effectively transfers synchronously Param gather does not overlap with computation Default bucket/live-param settings too small for the model // ds_config.json — let ZeRO-3 prefetch overlap compute\n{\n \"zero_optimization\": {\n \"stage\": 3,\n \"stage3_prefetch_bucket_size\": 5e8,\n \"stage3_max_live_parameters\": 1e9,\n \"stage3_param_persistence_threshold\": 1e5\n }\n}","action":"Increase stage3_prefetch_bucket_size and stage3_max_live_parameters","steps":["Increase stage3_prefetch_bucket_size and stage3_max_live_parameters","Tune stage3_param_persistence_threshold to keep small params resident","Upgrade DeepSpeed for prefetch overlap improvements","If the model fits, prefer ZeRO-2 for throughput"]},{"slug":"nccl-nvls-dualport-corruption","title":"NCCL NVLS Memory Corruption with Dual-Port NICs","category":"Communication","text":"NCCL NVLS Memory Corruption with Dual-Port NICs Communication free(): invalid next size (fast) all_reduce_perf hang or crash NVLS enabled + dual-port NICs (2 ports per HCA) nccl nvls dual-port-nic memory-corruption 2.19.4 all_reduce_perf hangs or crashes with dual-port NICs Regression from NCCL 2.16.5 to 2.18.5/2.19.3 Only with NVLS enabled NVLS lacked dual-port NIC transmission support Duplicate head-rank entries in the proxy loop Heap corruption from the duplicate entries","anchorText":"NCCL NVLS Memory Corruption with Dual-Port NICs free(): invalid next size (fast) all_reduce_perf hang or crash NVLS enabled + dual-port NICs (2 ports per HCA) nccl nvls dual-port-nic memory-corruption 2.19.4 NVLS lacked dual-port NIC transmission support Duplicate head-rank entries in the proxy loop Heap corruption from the duplicate entries ","action":"Upgrade to NCCL 2.19.4+ (dual-port NVLS fix)","steps":["Upgrade to NCCL 2.19.4+ (dual-port NVLS fix)","As a workaround, disable NVLS (NCCL_NVLS_ENABLE=0) or use a single port","Pin a known-good NCCL version"]},{"slug":"nccl-nic-ordering-segfault","title":"NCCL Random Segfault from Out-of-Order NIC Names","category":"Communication","text":"NCCL Random Segfault from Out-of-Order NIC Names Communication Random segfault during multi-node nccl-tests ibstat shows out-of-order NIC names (mlx5_3 before mlx5_0) Out-of-bounds access during NIC setup nccl infiniband nic-ordering segfault topology 2.23.4 Random segfaults during nccl-tests across 4+ H100 nodes Inconsistent — sometimes works, sometimes crashes Single-node runs are fine One node enumerates NICs out of order NCCL topology detection assumes ordered NIC names Out-of-bounds access during NIC setup on the misordered node","anchorText":"NCCL Random Segfault from Out-of-Order NIC Names Random segfault during multi-node nccl-tests ibstat shows out-of-order NIC names (mlx5_3 before mlx5_0) Out-of-bounds access during NIC setup nccl infiniband nic-ordering segfault topology 2.23.4 One node enumerates NICs out of order NCCL topology detection assumes ordered NIC names Out-of-bounds access during NIC setup on the misordered node # Identify the misordered NIC, then exclude it\nibstat | grep -i 'CA '\nexport NCCL_IB_HCA=^mlx5_3 # exclude the out-of-order device","action":"Exclude the misordered NIC: NCCL_IB_HCA=^mlx5_3","steps":["Exclude the misordered NIC: NCCL_IB_HCA=^mlx5_3","Physically reseat NICs to restore ordering","Upgrade to NCCL 2.23.4 which handles out-of-order names"]},{"slug":"nccl-socket-connection-abort","title":"NCCL socketStartConnect: Software caused connection abort","category":"Communication","text":"NCCL socketStartConnect: Software caused connection abort Communication Nccl socketStartConnect: Connect to x.x.x.x failed : Software caused connection abort Intermittent during communicator init Only on retry after an initial failure nccl socket connection-abort retry 2.24 Intermittent failures during NCCL init Connection abort on retry Timing-dependent (depends on first failure) First connect attempt fails (ECONNREFUSED/ETIMEDOUT), leaving the socket half-closed NCCL retries on the same stale fd Retry on the half-closed fd triggers ECONNABORTED","anchorText":"NCCL socketStartConnect: Software caused connection abort Nccl socketStartConnect: Connect to x.x.x.x failed : Software caused connection abort Intermittent during communicator init Only on retry after an initial failure nccl socket connection-abort retry 2.24 First connect attempt fails (ECONNREFUSED/ETIMEDOUT), leaving the socket half-closed NCCL retries on the same stale fd Retry on the half-closed fd triggers ECONNABORTED ","action":"Upgrade to NCCL 2.24 (closes and reopens the socket on retry)","steps":["Upgrade to NCCL 2.24 (closes and reopens the socket on retry)","Reduce transient connect failures (warm up peers, fix DNS/routing)","Retry the job if it is a one-off"]},{"slug":"nccl-blackwell-p2p-unsupported","title":"NCCL P2P Fails on RTX 5090 / Blackwell (SM120)","category":"Communication","text":"NCCL P2P Fails on RTX 5090 / Blackwell (SM120) Communication NCCL P2P connection setup fails on RTX 5090 CUDA device query shows SM120 Non-P2P collectives still function nccl blackwell rtx-5090 sm120 p2p P2P operations fail between two RTX 5090 GPUs Other operations work Blackwell-specific SM120 not handled in NCCL P2P topology detection Shared-memory maximum value wrong for Blackwell P2P path rejects the unrecognized architecture","anchorText":"NCCL P2P Fails on RTX 5090 / Blackwell (SM120) NCCL P2P connection setup fails on RTX 5090 CUDA device query shows SM120 Non-P2P collectives still function nccl blackwell rtx-5090 sm120 p2p SM120 not handled in NCCL P2P topology detection Shared-memory maximum value wrong for Blackwell P2P path rejects the unrecognized architecture ","action":"Use NCCL 2.26.x or newer with Blackwell support","steps":["Use NCCL 2.26.x or newer with Blackwell support","Patch shared-memory configuration for SM120 if building from source","Disable P2P (NCCL_P2P_DISABLE=1) as a functional fallback"]},{"slug":"nccl-mnnvl-stack-overrun","title":"NCCL MNNVL Init Segfault at Large World Size (Stack Overrun)","category":"Communication","text":"NCCL MNNVL Init Segfault at Large World Size (Stack Overrun) Communication SIGSEGV during NCCL communicator initialization Only at world_size >= 44 on GB200 NVL72 Crash in recursive topology traversal nccl mnnvl gb200 stack-overrun world-size 2.28 SIGSEGV during NCCL init at large world size (>=44) Smaller world sizes initialize fine Crash during init, not during collectives Recursive MNNVL topology search consumes deep stack With ulimit -s unlimited the recursion overruns into adjacent memory Stack size insufficient for the recursion depth at scale","anchorText":"NCCL MNNVL Init Segfault at Large World Size (Stack Overrun) SIGSEGV during NCCL communicator initialization Only at world_size >= 44 on GB200 NVL72 Crash in recursive topology traversal nccl mnnvl gb200 stack-overrun world-size 2.28 Recursive MNNVL topology search consumes deep stack With ulimit -s unlimited the recursion overruns into adjacent memory Stack size insufficient for the recursion depth at scale # Counterintuitive fix: a FINITE stack avoids the overrun\nulimit -s 8192\n# (ulimit -s unlimited makes the recursive topology search worse)","action":"Set a finite stack: ulimit -s 8192 (do NOT use unlimited)","steps":["Set a finite stack: ulimit -s 8192 (do NOT use unlimited)","Upgrade to NCCL 2.28+ with non-recursive topology search","Pin ulimit in the launcher/container"]},{"slug":"nccl-proxy-connect-ipv6","title":"NCCL Proxy Connect Failed (IPv6 Interference)","category":"Communication","text":"NCCL Proxy Connect Failed (IPv6 Interference) Communication ncclInternalError: Internal check failed. Last error: Proxy Connect failed IPv6 enabled on host NCCL cannot establish inter-node connections nccl ipv6 proxy-connect sockets dual-stack NCCL fails to connect between nodes Dual-stack (IPv4+IPv6) hosts Proxy thread connection failures NCCL proxy attempts an IPv6 connection The peer only responds on IPv4 Connection negotiation fails","anchorText":"NCCL Proxy Connect Failed (IPv6 Interference) ncclInternalError: Internal check failed. Last error: Proxy Connect failed IPv6 enabled on host NCCL cannot establish inter-node connections nccl ipv6 proxy-connect sockets dual-stack NCCL proxy attempts an IPv6 connection The peer only responds on IPv4 Connection negotiation fails # Force NCCL to use IPv4 sockets only\nexport NCCL_SOCKET_FAMILY=4\nexport NCCL_SOCKET_IFNAME=eth0","action":"Force IPv4: export NCCL_SOCKET_FAMILY=4","steps":["Force IPv4: export NCCL_SOCKET_FAMILY=4","Disable IPv6 on the host network","Pin NCCL_SOCKET_IFNAME to the IPv4 interface"]},{"slug":"nccl-three-nic-hang","title":"NCCL Hangs with Exactly Three InfiniBand NICs","category":"Communication","text":"NCCL Hangs with Exactly Three InfiniBand NICs Communication NCCL hang during bootstrap with 3 HCAs detected Ring/tree topology computation stalls No progress, eventual watchdog timeout nccl infiniband odd-nic hang topology NCCL hangs at init with exactly 3 IB HCAs 1, 2, or 4+ NICs work Path-finding gets stuck NIC selection algorithm mishandles an odd NIC count (3) Ring/tree topology path-finding loops or deadlocks Bootstrap never completes","anchorText":"NCCL Hangs with Exactly Three InfiniBand NICs NCCL hang during bootstrap with 3 HCAs detected Ring/tree topology computation stalls No progress, eventual watchdog timeout nccl infiniband odd-nic hang topology NIC selection algorithm mishandles an odd NIC count (3) Ring/tree topology path-finding loops or deadlocks Bootstrap never completes # Pin to two NICs so the path-finder doesn't deadlock on an odd count\nexport NCCL_IB_HCA=mlx5_0,mlx5_1","action":"Manually select an even number of NICs: NCCL_IB_HCA=mlx5_0,mlx5_1","steps":["Manually select an even number of NICs: NCCL_IB_HCA=mlx5_0,mlx5_1","Upgrade to a NCCL version with improved odd-NIC handling","Disable one NIC to make the count even"]},{"slug":"nvlink-partial-vs-nvswitch","title":"Partial NVLink Failure vs NVSwitch Failure (Localization)","category":"Hardware","text":"Partial NVLink Failure vs NVSwitch Failure (Localization) Hardware DCGM_FI_DEV_NVLINK_BANDWIDTH_TOTAL low on one or all GPUs NVLink error counters incrementing Slow all-reduce localized to one node nvlink nvswitch bandwidth hardware dcgm localization Collective bandwidth drops on a node Unclear if one GPU or the whole switch is at fault Need to localize before draining hardware A single GPU's NVLink connection degraded (one GPU low) Or the shared NVSwitch failed (all 8 GPUs low) Link errors / flaps reduce effective bandwidth","anchorText":"Partial NVLink Failure vs NVSwitch Failure (Localization) DCGM_FI_DEV_NVLINK_BANDWIDTH_TOTAL low on one or all GPUs NVLink error counters incrementing Slow all-reduce localized to one node nvlink nvswitch bandwidth hardware dcgm localization A single GPU's NVLink connection degraded (one GPU low) Or the shared NVSwitch failed (all 8 GPUs low) Link errors / flaps reduce effective bandwidth # Localize: dump per-GPU NVLink bandwidth + error counters\nnvidia-smi nvlink -gt d # throughput per link\nnvidia-smi nvlink -e # error counters per link\n# All 8 GPUs low => NVSwitch; a single GPU low => that GPU's NVLink","action":"Compare DCGM_FI_DEV_NVLINK_BANDWIDTH_TOTAL across all 8 GPUs: all low => NVSwitch problem; one low => that GPU's link","steps":["Compare DCGM_FI_DEV_NVLINK_BANDWIDTH_TOTAL across all 8 GPUs: all low => NVSwitch problem; one low => that GPU's link","Check NVLink error counters (nvidia-smi nvlink -e)","Drain and reset/RMA the implicated component","Re-run nccl-tests all_reduce to confirm recovery"]},{"slug":"param-plane-cable-link-down","title":"Parameter-Plane Cable Link Down","category":"Infrastructure","text":"Parameter-Plane Cable Link Down Infrastructure Link Down event on a parameter-plane cable/port NIC carrier lost / port state DOWN Sudden packet loss to one node network cable link-down infrastructure rdma parameter-plane A node drops out of distributed training RDMA collectives stall or fail on one node Link-state alarms on the backend fabric Cable, transceiver, or port hardware failure Optical module degradation Switch port fault on the backend plane","anchorText":"Parameter-Plane Cable Link Down Link Down event on a parameter-plane cable/port NIC carrier lost / port state DOWN Sudden packet loss to one node network cable link-down infrastructure rdma parameter-plane Cable, transceiver, or port hardware failure Optical module degradation Switch port fault on the backend plane ","action":"Identify the down link (switch port counters, NIC carrier state)","steps":["Identify the down link (switch port counters, NIC carrier state)","Reseat or replace the cable/transceiver","Fail the job over to remaining healthy nodes and drain the affected one","Open a hardware ticket for the cable/port"]},{"slug":"nic-link-degradation","title":"NIC Degradation (Link Speed Low / NIC Lost / GID Error)","category":"Infrastructure","text":"NIC Degradation (Link Speed Low / NIC Lost / GID Error) Infrastructure NIC Link Speed low (e.g. 100G port at 25G) NIC Lost / device disappears from the fabric Rnic Gid Error in logs nic link-speed degradation rdma infrastructure Collective throughput drops without an outright failure A NIC negotiates below its rated speed Intermittent NIC disappearance or GID errors Cable/transceiver degradation forcing a lower negotiated speed Firmware or PCIe issue dropping the NIC GID table / RoCE addressing errors","anchorText":"NIC Degradation (Link Speed Low / NIC Lost / GID Error) NIC Link Speed low (e.g. 100G port at 25G) NIC Lost / device disappears from the fabric Rnic Gid Error in logs nic link-speed degradation rdma infrastructure Cable/transceiver degradation forcing a lower negotiated speed Firmware or PCIe issue dropping the NIC GID table / RoCE addressing errors # Check negotiated link speed and NIC health\nethtool <iface> | grep -i speed\nibstat | grep -iE 'rate|state'","action":"Check negotiated vs rated link speed (ethtool / ibstat)","steps":["Check negotiated vs rated link speed (ethtool / ibstat)","Reseat/replace the transceiver or cable; update NIC firmware","Drain the node if the NIC keeps dropping","Validate GID configuration (show_gids)"]},{"slug":"power-supply-failure","title":"Power Supply Failure / Redundancy Lost","category":"Infrastructure","text":"Power Supply Failure / Redundancy Lost Infrastructure Power Supply Failure detected (BMC/IPMI sensor) PSU redundancy lost Voltage/current sensor out of range power-supply psu ipmi infrastructure hardware Node at risk of sudden power loss PSU redundancy lost (running on one supply) Power-related sensor alarms PSU hardware failure Power feed / PDU issue Thermal or load stress on the supply","anchorText":"Power Supply Failure / Redundancy Lost Power Supply Failure detected (BMC/IPMI sensor) PSU redundancy lost Voltage/current sensor out of range power-supply psu ipmi infrastructure hardware PSU hardware failure Power feed / PDU issue Thermal or load stress on the supply ","action":"Read BMC/IPMI power sensors to confirm the failed PSU","steps":["Read BMC/IPMI power sensors to confirm the failed PSU","Schedule a PSU replacement; drain the node before it hard-fails","Verify the redundant supply is carrying load","Check the rack PDU / power feed"]},{"slug":"fan-speed-critical","title":"Fan Speed Critical / Cooling Redundancy Lost","category":"Infrastructure","text":"Fan Speed Critical / Cooling Redundancy Lost Infrastructure Fan Speed Critical (sensor) Fan redundancy lost Rising GPU temperature with a fan alarm fan cooling thermal ipmi infrastructure GPU temperatures climbing on a node Fan fault or critical-speed sensor alarm Throughput dropping from thermal throttle Fan hardware failure or bearing wear Blocked airflow / dust Cooling redundancy lost","anchorText":"Fan Speed Critical / Cooling Redundancy Lost Fan Speed Critical (sensor) Fan redundancy lost Rising GPU temperature with a fan alarm fan cooling thermal ipmi infrastructure Fan hardware failure or bearing wear Blocked airflow / dust Cooling redundancy lost ","action":"Confirm via BMC fan sensors and GPU temperature trend","steps":["Confirm via BMC fan sensors and GPU temperature trend","Replace the failed fan; clear airflow obstructions","Drain the node if temperatures approach throttle thresholds","Correlate with GPU thermal-throttle events"]},{"slug":"checkpoint-io-stall-nfs","title":"Checkpoint I/O Stall (NFS RPC Saturation)","category":"Infrastructure","text":"Checkpoint I/O Stall (NFS RPC Saturation) Infrastructure NFS RPC slot table saturated (e.g. 128 slots full) 1-10% utilization of a 200 Gbps link during save (bandwidth paradox) Long checkpoint write duration / stall checkpoint nfs io-stall rpc infrastructure storage Checkpoint save takes far longer than expected Network shows low utilization yet I/O is the bottleneck All ranks block on the checkpoint write NFS RPC layer saturates its fixed slot table before bandwidth is used Many ranks issue small synchronous writes at once Metadata/RPC overhead dominates, not throughput","anchorText":"Checkpoint I/O Stall (NFS RPC Saturation) NFS RPC slot table saturated (e.g. 128 slots full) 1-10% utilization of a 200 Gbps link during save (bandwidth paradox) Long checkpoint write duration / stall checkpoint nfs io-stall rpc infrastructure storage NFS RPC layer saturates its fixed slot table before bandwidth is used Many ranks issue small synchronous writes at once Metadata/RPC overhead dominates, not throughput # Raise the NFS RPC slot table to relieve the saturation\nsysctl -w sunrpc.tcp_slot_table_entries=256\n# Then prefer sharded/async checkpointing over a single synchronous write","action":"Increase the NFS RPC slot table (sunrpc.tcp_slot_table_entries)","steps":["Increase the NFS RPC slot table (sunrpc.tcp_slot_table_entries)","Use asynchronous / sharded checkpointing to spread writes","Stage checkpoints to local NVMe then copy out of band","Reduce checkpoint frequency or overlap with compute"]},{"slug":"gsp-rpc-timeout-xid119","title":"GSP RPC Timeout (Xid 119)","category":"Hardware","text":"GSP RPC Timeout (Xid 119) Hardware Xid 119 in dmesg (GSP RPC timeout) nvidia-smi hangs or omits the GPU CUDA operations on the GPU stall xid-119 gsp rpc-timeout hardware gpu-reset A GPU becomes unresponsive mid-run Xid 119 appears in the kernel log CUDA calls to that GPU hang or error GSP firmware stopped responding to driver RPCs Firmware/driver interaction bug or load stall GPU enters an unresponsive state","anchorText":"GSP RPC Timeout (Xid 119) Xid 119 in dmesg (GSP RPC timeout) nvidia-smi hangs or omits the GPU CUDA operations on the GPU stall xid-119 gsp rpc-timeout hardware gpu-reset GSP firmware stopped responding to driver RPCs Firmware/driver interaction bug or load stall GPU enters an unresponsive state ","action":"Reset the GPU (nvidia-smi --gpu-reset) or reboot the node","steps":["Reset the GPU (nvidia-smi --gpu-reset) or reboot the node","Update GPU driver / GSP firmware to a stable combination","Drain the node and re-run a health check before reuse","Consider disabling GSP if a known-bad firmware (vendor guidance)"]},{"slug":"gpu-fallen-off-bus-xid79","title":"GPU Fallen Off the Bus (Xid 79)","category":"Hardware","text":"GPU Fallen Off the Bus (Xid 79) Hardware Xid 79: GPU has fallen off the bus nvidia-smi: GPU is lost / not listed ERR! or missing entry for the GPU xid-79 fallen-off-bus pcie hardware critical A GPU vanishes from nvidia-smi mid-run Training crashes with the GPU unavailable Xid 79 in the kernel log PCIe link lost (power instability, thermal, or physical seating) Failing GPU hardware Riser/connector fault","anchorText":"GPU Fallen Off the Bus (Xid 79) Xid 79: GPU has fallen off the bus nvidia-smi: GPU is lost / not listed ERR! or missing entry for the GPU xid-79 fallen-off-bus pcie hardware critical PCIe link lost (power instability, thermal, or physical seating) Failing GPU hardware Riser/connector fault ","action":"Reboot the node to attempt recovery (reset alone often insufficient)","steps":["Reboot the node to attempt recovery (reset alone often insufficient)","Check power and thermals (correlate with PSU/fan alarms)","Reseat the GPU / riser; if it recurs, RMA the GPU","Drain the node and run diagnostics before reuse"]},{"slug":"memory-row-remapping","title":"GPU Memory Row Remapping Event","category":"Hardware","text":"GPU Memory Row Remapping Event Hardware DCGM row remap count > 0 (or increasing) Xid messages referencing row remapping / ECC Pending remap requires a GPU reset to take effect row-remap ecc memory dcgm hardware ECC errors followed by a memory remap Remap count increasing over time Concern about GPU memory health A memory row developed uncorrectable/excessive errors The GPU remaps it to a spare row (finite supply) Underlying memory degradation","anchorText":"GPU Memory Row Remapping Event DCGM row remap count > 0 (or increasing) Xid messages referencing row remapping / ECC Pending remap requires a GPU reset to take effect row-remap ecc memory dcgm hardware A memory row developed uncorrectable/excessive errors The GPU remaps it to a spare row (finite supply) Underlying memory degradation ","action":"Treat a single remap as informational; reset the GPU to apply a pending remap","steps":["Treat a single remap as informational; reset the GPU to apply a pending remap","Track the remap count trend — rising counts mean degrading memory","Drain and RMA the GPU when remaps accumulate or spares run low","Correlate with ECC DBE (Xid 48/94/95) events"]},{"slug":"sdc-silent-degradation","title":"Silent Data Corruption: Silent Degradation (No NaN)","category":"Data Integrity","text":"Silent Data Corruption: Silent Degradation (No NaN) Data Integrity Loss settles above baseline with no NaN events Parameters drift from a reference run Validation metrics quietly degrade sdc silent-corruption data-integrity parameter-drift hardware Model trains 'fine' but underperforms No NaN, no crash, no error Final quality is quietly worse than expected Hardware fault flips bits in computation without triggering NaN/Inf Errors accumulate as parameter drift, not crashes NaN checks don't catch sub-NaN corruption","anchorText":"Silent Data Corruption: Silent Degradation (No NaN) Loss settles above baseline with no NaN events Parameters drift from a reference run Validation metrics quietly degrade sdc silent-corruption data-integrity parameter-drift hardware Hardware fault flips bits in computation without triggering NaN/Inf Errors accumulate as parameter drift, not crashes NaN checks don't catch sub-NaN corruption # SDC canary: compare a deterministic op against a known-good reference\nimport torch\nout = suspect_module(fixed_input) # on the suspect GPU\nref = out.detach().cpu()\n# rerun the same op/input on a healthy GPU; flag if max-abs diff exceeds tolerance\nassert (ref - golden).abs().max() < 1e-3, 'possible silent data corruption'","action":"Compare parameter L2 distance against a reference/replica run","steps":["Compare parameter L2 distance against a reference/replica run","Periodically recompute a deterministic submodule on suspect vs healthy nodes","Quarantine and swap the suspect GPU/node, then resume from a known-good checkpoint","Add redundant/duplicate computation checks on critical layers"]},{"slug":"sdc-gradual-parameter-drift","title":"Silent Data Corruption: Gradual Parameter Drift","category":"Data Integrity","text":"Silent Data Corruption: Gradual Parameter Drift Data Integrity Monotonically increasing parameter L2 distance from a reference run Loss visually identical to baseline No NaN, no error sdc parameter-drift data-integrity permanent-fault monotonic Loss curve looks indistinguishable from baseline Yet the model's parameters silently diverge Discovered only via cross-run comparison A permanent fault perturbs computation every step Errors accumulate as steadily growing parameter divergence Loss is insensitive enough to mask the drift","anchorText":"Silent Data Corruption: Gradual Parameter Drift Monotonically increasing parameter L2 distance from a reference run Loss visually identical to baseline No NaN, no error sdc parameter-drift data-integrity permanent-fault monotonic A permanent fault perturbs computation every step Errors accumulate as steadily growing parameter divergence Loss is insensitive enough to mask the drift ","action":"Compare parameters against a replica/reference periodically; flag monotonic divergence","steps":["Compare parameters against a replica/reference periodically; flag monotonic divergence","Once detected, identify and quarantine the faulty node","Roll back to a checkpoint taken before divergence began","Use deterministic execution to compare submodule outputs across nodes"]},{"slug":"node-rank-mismatch","title":"Node Rank Mismatch at NCCL Init","category":"Distributed Training","text":"Node Rank Mismatch at NCCL Init Distributed Training NCCL init blocks waiting for ranks that never join Mismatch between WORLD_SIZE and connected peers Hang at the first collective / barrier nccl rank-mismatch world-size ddp initialization Distributed init hangs or fails at startup world_size doesn't match connected peers Common with misconfigured launchers RANK/WORLD_SIZE/MASTER_ADDR misconfigured across nodes A launcher started the wrong number of processes A node failed to launch or register","anchorText":"Node Rank Mismatch at NCCL Init NCCL init blocks waiting for ranks that never join Mismatch between WORLD_SIZE and connected peers Hang at the first collective / barrier nccl rank-mismatch world-size ddp initialization RANK/WORLD_SIZE/MASTER_ADDR misconfigured across nodes A launcher started the wrong number of processes A node failed to launch or register # Fail fast on a rank mismatch instead of hanging forever\nimport os, datetime, torch.distributed as dist\nprint(os.environ['RANK'], os.environ['WORLD_SIZE'], os.environ['MASTER_ADDR'])\ndist.init_process_group('nccl', timeout=datetime.timedelta(minutes=5))","action":"Print RANK, WORLD_SIZE, MASTER_ADDR/PORT on every process and verify consistency","steps":["Print RANK, WORLD_SIZE, MASTER_ADDR/PORT on every process and verify consistency","Ensure exactly world_size processes start and can reach the master","Set a finite init timeout so a mismatch fails fast instead of hanging","Check the launcher (torchrun/SLURM) nnodes/nproc-per-node math"]},{"slug":"roce-mtu-mismatch","title":"RoCE MTU Mismatch — Completion Error 12 / Vendor Err 129","category":"Communication","text":"RoCE MTU Mismatch — Completion Error 12 / Vendor Err 129 Communication NCCL WARN Got completion with error 12, vendor err 129 ibv_post_send fails on specific node pairs but not others NCCL_DEBUG=INFO shows IB transport errors during channel setup ibstat shows active_mtu 2048 but switch port is configured for 4096 or vice versa roce mtu infiniband rdma nccl completion-error vendor-err-129 NCCL init succeeds but the first all-reduce or all-gather fails with completion error 12 Training hangs at NCCL communication with ibv completion errors in dmesg Single-node training works but multi-node RoCE fails immediately The NIC MTU (active_mtu from ibstat) does not match the switch port MTU, causing RDMA packet fragmentation or drops that manifest as completion error 12 Vendor error 129 is Mellanox-specific: the NIC's firmware rejected the packet because the MTU exceeded the port's configured maximum After a switch firmware upgrade, the default MTU may change from 2048 to 4096 without updating the NIC configuration Docker containers may inherit the host MTU but the RoCE device has a different MTU configured","anchorText":"RoCE MTU Mismatch — Completion Error 12 / Vendor Err 129 NCCL WARN Got completion with error 12, vendor err 129 ibv_post_send fails on specific node pairs but not others NCCL_DEBUG=INFO shows IB transport errors during channel setup ibstat shows active_mtu 2048 but switch port is configured for 4096 or vice versa roce mtu infiniband rdma nccl completion-error vendor-err-129 The NIC MTU (active_mtu from ibstat) does not match the switch port MTU, causing RDMA packet fragmentation or drops that manifest as completion error 12 Vendor error 129 is Mellanox-specific: the NIC's firmware rejected the packet because the MTU exceeded the port's configured maximum After a switch firmware upgrade, the default MTU may change from 2048 to 4096 without updating the NIC configuration Docker containers may inherit the host MTU but the RoCE device has a different MTU configured ","action":"Verify MTU on both ends: run ibstat on every node and check active_mtu; compare with switch port MTU configuration","steps":["Verify MTU on both ends: run ibstat on every node and check active_mtu; compare with switch port MTU configuration","Set MTU to 4096 on both NICs and switch ports for optimal RoCE performance: sudo ip link set dev ib0 mtu 4096","After changing MTU, restart the RDMA subsystem: sudo systemctl restart opensmd or reboot the node","Set NCCL_IB_RETRY_CNT=7 and NCCL_IB_TIMEOUT=22 to increase RDMA retry tolerance while the MTU is being fixed","Use Denpex to validate MTU consistency across all nodes before launching multi-node training"]},{"slug":"nccl-mismatched-collective","title":"NCCL Mismatched Collective — Different Ranks Executing Different Operations","category":"Communication","text":"NCCL Mismatched Collective — Different Ranks Executing Different Operations Communication NCCL error: Mismatched collective detected at rank X NCCL_DEBUG=INFO shows different collective types on different ranks at the same step Training hangs before the mismatch error if ranks are waiting at different barriers The error occurs after a conditional branch that differs across ranks nccl collective mismatch ddp fsdp programming-error synchronization Training crashes with 'Mismatched collective' error from NCCL One rank calls all-reduce while another calls all-gather at the same collective step The error is deterministic and reproducible with the same code path A conditional branch (if/else) causes different ranks to execute different collective operations at the same point in the training loop One rank skips a collective (e.g., skips backward pass for a batch with no loss) while other ranks execute it Dynamic model architectures (MoE routing, early exit) produce different collective patterns on different ranks Mismatched data loading: one rank processes a different number of batches, causing it to be one collective ahead or behind","anchorText":"NCCL Mismatched Collective — Different Ranks Executing Different Operations NCCL error: Mismatched collective detected at rank X NCCL_DEBUG=INFO shows different collective types on different ranks at the same step Training hangs before the mismatch error if ranks are waiting at different barriers The error occurs after a conditional branch that differs across ranks nccl collective mismatch ddp fsdp programming-error synchronization A conditional branch (if/else) causes different ranks to execute different collective operations at the same point in the training loop One rank skips a collective (e.g., skips backward pass for a batch with no loss) while other ranks execute it Dynamic model architectures (MoE routing, early exit) produce different collective patterns on different ranks Mismatched data loading: one rank processes a different number of batches, causing it to be one collective ahead or behind ","action":"Ensure all ranks execute the same collectives in the same order: remove conditional collectives or add dummy collectives on the non-active path","steps":["Ensure all ranks execute the same collectives in the same order: remove conditional collectives or add dummy collectives on the non-active path","Use NCCL_DEBUG_SUBSYS=COLL to trace which collectives each rank is calling and find the divergence point","Synchronize data loading: ensure all ranks process the same number of batches per epoch","For MoE: use expert-parallel collectives that are consistent across all ranks regardless of routing","Use Denpex to detect the exact rank and step where the collective mismatch occurs"]},{"slug":"ofi-memlock-exhaustion","title":"OFI Memlock Exhaustion — RDMA Memory Registration Failure","category":"Communication","text":"OFI Memlock Exhaustion — RDMA Memory Registration Failure Communication NCCL WARN NET/OFI Unable to register memory RC:12 NCCL init timeout after the memory registration error Training works when switching to Socket transport: NCCL_NET=Socket ulimit -l shows a low memlock value (e.g., 64KB or 65536) ofi memlock rdma memory-registration nccl efa ulimit NCCL initialization fails with 'Unable to register memory' on the OFI transport Training works with NCCL_NET=Socket but fails with the default OFI/IB transport The error appears on all ranks simultaneously during NCCL channel setup The OS per-process memlock limit (ulimit -l) is too low for RDMA memory registration RDMA requires pinning (locking) GPU memory in physical RAM so the NIC can DMA directly to it The default memlock limit on many Linux distributions is 64KB, which is far too low for NCCL's RDMA buffers Containers inherit the host's memlock limit unless explicitly overridden","anchorText":"OFI Memlock Exhaustion — RDMA Memory Registration Failure NCCL WARN NET/OFI Unable to register memory RC:12 NCCL init timeout after the memory registration error Training works when switching to Socket transport: NCCL_NET=Socket ulimit -l shows a low memlock value (e.g., 64KB or 65536) ofi memlock rdma memory-registration nccl efa ulimit The OS per-process memlock limit (ulimit -l) is too low for RDMA memory registration RDMA requires pinning (locking) GPU memory in physical RAM so the NIC can DMA directly to it The default memlock limit on many Linux distributions is 64KB, which is far too low for NCCL's RDMA buffers Containers inherit the host's memlock limit unless explicitly overridden ","action":"Set memlock to unlimited in /etc/security/limits.conf: add '* soft memlock unlimited' and '* hard memlock unlimited'","steps":["Set memlock to unlimited in /etc/security/limits.conf: add '* soft memlock unlimited' and '* hard memlock unlimited'","For Docker: pass --ulimit memlock=-1:-1 to docker run","For Kubernetes: set container securityContext: capabilities: add: ['IPC_LOCK']","Verify the fix: run ulimit -l in the training process and confirm it shows 'unlimited' or a very large value","Use Denpex to validate memlock configuration across all nodes before launching training"]},{"slug":"docker-shm-exhaustion","title":"Docker /dev/shm Exhaustion — NCCL Shared Memory Allocation Failure","category":"Infrastructure","text":"Docker /dev/shm Exhaustion — NCCL Shared Memory Allocation Failure Infrastructure NCCL WARN posix_fallocate failed: No space left on device ncclShmemOpen failed: cannot allocate shared memory segment df -h /dev/shm shows 64MB (Docker default) and 100% usage Training crashes during NCCL channel or proxy setup docker shm shared-memory nccl container ipc kubernetes NCCL initialization fails with 'No space left on device' inside Docker containers Training works on bare metal but fails inside Docker with shared memory errors The container has plenty of disk space but /dev/shm is exhausted Docker sets /dev/shm to 64MB by default, which is too small for NCCL's inter-process shared memory buffers NCCL allocates shared memory segments for proxy threads, communication channels, and collnet buffers DataLoader with num_workers > 0 also uses /dev/shm for tensor sharing with the main process Multiple NCCL communicators (e.g., FSDP sub-communicators) each allocate their own shared memory segments","anchorText":"Docker /dev/shm Exhaustion — NCCL Shared Memory Allocation Failure NCCL WARN posix_fallocate failed: No space left on device ncclShmemOpen failed: cannot allocate shared memory segment df -h /dev/shm shows 64MB (Docker default) and 100% usage Training crashes during NCCL channel or proxy setup docker shm shared-memory nccl container ipc kubernetes Docker sets /dev/shm to 64MB by default, which is too small for NCCL's inter-process shared memory buffers NCCL allocates shared memory segments for proxy threads, communication channels, and collnet buffers DataLoader with num_workers > 0 also uses /dev/shm for tensor sharing with the main process Multiple NCCL communicators (e.g., FSDP sub-communicators) each allocate their own shared memory segments ","action":"Pass --shm-size=2g (minimum) or --shm-size=8g to docker run for NCCL training","steps":["Pass --shm-size=2g (minimum) or --shm-size=8g to docker run for NCCL training","For Kubernetes: set emptyDir volume with medium: Memory and sizeLimit on /dev/shm","Alternative: use --ipc=host to share the host's /dev/shm (has security implications in multi-tenant environments)","Verify the fix: run df -h /dev/shm inside the container and confirm it shows the increased size","Use Denpex to validate /dev/shm configuration across all containerized training nodes"]},{"slug":"acs-disabled-gdr-failure","title":"ACS Disabled — GPU Direct RDMA Failure (Vendor Err 81)","category":"Communication","text":"ACS Disabled — GPU Direct RDMA Failure (Vendor Err 81) Communication NCCL WARN Got completion with error 4, vendor err 81 NCCL_DEBUG=INFO shows GDR registration failures Performance is significantly worse than expected on RoCE/IB fabrics ibv_devinfo shows the NIC supports GDR but NCCL cannot use it acs gdr gpu-direct-rdma rdma nccl bios pcie vendor-err-81 NCCL RDMA operations fail with completion error 4 and vendor error 81 Multi-node training with GDR enabled crashes or performs very poorly The same training works with GPU Direct RDMA disabled ACS (Access Control Services) is disabled in the server BIOS, which prevents peer-to-peer DMA between the GPU and NIC across the PCIe fabric Without ACS, the IOMMU cannot validate DMA transactions between the GPU and NIC, so the RDMA completion fails with vendor error 81 ACS is sometimes disabled for PCIe passthrough (e.g., for VM device assignment), which has the side effect of breaking GPU Direct RDMA NCCL attempts to use GDR by default on supported hardware, and the failure is not gracefully handled on all NCCL versions","anchorText":"ACS Disabled — GPU Direct RDMA Failure (Vendor Err 81) NCCL WARN Got completion with error 4, vendor err 81 NCCL_DEBUG=INFO shows GDR registration failures Performance is significantly worse than expected on RoCE/IB fabrics ibv_devinfo shows the NIC supports GDR but NCCL cannot use it acs gdr gpu-direct-rdma rdma nccl bios pcie vendor-err-81 ACS (Access Control Services) is disabled in the server BIOS, which prevents peer-to-peer DMA between the GPU and NIC across the PCIe fabric Without ACS, the IOMMU cannot validate DMA transactions between the GPU and NIC, so the RDMA completion fails with vendor error 81 ACS is sometimes disabled for PCIe passthrough (e.g., for VM device assignment), which has the side effect of breaking GPU Direct RDMA NCCL attempts to use GDR by default on supported hardware, and the failure is not gracefully handled on all NCCL versions ","action":"Enable ACS in the server BIOS: look for 'ACS Enable' or 'Access Control Services' under PCIe settings","steps":["Enable ACS in the server BIOS: look for 'ACS Enable' or 'Access Control Services' under PCIe settings","If ACS cannot be enabled: disable GPU Direct RDMA with NCCL_IB_DISABLE=1 or NCCL_NET_GDR_LEVEL=0","After enabling ACS: verify GDR works with NCCL_DEBUG=INFO and check that NCCL uses the NVLS/IB transport with GDR","Update the BIOS to the latest version: some vendors have fixed ACS-related GDR issues in firmware updates","Use Denpex to validate GDR functionality across all nodes and flag ACS misconfiguration"]},{"slug":"nccl-topology-xml-missing-busid","title":"NCCL Topology XML Missing NIC Bus ID","category":"Communication","text":"NCCL Topology XML Missing NIC Bus ID Communication NCCL WARN Attribute busid of node nic not found NCCL uses Socket transport instead of IB despite ibstat showing active NICs NCCL_TOPO_FILE environment variable has no effect or contains invalid bus IDs Performance regression after a driver or firmware update nccl topology busid nic infiniband efa pci NCCL falls back to Socket transport despite InfiniBand NICs being available NCCL_WARN: Attribute busid of node nic not found in topology Multi-node training performance is much worse than expected on IB clusters The NIC's PCI bus ID is not populated in sysfs or the NCCL topology XML, preventing NCCL from associating the NIC with the correct NUMA node and GPU EFA devices on AWS don't have standard PCI bus IDs that NCCL can discover Virtual NICs in VMs may not expose PCI topology information After a driver upgrade, the NIC's PCI path may change, invalidating the cached topology XML","anchorText":"NCCL Topology XML Missing NIC Bus ID NCCL WARN Attribute busid of node nic not found NCCL uses Socket transport instead of IB despite ibstat showing active NICs NCCL_TOPO_FILE environment variable has no effect or contains invalid bus IDs Performance regression after a driver or firmware update nccl topology busid nic infiniband efa pci The NIC's PCI bus ID is not populated in sysfs or the NCCL topology XML, preventing NCCL from associating the NIC with the correct NUMA node and GPU EFA devices on AWS don't have standard PCI bus IDs that NCCL can discover Virtual NICs in VMs may not expose PCI topology information After a driver upgrade, the NIC's PCI path may change, invalidating the cached topology XML ","action":"Set NCCL_TOPO_FILE to a manually created topology XML that includes the correct NIC bus IDs","steps":["Set NCCL_TOPO_FILE to a manually created topology XML that includes the correct NIC bus IDs","Generate the topology XML with NCCL's built-in tool: nvidia-smi -q && python -c 'import torch; torch.distributed.init_process_group()' to trigger topology discovery","Override the NIC bus ID: set NCCL_IB_HCA explicitly to the correct device name (e.g., NCCL_IB_HCA=mlx5_0,mlx5_1)","For EFA: set NCCL_NET=AWS-EFA and ensure the AWS OFI plugin is installed","Use Denpex to validate NCCL topology detection and flag missing bus IDs before training"]},{"slug":"nccl-buffsize-oversized","title":"NCCL BUFFSIZE Oversized — Socket Transport Stall","category":"Communication","text":"NCCL BUFFSIZE Oversized — Socket Transport Stall Communication NCCL hangs during all-reduce or all-gather after NCCL_BUFFSIZE was manually increased NCCL_DEBUG=INFO shows Socket transport timeouts or retransmissions Watchdog timeout after increasing NCCL_BUFFSIZE Training works with default NCCL_BUFFSIZE but hangs with custom value nccl buffsize socket buffer configuration tuning Training hangs at NCCL communication after setting NCCL_BUFFSIZE to a very large value Socket transport stalls while IB transport works fine with the same buffer size Increasing NCCL_BUFFSIZE made performance worse instead of better NCCL_BUFFSIZE controls the per-channel buffer size for NCCL communication When Socket transport is used, the buffer size must fit within the OS socket buffer limits (rmem_max, wmem_max) Setting NCCL_BUFFSIZE to a value larger than the OS socket buffer limits causes the socket write to block indefinitely IB transport has its own memory registration and doesn't have this socket buffer limit","anchorText":"NCCL BUFFSIZE Oversized — Socket Transport Stall NCCL hangs during all-reduce or all-gather after NCCL_BUFFSIZE was manually increased NCCL_DEBUG=INFO shows Socket transport timeouts or retransmissions Watchdog timeout after increasing NCCL_BUFFSIZE Training works with default NCCL_BUFFSIZE but hangs with custom value nccl buffsize socket buffer configuration tuning NCCL_BUFFSIZE controls the per-channel buffer size for NCCL communication When Socket transport is used, the buffer size must fit within the OS socket buffer limits (rmem_max, wmem_max) Setting NCCL_BUFFSIZE to a value larger than the OS socket buffer limits causes the socket write to block indefinitely IB transport has its own memory registration and doesn't have this socket buffer limit ","action":"Remove the custom NCCL_BUFFSIZE setting and let NCCL use its default value","steps":["Remove the custom NCCL_BUFFSIZE setting and let NCCL use its default value","If custom buffer size is needed: increase OS socket buffer limits with 'sudo sysctl -w net.core.rmem_max=2147483647' and 'sudo sysctl -w net.core.wmem_max=2147483647'","For Socket transport: keep NCCL_BUFFSIZE at or below 4MB (4194304)","Switch to IB transport if available: NCCL_NET=IB handles large buffer sizes correctly","Use Denpex to validate NCCL configuration and flag oversized buffer settings"]},{"slug":"nccl-topology-regression","title":"NCCL Topology Detection Regression (v2.18.3+)","category":"Communication","text":"NCCL Topology Detection Regression (v2.18.3+) Communication NCCL topology XML shows different GPU-NIC assignments after upgrading NCCL Performance regression of 10-50% on multi-node training after NCCL upgrade NCCL_DEBUG=INFO shows different channel counts or algorithm selections than before Downgrading NCCL to v2.18.1 restores performance nccl topology regression version upgrade performance Training performance degrades after upgrading NCCL to v2.18.3 or later NCCL uses a different (worse) topology than the previous version GPU-NIC affinity is incorrect in the NCCL topology XML after the upgrade NCCL v2.18.3 changed the topology detection algorithm to handle more complex NIC configurations, but the new algorithm produces incorrect GPU-NIC affinity on some systems The regression causes NCCL to assign GPUs to NICs across NUMA boundaries instead of within the same NUMA node On systems with NVSwitch, the regression may incorrectly prefer P2P transport over the faster NVLS transport The fix was included in NCCL v2.22.x but not backported to v2.18.x or v2.19.x","anchorText":"NCCL Topology Detection Regression (v2.18.3+) NCCL topology XML shows different GPU-NIC assignments after upgrading NCCL Performance regression of 10-50% on multi-node training after NCCL upgrade NCCL_DEBUG=INFO shows different channel counts or algorithm selections than before Downgrading NCCL to v2.18.1 restores performance nccl topology regression version upgrade performance NCCL v2.18.3 changed the topology detection algorithm to handle more complex NIC configurations, but the new algorithm produces incorrect GPU-NIC affinity on some systems The regression causes NCCL to assign GPUs to NICs across NUMA boundaries instead of within the same NUMA node On systems with NVSwitch, the regression may incorrectly prefer P2P transport over the faster NVLS transport The fix was included in NCCL v2.22.x but not backported to v2.18.x or v2.19.x ","action":"Upgrade NCCL to v2.22.0 or later where the topology regression is fixed","steps":["Upgrade NCCL to v2.22.0 or later where the topology regression is fixed","If upgrade is not possible: set NCCL_TOPO_FILE to a pre-generated topology XML from the working NCCL version","Override the GPU-NIC affinity explicitly with NCCL_IB_HCA to match the correct NUMA assignment","As a diagnostic: compare NCCL topology XML output between the old and new NCCL versions to confirm the regression","Use Denpex to detect NCCL version-specific regressions and recommend the correct upgrade path"]},{"slug":"deepspeed-bf16-norm-underflow","title":"DeepSpeed bf16 Gradient Norm Underflow","category":"Training Stability","text":"DeepSpeed bf16 Gradient Norm Underflow Training Stability AssertionError: all_groups_norm > 0 in DeepSpeed gradient norm computation Training runs fine with fp16 but crashes with bf16 The error occurs at the same step consistently Gradient norms reported as 0.0 in DeepSpeed logs even though the model is learning deepspeed bf16 gradient-norm underflow precision numerical-stability Training crashes with 'assert all_groups_norm > 0' when using bf16 precision with DeepSpeed The error occurs even though gradients are non-zero (they're just too small for bf16 to represent) Switching to fp16 resolves the issue but bf16 is preferred for training stability bf16 has fewer mantissa bits (7) than fp16 (10), reducing the smallest representable positive number When gradient norms are very small (common with large models, small LR, or gradient accumulation), they can underflow to zero in bf16 DeepSpeed's gradient norm assertion assumes non-zero norms, which is not guaranteed with bf16 precision The gradient norm is computed in bf16 precision instead of being upcast to fp32 for the norm calculation","anchorText":"DeepSpeed bf16 Gradient Norm Underflow AssertionError: all_groups_norm > 0 in DeepSpeed gradient norm computation Training runs fine with fp16 but crashes with bf16 The error occurs at the same step consistently Gradient norms reported as 0.0 in DeepSpeed logs even though the model is learning deepspeed bf16 gradient-norm underflow precision numerical-stability bf16 has fewer mantissa bits (7) than fp16 (10), reducing the smallest representable positive number When gradient norms are very small (common with large models, small LR, or gradient accumulation), they can underflow to zero in bf16 DeepSpeed's gradient norm assertion assumes non-zero norms, which is not guaranteed with bf16 precision The gradient norm is computed in bf16 precision instead of being upcast to fp32 for the norm calculation ","action":"Upgrade DeepSpeed to v0.13.5 or later which fixes the bf16 norm underflow","steps":["Upgrade DeepSpeed to v0.13.5 or later which fixes the bf16 norm underflow","If upgrade is not possible: use fp16 instead of bf16 (with appropriate loss scaling)","Set gradient_accumulation_steps to a smaller value to produce larger per-step gradients","Increase the learning rate slightly so gradients don't underflow","Use Denpex to detect bf16-specific numerical issues and recommend the correct precision setting"]},{"slug":"deepspeed-moe-leaf-module","title":"DeepSpeed MoE + ZeRO-3 Hang — Missing Leaf Module Marking","category":"Distributed Training","text":"DeepSpeed MoE + ZeRO-3 Hang — Missing Leaf Module Marking Distributed Training Training hangs with NCCL timeout during the first forward pass through MoE layers nvidia-smi shows 0% GPU utilization on all ranks during the hang DeepSpeed logs show parameter gathering initiated but never completing for MoE parameters No error output — the job just hangs indefinitely deepspeed moe zero3 leaf-module hang mixture-of-experts parameter-gathering Training with DeepSpeed ZeRO-3 and MoE models hangs at the first forward pass NCCL timeout occurs during parameter gathering for MoE layers The hang is specific to MoE models — non-MoE models with the same ZeRO-3 config work fine ZeRO-3 parameter gathering traverses the model hierarchy recursively, entering MoE router logic during the gather phase MoE routers have conditional parameter access (only active experts per token), which creates deadlocks when ZeRO-3 tries to gather all expert parameters simultaneously Without leaf module marking, ZeRO-3 doesn't know to treat MoE blocks as atomic units that should not be recursively decomposed The hang occurs because rank 0 is waiting for rank 1's expert parameters while rank 1 is stuck in the MoE router","anchorText":"DeepSpeed MoE + ZeRO-3 Hang — Missing Leaf Module Marking Training hangs with NCCL timeout during the first forward pass through MoE layers nvidia-smi shows 0% GPU utilization on all ranks during the hang DeepSpeed logs show parameter gathering initiated but never completing for MoE parameters No error output — the job just hangs indefinitely deepspeed moe zero3 leaf-module hang mixture-of-experts parameter-gathering ZeRO-3 parameter gathering traverses the model hierarchy recursively, entering MoE router logic during the gather phase MoE routers have conditional parameter access (only active experts per token), which creates deadlocks when ZeRO-3 tries to gather all expert parameters simultaneously Without leaf module marking, ZeRO-3 doesn't know to treat MoE blocks as atomic units that should not be recursively decomposed The hang occurs because rank 0 is waiting for rank 1's expert parameters while rank 1 is stuck in the MoE router ","action":"Mark MoE blocks as leaf modules: deepspeed.zero.set_z3_leaf_modules(model, [MoEBlockClass])","steps":["Mark MoE blocks as leaf modules: deepspeed.zero.set_z3_leaf_modules(model, [MoEBlockClass])","Upgrade DeepSpeed to v0.14.0+ which has better MoE+ZeRO-3 integration","If using Hugging Face: wrap MoE layers with deepspeed.zero.HfDeepSpeedConfig for automatic leaf module detection","As a diagnostic: try ZeRO stage 2 (which doesn't require parameter gathering) to confirm the hang is ZeRO-3 specific","Use Denpex to detect MoE+ZeRO-3 hang patterns and recommend the leaf module configuration"]},{"slug":"deepspeed-overlap-contig-nan","title":"DeepSpeed NaN from overlap_comm + contiguous_gradients","category":"Training Stability","text":"DeepSpeed NaN from overlap_comm + contiguous_gradients Training Stability grad_norm is nan reported by DeepSpeed Training loss becomes NaN after a variable number of steps The NaN disappears when setting overlap_comm: false or contiguous_gradients: false DeepSpeed logs show NaN in gradient norm calculation deepspeed nan overlap-comm contiguous-gradients race-condition zero3 gradient Training produces NaN gradient norms when both overlap_comm and contiguous_gradients are True in DeepSpeed ZeRO-3 Disabling either flag individually resolves the NaN The NaN appears non-deterministically, making it hard to reproduce When overlap_comm is True, DeepSpeed starts communication (all-reduce) for one gradient bucket while the next bucket is being computed When contiguous_gradients is True, DeepSpeed copies gradients into a contiguous buffer for efficient communication The race: the communication overlap reads from the contiguous buffer while it's being rewritten by the next bucket's gradient copy, producing corrupted (NaN) values This is a buffer reuse race condition in DeepSpeed's gradient communication pipeline","anchorText":"DeepSpeed NaN from overlap_comm + contiguous_gradients grad_norm is nan reported by DeepSpeed Training loss becomes NaN after a variable number of steps The NaN disappears when setting overlap_comm: false or contiguous_gradients: false DeepSpeed logs show NaN in gradient norm calculation deepspeed nan overlap-comm contiguous-gradients race-condition zero3 gradient When overlap_comm is True, DeepSpeed starts communication (all-reduce) for one gradient bucket while the next bucket is being computed When contiguous_gradients is True, DeepSpeed copies gradients into a contiguous buffer for efficient communication The race: the communication overlap reads from the contiguous buffer while it's being rewritten by the next bucket's gradient copy, producing corrupted (NaN) values This is a buffer reuse race condition in DeepSpeed's gradient communication pipeline ","action":"Disable one of the two flags: set overlap_comm: false or contiguous_gradients: false in the DeepSpeed config","steps":["Disable one of the two flags: set overlap_comm: false or contiguous_gradients: false in the DeepSpeed config","If communication overlap is needed for performance: disable contiguous_gradients and accept the slightly higher communication overhead","Upgrade DeepSpeed to v0.14.0+ which includes fixes for the overlap+contiguous race condition","As a diagnostic: run with NCCL_DEBUG=INFO to confirm the NaN correlates with communication overlap timing","Use Denpex to detect the specific flag combination causing NaN and recommend the correct configuration"]},{"slug":"deepspeed-fusedadam-h100-illegal-mem","title":"DeepSpeed FusedAdam Illegal Memory Access on H100","category":"Memory","text":"DeepSpeed FusedAdam Illegal Memory Access on H100 Memory CUDA error: illegal memory access in FusedAdam kernel Training crashes during the optimizer step (not forward or backward) nvidia-smi shows the GPU in a bad state after the crash The error is deterministic: it happens at the same step every time deepspeed fusedadam h100 sm90 illegal-memory optimizer cuda Training on H100 with DeepSpeed FusedAdam crashes with 'CUDA error: illegal memory access' The same training works on A100 with FusedAdam Switching to the standard PyTorch AdamW optimizer resolves the crash FusedAdam's CUDA kernels were compiled for SM80 (A100) and are not compatible with SM90 (H100) architecture H100's SM90 architecture has different shared memory layout and warp scheduling that causes out-of-bounds access in FusedAdam kernels The illegal memory access occurs in the Adam update kernel when updating optimizer states (momentum, variance) for large parameter tensors DeepSpeed's FusedAdam has not been updated to support SM90 architecture natively","anchorText":"DeepSpeed FusedAdam Illegal Memory Access on H100 CUDA error: illegal memory access in FusedAdam kernel Training crashes during the optimizer step (not forward or backward) nvidia-smi shows the GPU in a bad state after the crash The error is deterministic: it happens at the same step every time deepspeed fusedadam h100 sm90 illegal-memory optimizer cuda FusedAdam's CUDA kernels were compiled for SM80 (A100) and are not compatible with SM90 (H100) architecture H100's SM90 architecture has different shared memory layout and warp scheduling that causes out-of-bounds access in FusedAdam kernels The illegal memory access occurs in the Adam update kernel when updating optimizer states (momentum, variance) for large parameter tensors DeepSpeed's FusedAdam has not been updated to support SM90 architecture natively ","action":"Switch from FusedAdam to PyTorch's native AdamW: optimizer = torch.optim.AdamW(model.parameters(), lr=1e-4)","steps":["Switch from FusedAdam to PyTorch's native AdamW: optimizer = torch.optim.AdamW(model.parameters(), lr=1e-4)","If FusedAdam performance is needed: use the Triton-based FusedAdam from apex or the torch.compile-based optimizer","Update DeepSpeed to the latest version which may include SM90 FusedAdam fixes","Compile FusedAdam from source with TORCH_CUDA_ARCH_LIST='9.0' to target SM90","Use Denpex to detect H100+FusedAdam incompatibility and recommend the correct optimizer"]},{"slug":"deepspeed-checkpoint-engine-hang","title":"DeepSpeed DecoupledCheckpointEngine Infinite Hang","category":"Data Integrity","text":"DeepSpeed DecoupledCheckpointEngine Infinite Hang Data Integrity Training stops making progress at a checkpoint save step nvidia-smi shows 0% GPU utilization on all ranks DeepSpeed logs show 'Saving checkpoint...' but no completion message ps aux shows the checkpoint writer thread stuck in D (uninterruptible sleep) state No timeout or error is triggered — the job just hangs forever deepspeed checkpoint hang deadlock decoupled async nvme Training hangs indefinitely during checkpoint save with no error output The checkpoint save operation never completes, blocking all training progress GPU utilization drops to 0% during the hang while the process remains alive The async checkpoint writer thread deadlocks when the checkpoint queue is full and the main thread tries to enqueue another checkpoint NVMe offload checkpoint writes to slow storage (NFS) can block the writer thread indefinitely A race condition in the DecoupledCheckpointEngine causes the writer thread to wait for a signal that is never sent Memory pressure causes the checkpoint buffer allocation to block, which in turn blocks the writer thread","anchorText":"DeepSpeed DecoupledCheckpointEngine Infinite Hang Training stops making progress at a checkpoint save step nvidia-smi shows 0% GPU utilization on all ranks DeepSpeed logs show 'Saving checkpoint...' but no completion message ps aux shows the checkpoint writer thread stuck in D (uninterruptible sleep) state No timeout or error is triggered — the job just hangs forever deepspeed checkpoint hang deadlock decoupled async nvme The async checkpoint writer thread deadlocks when the checkpoint queue is full and the main thread tries to enqueue another checkpoint NVMe offload checkpoint writes to slow storage (NFS) can block the writer thread indefinitely A race condition in the DecoupledCheckpointEngine causes the writer thread to wait for a signal that is never sent Memory pressure causes the checkpoint buffer allocation to block, which in turn blocks the writer thread ","action":"Switch from DecoupledCheckpointEngine to the synchronous checkpoint engine","steps":["Switch from DecoupledCheckpointEngine to the synchronous checkpoint engine","Reduce checkpoint save frequency to avoid filling the async queue","Ensure checkpoint storage (NFS, local SSD) has adequate I/O bandwidth and free space","Set a checkpoint save timeout in DeepSpeed config to convert the hang into a timed error","Use Denpex to monitor checkpoint save duration and alert when saves exceed expected time"]},{"slug":"deepspeed-fd-leak-enospc","title":"DeepSpeed FastFileWriter File Descriptor Leak — Phantom ENOSPC","category":"Data Integrity","text":"DeepSpeed FastFileWriter File Descriptor Leak — Phantom ENOSPC Data Integrity OSError: [Errno 28] No space left on device when saving checkpoints df -h shows ample free space on the checkpoint storage lsof -p <pid> | wc -l shows file descriptor count near the ulimit -n limit The error occurs after a specific number of checkpoint saves (consistent across runs) deepspeed fd-leak enospc file-descriptor checkpoint fastfilewriter Training fails with 'No space left on device' but df shows plenty of free space The error occurs after many checkpoint saves, not at the beginning lsof shows thousands of open file descriptors for checkpoint files that should have been closed FastFileWriter opens checkpoint files for writing but doesn't close them after the write completes Each checkpoint save opens new file descriptors without closing the previous ones After enough saves, the per-process file descriptor limit (ulimit -n, typically 1024 or 65536) is exceeded The OS returns ENOSPC because the filesystem can't allocate new inodes or directory entries when fd limits are hit","anchorText":"DeepSpeed FastFileWriter File Descriptor Leak — Phantom ENOSPC OSError: [Errno 28] No space left on device when saving checkpoints df -h shows ample free space on the checkpoint storage lsof -p <pid> | wc -l shows file descriptor count near the ulimit -n limit The error occurs after a specific number of checkpoint saves (consistent across runs) deepspeed fd-leak enospc file-descriptor checkpoint fastfilewriter FastFileWriter opens checkpoint files for writing but doesn't close them after the write completes Each checkpoint save opens new file descriptors without closing the previous ones After enough saves, the per-process file descriptor limit (ulimit -n, typically 1024 or 65536) is exceeded The OS returns ENOSPC because the filesystem can't allocate new inodes or directory entries when fd limits are hit ","action":"Upgrade DeepSpeed to v0.14.3+ which includes the FastFileWriter fd leak fix","steps":["Upgrade DeepSpeed to v0.14.3+ which includes the FastFileWriter fd leak fix","Increase the per-process fd limit: ulimit -n 1048576","Reduce checkpoint save frequency to slow the fd leak","As a workaround: periodically restart the training process from the latest checkpoint to reset fd count","Use Denpex to monitor fd count trends during training and alert on approaching limits"]},{"slug":"deepspeed-universal-checkpoint-sort","title":"DeepSpeed Universal Checkpoint Lexicographic Sort Corruption","category":"Data Integrity","text":"DeepSpeed Universal Checkpoint Lexicographic Sort Corruption Data Integrity Training loss diverges after resuming from a universal checkpoint with 10+ ranks Model accuracy drops significantly after checkpoint resume Parameter checksums differ across ranks after loading the same universal checkpoint The issue only appears with 10+ ranks (rank_10 sorts before rank_2 lexicographically) deepspeed checkpoint sort corruption universal-checkpoint lexicographic silent Model weights are silently corrupted after loading a universal checkpoint Training produces wrong results or diverges after resuming from a universal checkpoint The corruption is invisible — no error is raised during checkpoint loading DeepSpeed sorts rank directories using lexicographic order: rank_0, rank_1, rank_10, rank_11, ..., rank_2, rank_3, ... Numeric order should be: rank_0, rank_1, rank_2, rank_3, ..., rank_10, rank_11, ... When rank_10 is loaded in rank_2's position, all parameters from rank_10 onwards are loaded into the wrong rank This silently corrupts model weights because each rank's optimizer states (momentum, variance) are mismatched with its parameters","anchorText":"DeepSpeed Universal Checkpoint Lexicographic Sort Corruption Training loss diverges after resuming from a universal checkpoint with 10+ ranks Model accuracy drops significantly after checkpoint resume Parameter checksums differ across ranks after loading the same universal checkpoint The issue only appears with 10+ ranks (rank_10 sorts before rank_2 lexicographically) deepspeed checkpoint sort corruption universal-checkpoint lexicographic silent DeepSpeed sorts rank directories using lexicographic order: rank_0, rank_1, rank_10, rank_11, ..., rank_2, rank_3, ... Numeric order should be: rank_0, rank_1, rank_2, rank_3, ..., rank_10, rank_11, ... When rank_10 is loaded in rank_2's position, all parameters from rank_10 onwards are loaded into the wrong rank This silently corrupts model weights because each rank's optimizer states (momentum, variance) are mismatched with its parameters ","action":"Upgrade DeepSpeed to v0.14.0+ which fixes the lexicographic sort to numeric sort","steps":["Upgrade DeepSpeed to v0.14.0+ which fixes the lexicographic sort to numeric sort","If upgrade is not possible: rename rank directories with zero-padded names (rank_00, rank_01, ..., rank_09, rank_10) before loading","Validate checkpoint integrity after loading: compare parameter checksums across ranks","As a diagnostic: save a small test checkpoint with 10+ ranks and verify that rank_2 loads rank_2's data","Use Denpex to validate universal checkpoint integrity and detect sort-related corruption"]},{"slug":"deepspeed-pytorch25-parameters-dict","title":"DeepSpeed ZeRO-3 + PyTorch 2.5 _parameters Dict Error","category":"Distributed Training","text":"DeepSpeed ZeRO-3 + PyTorch 2.5 _parameters Dict Error Distributed Training AttributeError: 'dict' object has no attribute '_in_forward' Error occurs in deepspeed/runtime/zero/partition_parameters.py DeepSpeed ZeRO-3 initialization fails but ZeRO-2 works Downgrading to PyTorch 2.4 resolves the issue deepspeed pytorch-2.5 parameters version compatibility zero3 Training crashes with 'dict object has no attribute _in_forward' after upgrading to PyTorch 2.5 DeepSpeed ZeRO-3 init fails on models that worked with PyTorch 2.4 The error occurs during DeepSpeed's parameter partitioning, not in the model code PyTorch 2.5 changed the internal representation of module._parameters from a plain dict to a custom class that tracks parameter access patterns DeepSpeed ZeRO-3's partition_parameters.py accesses _parameters as a plain dict, expecting dict methods only The new _parameters class in PyTorch 2.5 has additional attributes (like _in_forward) that DeepSpeed doesn't expect When DeepSpeed tries to access _in_forward on what it thinks is a plain dict, the AttributeError is raised","anchorText":"DeepSpeed ZeRO-3 + PyTorch 2.5 _parameters Dict Error AttributeError: 'dict' object has no attribute '_in_forward' Error occurs in deepspeed/runtime/zero/partition_parameters.py DeepSpeed ZeRO-3 initialization fails but ZeRO-2 works Downgrading to PyTorch 2.4 resolves the issue deepspeed pytorch-2.5 parameters version compatibility zero3 PyTorch 2.5 changed the internal representation of module._parameters from a plain dict to a custom class that tracks parameter access patterns DeepSpeed ZeRO-3's partition_parameters.py accesses _parameters as a plain dict, expecting dict methods only The new _parameters class in PyTorch 2.5 has additional attributes (like _in_forward) that DeepSpeed doesn't expect When DeepSpeed tries to access _in_forward on what it thinks is a plain dict, the AttributeError is raised ","action":"Upgrade DeepSpeed to v0.15.0+ which is compatible with PyTorch 2.5's _parameters changes","steps":["Upgrade DeepSpeed to v0.15.0+ which is compatible with PyTorch 2.5's _parameters changes","If DeepSpeed upgrade is not possible: downgrade PyTorch to 2.4.x","As a workaround: apply the DeepSpeed patch that handles the new _parameters class","Use Denpex to detect DeepSpeed+PyTorch version incompatibilities before training starts"]},{"slug":"deepspeed-sigbus-exit-code-7","title":"DeepSpeed SIGBUS Exit Code -7 After Checkpoint Load","category":"Data Integrity","text":"DeepSpeed SIGBUS Exit Code -7 After Checkpoint Load Data Integrity Process exits with code -7 (SIGBUS) or signal 7 No Python traceback or error message — just an abrupt exit dmesg shows SIGBUS or Bus error on the training process The crash occurs immediately after torch.load() or DeepSpeed checkpoint load NFS mount points show as stale or unmounted in df output deepspeed sigbus exit-code-7 checkpoint nfs mmap filesystem Training crashes with exit code -7 (SIGBUS) immediately after loading a checkpoint No Python traceback is produced — the process dies at the OS level All ranks crash simultaneously, suggesting a shared checkpoint file issue A memory-mapped checkpoint file was truncated or deleted while the process still had it mapped, causing SIGBUS when the process tries to read the missing pages NFS mount became stale or was unmounted between checkpoint save and load, making the mapped file inaccessible DeepSpeed's NVMe offload uses mmap for checkpoint buffers, and the underlying file was deleted or moved Disk corruption or filesystem journaling errors made the checkpoint file partially inaccessible","anchorText":"DeepSpeed SIGBUS Exit Code -7 After Checkpoint Load Process exits with code -7 (SIGBUS) or signal 7 No Python traceback or error message — just an abrupt exit dmesg shows SIGBUS or Bus error on the training process The crash occurs immediately after torch.load() or DeepSpeed checkpoint load NFS mount points show as stale or unmounted in df output deepspeed sigbus exit-code-7 checkpoint nfs mmap filesystem A memory-mapped checkpoint file was truncated or deleted while the process still had it mapped, causing SIGBUS when the process tries to read the missing pages NFS mount became stale or was unmounted between checkpoint save and load, making the mapped file inaccessible DeepSpeed's NVMe offload uses mmap for checkpoint buffers, and the underlying file was deleted or moved Disk corruption or filesystem journaling errors made the checkpoint file partially inaccessible ","action":"Check if the checkpoint file still exists and has the correct size: ls -la <checkpoint_path>","steps":["Check if the checkpoint file still exists and has the correct size: ls -la <checkpoint_path>","Verify NFS mount health: df -h and ls on the checkpoint directory","If the file was deleted: restore from backup or resume from a previous checkpoint","If NFS is stale: remount the filesystem (sudo mount -o remount <mount_point>)","Use Denpex to validate checkpoint file integrity before loading and detect filesystem issues"]},{"slug":"deepspeed-zero3-small-param-partition-bug","title":"DeepSpeed ZeRO-3 Small Parameter Partition Bug","category":"Distributed Training","text":"DeepSpeed ZeRO-3 Small Parameter Partition Bug Distributed Training UnboundLocalError: local variable 'partition_dim' referenced before assignment in partition_parameters.py Error occurs in deepspeed/runtime/zero/partition_parameters.py around line 862 Crash happens during DeepSpeed engine initialization, before training starts Reducing the number of GPUs or switching to ZeRO-2 avoids the crash deepspeed zero3 partition small-parameter unboundlocalerror partition-parameters Training crashes with UnboundLocalError in DeepSpeed's partition_parameters.py The error occurs during ZeRO-3 initialization for models with very small parameters (e.g., bias vectors, layer norm parameters) The same model works with fewer GPUs but crashes when scaled to more GPUs ZeRO-3 partitions each parameter's elements equally across GPUs. When a parameter has fewer elements than the number of GPUs, the partition calculation produces zero or negative slice sizes The partition_dim variable is never assigned when the parameter is too small to partition, causing the UnboundLocalError Small parameters (e.g., a bias vector with 8 elements on 16 GPUs) cannot be meaningfully partitioned The bug is in DeepSpeed's partition_parameters.py which doesn't handle the edge case of parameters smaller than world_size","anchorText":"DeepSpeed ZeRO-3 Small Parameter Partition Bug UnboundLocalError: local variable 'partition_dim' referenced before assignment in partition_parameters.py Error occurs in deepspeed/runtime/zero/partition_parameters.py around line 862 Crash happens during DeepSpeed engine initialization, before training starts Reducing the number of GPUs or switching to ZeRO-2 avoids the crash deepspeed zero3 partition small-parameter unboundlocalerror partition-parameters ZeRO-3 partitions each parameter's elements equally across GPUs. When a parameter has fewer elements than the number of GPUs, the partition calculation produces zero or negative slice sizes The partition_dim variable is never assigned when the parameter is too small to partition, causing the UnboundLocalError Small parameters (e.g., a bias vector with 8 elements on 16 GPUs) cannot be meaningfully partitioned The bug is in DeepSpeed's partition_parameters.py which doesn't handle the edge case of parameters smaller than world_size ","action":"Upgrade DeepSpeed to v0.12.0+ which handles small parameter partitioning correctly","steps":["Upgrade DeepSpeed to v0.12.0+ which handles small parameter partitioning correctly","Set contiguous_gradients: false in the DeepSpeed config to avoid partitioning small parameters","Reduce the number of GPUs to be less than or equal to the smallest parameter's element count","Use deepspeed.zero.SetZ3LeafModules() to mark modules with small parameters as leaf modules that bypass partitioning","Use Denpex to detect small parameter partition issues and recommend the correct configuration"]},{"slug":"cuda-device-side-assert","title":"CUDA Device-Side Assert Triggered","category":"Training Stability","text":"CUDA Device-Side Assert Triggered Training Stability RuntimeError: CUDA error: device-side assert triggered Assertion `srcIndex < srcSelectDimSize` failed nll_loss / index_select / embedding op in the trace Error disappears on CPU but the loss is wrong (label out of range) device-side-assert cuda index out of bounds embedding cross-entropy training-stability srcindex Training crashes with 'CUDA error: device-side assert triggered' The reported line is unrelated to the real fault Subsequent CUDA calls all fail until the process restarts A label or index is outside the valid range for an embedding/loss layer (e.g. label == num_classes, or pad id >= vocab_size) CUDA executes asynchronously, so the trace points at the next synchronizing call, not the faulting kernel Target tensor contains -1 or a sentinel id the loss does not ignore Vocabulary/num_classes drifted from the data after a config change","anchorText":"CUDA Device-Side Assert Triggered RuntimeError: CUDA error: device-side assert triggered Assertion `srcIndex < srcSelectDimSize` failed nll_loss / index_select / embedding op in the trace Error disappears on CPU but the loss is wrong (label out of range) device-side-assert cuda index out of bounds embedding cross-entropy training-stability srcindex A label or index is outside the valid range for an embedding/loss layer (e.g. label == num_classes, or pad id >= vocab_size) CUDA executes asynchronously, so the trace points at the next synchronizing call, not the faulting kernel Target tensor contains -1 or a sentinel id the loss does not ignore Vocabulary/num_classes drifted from the data after a config change # Make the async assert point at the real kernel, then bound-check inputs\n# CUDA_LAUNCH_BLOCKING=1 python train.py\n\nassert targets.min() >= 0 and targets.max() < num_classes, \\\n f\"label out of range: [{targets.min()}, {targets.max()}] vs num_classes={num_classes}\"\nassert input_ids.max() < model.get_input_embeddings().num_embeddings, \\\n \"token id exceeds vocab size — tokenizer/model mismatch\"\nloss = F.cross_entropy(logits, targets, ignore_index=pad_id)","action":"Re-run with CUDA_LAUNCH_BLOCKING=1 to make the trace point at the real kernel","steps":["Re-run with CUDA_LAUNCH_BLOCKING=1 to make the trace point at the real kernel","Assert label ranges before the loss: assert targets.max() < num_classes and targets.min() >= 0","Check embedding inputs: assert input_ids.max() < embedding.num_embeddings","Set ignore_index correctly in CrossEntropyLoss for pad tokens","Reproduce the failing batch on CPU to get a precise Python-level error"]},{"slug":"straggler-slow-rank-detection","title":"Straggler / Slow-Rank Detection","category":"Reliability","text":"Straggler / Slow-Rank Detection Reliability One rank's step time is consistently 1.5-10x the median GPU utilization on most ranks oscillates between 100% and 0% (waiting) MFU / tokens-per-second degrades over hours without a config change No NCCL timeout fires because the slow rank still responds, just late straggler slow-rank reliability distributed throughput gray-failure stall Throughput drops with no error in the logs All ranks block at all-reduce waiting on one slow rank Step time is dominated by the slowest rank A degraded GPU (thermal throttle, ECC row-remap, reduced clocks) on one rank Slower network path (oversubscribed link, bad cable, PFC pause storms) to one node CPU/dataloader contention starving one rank's input pipeline NUMA / PCIe misconfiguration on a single host","anchorText":"Straggler / Slow-Rank Detection One rank's step time is consistently 1.5-10x the median GPU utilization on most ranks oscillates between 100% and 0% (waiting) MFU / tokens-per-second degrades over hours without a config change No NCCL timeout fires because the slow rank still responds, just late straggler slow-rank reliability distributed throughput gray-failure stall A degraded GPU (thermal throttle, ECC row-remap, reduced clocks) on one rank Slower network path (oversubscribed link, bad cable, PFC pause storms) to one node CPU/dataloader contention starving one rank's input pipeline NUMA / PCIe misconfiguration on a single host # Lightweight per-rank straggler check — log ranks slower than 1.5x median\nimport torch, torch.distributed as dist\n\nstep_t = torch.tensor([elapsed], device='cuda')\ngathered = [torch.zeros_like(step_t) for _ in range(dist.get_world_size())]\ndist.all_gather(gathered, step_t)\ntimes = torch.stack(gathered).flatten()\nmedian = times.median()\nslow = (times > 1.5 * median).nonzero().flatten().tolist()\nif dist.get_rank() == 0 and slow:\n print(f'[straggler] ranks {slow} slower than 1.5x median ({median.item():.3f}s)')","action":"Instrument per-rank step time and flag any rank > 1.5x the median (this is what Denpex straggler detection does)","steps":["Instrument per-rank step time and flag any rank > 1.5x the median (this is what Denpex straggler detection does)","Check the suspect GPU: nvidia-smi -q -d CLOCK,TEMPERATURE,ECC for throttling or remaps","Compare per-rank NCCL all-reduce latency to isolate network vs compute","Drain the slow node (scontrol update nodename=<n> state=drain) and resume from checkpoint","Adopt elastic / redundant training so a straggler can be evicted without a full restart"]}]}
|