hipengine 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hipengine-0.2.0 → hipengine-0.2.1}/CHANGELOG.md +41 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/PKG-INFO +12 -5
- {hipengine-0.2.0 → hipengine-0.2.1}/README.md +11 -4
- {hipengine-0.2.0 → hipengine-0.2.1}/WORKLOG.md +79 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/API.md +14 -4
- hipengine-0.2.1/hipengine/generation/qwen35_paro.py +298 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/llm.py +45 -22
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/loading/__init__.py +2 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/loading/gguf.py +2 -1
- hipengine-0.2.1/hipengine/loading/hf_cache.py +121 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/loading/safetensors.py +3 -26
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/server/__main__.py +34 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/server/api.py +241 -23
- {hipengine-0.2.0 → hipengine-0.2.1}/pyproject.toml +1 -1
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/llamacpp_bench_with_peak.py +2 -3
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_generation_qwen35_paro.py +40 -2
- hipengine-0.2.1/tests/test_hf_cache.py +63 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_llm_generate.py +88 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_server_api.py +76 -2
- hipengine-0.2.0/hipengine/generation/qwen35_paro.py +0 -155
- {hipengine-0.2.0 → hipengine-0.2.1}/.gitattributes +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/.github/workflows/publish.yml +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/.gitignore +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/AGENTS.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/CLAUDE.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/LICENSE +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/7900XTX.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/CHANGELOG.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/MTP.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/README.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/W7900.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/configs/llamacpp-mtp-qwen36-27b.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/prompts/mtpbench-code-general-ja.jsonl +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/.gitkeep +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-13-hipengine-qwen35-paro-optimal-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-13-source-lineage-qwen35-paro-optimal-4k-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-13-source-lineage-qwen35-paro-optimal-512-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-14-hipengine-qwen35-c1-parent-fixture-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-14-hipengine-qwen35-cn-correctness-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-14-hipengine-qwen35-paro-512-128-ab-fused-lmhead128-graph-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-14-hipengine-qwen35-paro-512-128-c1-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-14-hipengine-qwen35-paro-512-128-graph-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-14-hipengine-qwen35-paro-512-128-linear-qkv-z-full-qk-fused-graph-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-14-hipengine-qwen35-paro-512-128-linear-qkv-z-fused-graph-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-14-hipengine-qwen35-paro-512-128-lmhead128-qk-qkvz-fused-graph-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-14-hipengine-qwen35-paro-512-128-tokenizer-cache-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-0p8b-paro-512-128-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c1-parent-fixture-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c1-parent-mixed-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c1-router-qnorm-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c1-scheduler-serial-bench-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c2-native-compact-prefill-correctness-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c2-scheduler-serial-bench-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c2-scheduler-serial-runner-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c2-serial-slot-runner-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c4-native-compact-prefill-correctness-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c4-scheduler-serial-bench-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c4-scheduler-serial-runner-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c8-native-compact-prefill-correctness-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c8-scheduler-serial-bench-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-c8-scheduler-serial-runner-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-cn-generated-equality-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-dflash-ddtree-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-linear-attn-segment-prefill-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefill-compact-c8-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefill-full-attn-boundary-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefill-full-single-request-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefill-multiloop-512-4k-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefill-plan-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-layer0-attention-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-layer0-attn-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-layer0-decode-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-layer0-gated-recurrent-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-layer0-stage-bisect-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-layer3-fullattn-stage-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-prefill-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-scratch-restore-sweep.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-serial-fullattn-layer4-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-serial-suffix-full40-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-native-prefix-sweep-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-prefix-bisect-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-15-hipengine-qwen35-varlen-full-attn-prefill-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-gguf-q4k-pack8-bf16out-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-gguf-q4k-pack8-gemv-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-gguf-qwen35-e2e-correctness-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-gguf-vs-paro-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-qwen35-aotriton-cast-glue-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-qwen35-aotriton-gate-rotate-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-qwen35-aotriton-threshold-sweep-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-qwen35-aotriton-v3-memory-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-qwen35-aotriton-v3-prefill-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-qwen35-comparison-tables-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-qwen35-decode-graph-replay-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-qwen35-long-checkpoint-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-16-hipengine-qwen35-prefill-chunking-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gfx1151-shisa-qwen36-packed-canonical-sweep-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gfx1151-shisa-qwen36-packed-chunk256-sweep-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gguf-aotriton-v3-prefill-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gguf-bulk-prefill-q4km-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gguf-decode-graph-replay-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gguf-full-attn-gpu-prelude-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gguf-local-quant-coverage-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gguf-prefill-projection-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gguf-q4km-parity-benchmark-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-gguf-resident-session-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-08b-gfx1151-dense-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-35b-qwen36-27b-gfx1151-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d11-rotate-dual-pack8-fusion-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d12-rmsnorm-producer-fusion-deferred.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d13-same-input-projection-fusions-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d14-selected-moe-postop-fold-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d15-router-coop-fold-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d16-kv-pack8-fusion-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d21-marlin-k-qweight-neutral-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d31-d33-grouped-gqa-long-context-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d42-dispatch-cap-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d44-launch-bounds-deferred.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d51-gdn-decode-audit.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-d52-w8a16-decode-audit.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-p33-moe-metadata-fanout-deferred.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-p52-prefill-chunk-autotune-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-qwen36-p11-rocblas-ab-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-qwen36-p12-shared-gate-up-token-tile-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-qwen36-p13-shared-down-combine-token-tile-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-qwen36-p14-moe-wmma-threshold-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-qwen36-p16-prefill-mcumode-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-qwen36-p31-gdn-rotate-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-qwen36-p32-router-sigmoid-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-qwen36-paro-dual-format-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-qwen36-w1-unroll600-ablation-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen35-rocprof-amdahl-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-512-128-paro-blocker-profile.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-bench-paro-comparison-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-bulk-moe-prefill-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-bulk-parity-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-decode-pack8-raw-partial.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-decode-profile-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-expert-pack8-sidecar-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-fast-bulk-default-promoted-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-full-attn-parity-fixed-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-intake-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-linear-recurrent-parity-fixed-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-native-attention-bulk-moe-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-public-generate-smoke.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-35b-a3b-q4km-selected-device-experts-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-packed-shared-decode-fusion-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-shisa-force-legacy-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-hipengine-qwen36-shisa-packed-vs-legacy-refresh-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-llamacpp-hip-qwen36-peak.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-llamacpp-upstream-gfx1151-qwen36-gguf-rerun-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-17-llamacpp-vulkan-qwen36-peak.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-gfx1100-qwen36-27b-paro-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-gfx1100-shisa-qwen36-packed-gt1k-default-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-p8_2-dense-q4k-wmma-prefill-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen35-gt1k-prefill-chunk-policy-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen35-int8-kv-128k-quality-perf-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen35-int8-kv-256k-capacity-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen35-int8-kv-256k-single-buffer-capacity-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen35-int8-kv-aotriton-query-reuse-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen35-int8-kv-scratch-release-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p8-compact-moe-wmma-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_a3-gdn-k2-chain-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_c1-wmma-tile-sweep-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_c10-combined-gap-analysis.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_c11-hot-expert-final-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_c3-selected-moe-profile.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_c4-q4-hot-fulltile-v1-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_c5-q4-sidemeta-v1-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_c7-q5-opt-v1-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_c8-q6-retain-legacy.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-p9_c9-tail-no-padding-not-retained.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-18-hipengine-qwen36-35b-a3b-q4km-prefill-q8-wmma-p8_1.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-35b-a3b-q4km-p9_b7-decode-gemv-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-35b-a3b-q4km-p9_c12-q4t16-repack-design.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-35b-a3b-q4km-p9_c13-q4t16-materializer.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-35b-a3b-q4km-p9_e2-e2e-correctness-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-35b-a3b-q4km-p9_h1-fastpath-safety-correctness-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-35b-a3b-q4km-p9_h2-decode-repack-design.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-35b-a3b-q4km-p9_h3-t16-512x128-bench.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-35b-a3b-q4km-p9_h3-t16-e2e-correctness-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-35b-a3b-q4km-p9_h3-t16-rocprof-512x16-summary.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-hipengine-qwen36-packed-int8-kv-readme-memory-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-19-llamacpp-mtp-qwen36-27b-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p10-b5-p9-e2e-gate-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p10-b6-acceptance-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p10-wave1-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_c14-q4t16-selected-wmma-prototype.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_c15-q4t16-replay-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_c16-selected-moe-alternatives.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_c17-no-q4-redesign-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d1-router-split-coop.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d10-q8t16-dual-split-64.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d11-rejected-q8t16-shared-silu.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d15-dense-dual-alpha-beta.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d16-q8t16-f32-ssm-out.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d17-key-bf16-rope.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d18-splitk-gqa-gate.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d2-bf16-key-rope-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d4-q4t16-silu-decode.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d6-q8t16-pair-dispatch.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d7-q8t16-qkv-gate-pair.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d8-q8t16-dcache-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_d9-q8t16-triple-qkv.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_g1-final-acceptance-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_h3-rejected-attn-gate-fusion.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_h3-rejected-attn128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_h3-rejected-q4t16-silu256.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_h3-rejected-q5q6-direct-probes.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-20-hipengine-qwen36-35b-a3b-q4km-p9_h3-rejected-q6dense-dpreload.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-aotriton-v2-v3-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-d10-splitk-rocprof.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-d11-comparison-review.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-d8-splitk-decode.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-d9-splitk-sweep.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-decode-repack-residency-audit.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-long-context-chunked-smoke.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-long-context-preflight-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-memory-decode-pass-review.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-no-prefill-scratch-kv.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-q8-t16-scale-broadcast-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-r0-rocprof-baseline.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-r1-post-x1-rocprof.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-retained-safe-mode.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-selected-moe-t16-launchbounds-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-task39-32k-smoke.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-task40-prefill-vs-paro-diagnosis.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-hipengine-qwen36-35b-a3b-q4km-p10-x1-correctness-plus-x2-wmma-blocker.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-21-local-rx7900xtx-gguf-vs-paro-memory-comparison.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-after-memory-decode-pass-review.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-direct-selected-moe-c1-4k128-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-final-gate-4k128-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-q8-t16-decode-probes-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-q8-t16-second-pass-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-router256-4k128-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-selected-moe-down64-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-selected-moe-t16-qk256-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-small-kernel-second-pass-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-22-hipengine-qwen36-35b-a3b-q4km-q4ks-small-kernel-third-pass-rejected.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-hipengine-qwen36-35b-a3b-q4ks-persistent-session-w7900-gap-review.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-hipengine-qwen36-35b-a3b-q4ks-w7900-cold-start-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-hipengine-qwen36-35b-a3b-q4ks-w7900-readme-sweep-accepted.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-hipengine-qwen36-35b-a3b-q4ks-w7900-therock713-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-hipengine-w7900-therock713-gguf-paro-512-4k-spot-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-paro-512-prefill-workspace-overlap-rootcause-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-paro-prefill-workspace-overlap-threshold-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-gguf-q4km-1024-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-gguf-q4km-131072-128-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-gguf-q4km-32768-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-gguf-q4km-4096-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-gguf-q4km-512-128-rerun.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-gguf-q4km-65536-128-blocked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-gguf-q4ks-4096-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-paro-bf16kv-1024-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-paro-bf16kv-131072-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-paro-bf16kv-32768-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-paro-bf16kv-4096-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-paro-bf16kv-512-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-paro-bf16kv-65536-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-paro-int8kv-131072-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-hipengine-paro-int8kv-65536-128.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-llamacpp-hip-q4km-f16kv-sweep.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-llamacpp-hip-q4km-q8kv-maxctx.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-llamacpp-vulkan-q4km-f16kv-sweep.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-llamacpp-vulkan-q4km-q8kv-maxctx.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-paro-v011-current-regression-check.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-rocm7130423-current-head-paro-512-4k-check.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-rocm7130423-pure-current-head-paro-512-4k-check.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-rocm7130423-pure-packed-qwen36-paro-512-4k-check.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-rx7900xtx-rocm714-current-head-paro-512-4k-check.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-23-w7900-hipengine-therock713-paro-gguf-sweep-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-25-w7900-hipengine-gguf-q4ks-readme-persistent-5run.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-25-w7900-hipengine-paro-readme-persistent-5run.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/benchmarks/results/2026-05-25-w7900-hipengine-readme-persistent-5run-diagnostic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/BENCHMARK.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/DFLASH.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/ENVS.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/GGUF.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/GGUF_DECODE_REPACK.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/IMPLEMENTATION.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/KERNELS.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/KVCACHE.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/LESSONS-LEARNED.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/MARLIN.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/MTP.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/OPTIMIZE-DENSE.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/OPTIMIZE.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/PLAN.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/PREFILL.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/PUBLISH.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/README.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/RELAXED.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/ROOFLINE-gfx1151.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/ROOFLINE.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/SPECULATIVE-DECODE.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/TESTING.md +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/docs/source_lineage.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/fixtures/qwen35_paro/parent_512_32_seed1234.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hatch_build.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/benchmark/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/benchmark/correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/core/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/core/build.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/core/device.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/core/dtype.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/core/hip.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/core/memory.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/core/rocblas.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/core/tensor.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/dispatch/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/dispatch/batch.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/dispatch/fusion.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/dispatch/kv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/distributed/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/generation/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/generation/batch_scheduler.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/generation/qwen35_gguf.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/generation/registry.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/backends.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/cpu_reference/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/cpu_reference/fixtures.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/cpu_reference/ops.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/cuda_sm86/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_release.toml +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/MANIFEST.vendor.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/_internal/aiter_hip_common.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/_internal/flash/aiter.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/_internal/kernel_cluster.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/_internal/lazy_tensor_internal.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/_internal/packed_kernel.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/_internal/triton_kernel.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/_internal/util.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/config.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/cpp_tune.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/dtypes.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/flash.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/runtime.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/util.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/include/aotriton/v2/flash.h +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_F_F_0_0___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_F_F_0_1___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_F_F_3_0___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_F_T_0_0___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_F_T_0_1___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_F_T_3_0___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_T_F_0_0___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_T_F_0_1___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_T_F_3_0___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_T_T_0_0___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_T_T_0_1___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/aotriton.images/amd-gfx11xx/flash/attn_fwd/FONLY__/357/274/212bf16@16_256_T_T_3_0___gfx11xx.aks2" +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/libaotriton_v2.so +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_runtime/0.11.2b/lib/libaotriton_v2.so.0.11.2 +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_wrap.cc +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/aotriton_wrap.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/paged_attn_decode.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/paged_attn_decode.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/paged_kv_write.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/attention/paged_kv_write.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/common/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/convert/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/convert/cast.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/convert/cast.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/fused/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/fused/gguf_ops.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/fused/gguf_ops.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/fused/paro_combine.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/fused/paro_combine.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/fused/paro_silu.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/fused/paro_silu.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear/dense_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear/dense_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear/lm_head.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear/lm_head.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear_attn/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear_attn/conv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear_attn/conv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear_attn/gdn.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/linear_attn/gdn.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/moe/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/moe/group_scatter.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/moe/group_scatter.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/moe/prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/moe/router.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/moe/router.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/norm/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/norm/rmsnorm.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/norm/rmsnorm.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_expert_pack8_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_expert_pack8_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_k_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_k_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_k_selected_pack8_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_k_selected_pack8_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_k_selected_prefill.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_k_selected_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_k_t16_selected_prefill.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_k_t16_selected_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_pack8_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_pack8_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_prefill.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_selected_pack8_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_selected_pack8_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_selected_prefill.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_selected_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_t16_selected_prefill.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q4_k_t16_selected_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q6_k_embedding.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q6_k_embedding.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q6_k_pack8_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q6_k_pack8_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q6_k_t16_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q6_k_t16_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q8_0_pack8_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q8_0_pack8_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q8_0_prefill.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q8_0_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q8_0_t16_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q8_0_t16_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q8_0_t16_prefill.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_q8_0_t16_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_t16_selected_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/gguf_t16_selected_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/paro_awq_gemv.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/paro_awq_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/paro_marlin_k.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/paro_marlin_k.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/w8a16_linear.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/quant/w8a16_linear.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/rotary/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/rotary/paro_rotate.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/rotary/paro_rotate.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/rotary/qwen35_rotary.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/rotary/qwen35_rotary.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/runtime/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/runtime/state.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/runtime/state.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/smoke/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/smoke/smoke_add.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/smoke/smoke_add.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/wmma/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/wmma/paro_awq_wmma.hip +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1100/wmma/paro_awq_wmma.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/hip_gfx1151/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kernels/registry.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kvcache/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kvcache/policy.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/kvcache/spans.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/layers/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/layers/base.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/loading/materialize.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/loading/qwen35_gguf.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/loading/qwen35_gguf_expert_sidecar.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/loading/qwen35_gguf_materialize.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/loading/qwen35_paro.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/models/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/models/base.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/models/qwen35.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/models/registry.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/models/toy.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/base.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/bf16.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/fp16.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/gguf.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/gguf_k.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/gguf_q4_k.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/gguf_t16.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/registry.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/quant/w4_paro.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/runtime/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/runtime/gguf_embedding.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/runtime/gguf_linear.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/runtime/prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/runtime/qwen35_gguf_runner.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/runtime/qwen35_paro.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/runtime/qwen35_paro_runner.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/runtime/workspace.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/server/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/speculative/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/speculative/interfaces.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/tokenization/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/tokenization/gguf.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/util/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/hipengine/util/amdgpu_vram.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/__init__.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/check_fixtures.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/check_lineage.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/fetch_aotriton.sh +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/gdn_decode_probe.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/gguf_k_gemv_smoke.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/gguf_prefill_projection_smoke.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/gguf_q6_k_embedding_smoke.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/inspect_gguf.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/llamacpp_mtp_bench.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_batch_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_batch_packed_prefill_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_batch_serial_bench.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_batch_serial_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_compare_tables.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_decode_graph_fixture_gate.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_dflash_ddtree_blocker.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_e2e_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_aotriton_prefill_sweep.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_bench.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_build_expert_sidecar.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_bulk_parity.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_decode_graph_smoke.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_e2e_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_expert_pack8_smoke.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_moe_replay.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_p9_e2e_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_gguf_rocprof_summary.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_kv_e2e_fixture_gate.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_kv_int8_accuracy.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_kv_policy_args.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_native_compact_prefill_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_native_prefill_boundary.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_native_prefill_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_native_prefill_fixture_gate.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_native_prefill_fullattn_stage_probe.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_native_prefill_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_native_prefill_stage_probe.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_paro_bench.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_paro_next_token.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_paro_packed_bench.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_readme_sweep.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/qwen35_rocprof_audit.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/resolve_worklog_conflict.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/smoke.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/strip_paro_safetensors.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/vendor_aotriton.sh +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/scripts/w8a16_decode_probe.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/_gguf_synthetic_weights.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/conftest.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/cpu_reference/attention_decode_masked.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/cpu_reference/full_attn_prefill_causal_gqa_gate.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/cpu_reference/kv_int8_dequant_per_token_head.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/cpu_reference/linear_basic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/cpu_reference/paged_attn_decode_int8_per_token_head.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/cpu_reference/rmsnorm_basic.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/cpu_reference/rotate_split_half.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/gguf/qwen35_0_8b_q4_1_e2e.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/gguf/qwen35_0_8b_q4_k_m_e2e.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/gguf/qwen35_0_8b_q8_0_e2e.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/gguf/qwen35_0_8b_ud_q4_k_xl_e2e.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/gguf/qwen36_35b_a3b_q4km_p9_e2e.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/fixtures/gguf/qwen36_35b_a3b_q4km_smoke.json +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_aotriton_discovery.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_build.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_cast_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_check_lineage.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_cpu_reference.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_dense_gemv_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_dispatch_batch.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_fusion_spike.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_generation_batch_scheduler.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gfx1151_backend.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_e2e_acceptance.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_embedding_dispatch.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_expert_pack8_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_gemv_decode_dispatch.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_k_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_k_selected_pack8_gemv_decode.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_k_selected_wmma_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_k_t16_selected_wmma_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_linear_dispatch.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_ops.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q4_k_gemv.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q4_k_pack8_gemv_decode.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q4_k_selected_dual_pack8_gemv_decode.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q4_k_selected_wmma_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q4_k_t16_selected_wmma_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q4_k_tile16_repack.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q4_k_wmma_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q6_k_embedding.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q6_k_pack8_gemv_decode.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q6_k_t16_gemv_decode.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q8_0_pack8_gemv_decode.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q8_0_t16_gemv_decode.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q8_0_t16_wmma_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q8_0_wmma_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_q8_0_wmma_prefill_dual.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_quant_layout.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_reader.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_t16_repack.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_gguf_t16_selected_gemv_decode.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_hip_runtime.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_kernel_registry.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_kv_dispatch.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_kvcache_policy.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_kvcache_spans.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_llm_gguf_generate_path.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_lm_head_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_loading_materialize.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_loading_safetensors.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_memory_stats.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_model_quant_and_imports.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_paro_awq_gemv_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_paro_awq_wmma_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_paro_combine_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_paro_rotate_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_paro_silu_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_bench_memory_audit.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_decode_state.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_chunked_prefill.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_compact_moe_gemv_routing.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_compact_moe_wmma_resolver.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_compact_moe_wmma_routing.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_decode_graph_policy.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_decode_repack_dispatch.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_decode_repack_semantics.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_expert_sidecar.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_fastpath_safety.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_full_attention_gpu.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_gdn_prefill_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_gdn_prefill_routing.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_mapping.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_materialize.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_moe_replay.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_p10_x2_layer_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_p9_e2e_correctness.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_rocprof_summary.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_runner.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_gguf_tokenizer.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_kv_e2e_fixture_gate.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_kv_int8_accuracy.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_linear_attn_conv_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_linear_attn_gdn_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_moe_group_scatter_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_native_prefill_boundary.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_native_prefill_fullattn_stage_probe.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_paged_attn_decode_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_paged_kv_write_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_paro_layout.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_paro_marlin_k.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_prefill_workspace_policy.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_resident_batch_layout.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_rmsnorm_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_rotary_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_qwen35_router_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_runtime_state_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_runtime_workspace.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_smoke_add_plan.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_speculative_interfaces.py +0 -0
- {hipengine-0.2.0 → hipengine-0.2.1}/tests/test_w8a16_linear_plan.py +0 -0
|
@@ -6,6 +6,47 @@ This changelog is for package/API releases. Performance rollup history remains i
|
|
|
6
6
|
[`benchmarks/CHANGELOG.md`](benchmarks/CHANGELOG.md), with detailed benchmark
|
|
7
7
|
evidence under [`benchmarks/results/`](benchmarks/results/).
|
|
8
8
|
|
|
9
|
+
## v0.2.1 - 2026-05-25
|
|
10
|
+
|
|
11
|
+
Patch release improving server session management, streaming, and
|
|
12
|
+
OpenAI-compatible reasoning output.
|
|
13
|
+
|
|
14
|
+
### Added
|
|
15
|
+
|
|
16
|
+
- Eager model warmup on server startup: the configured model and a short
|
|
17
|
+
warmup generation run before uvicorn reports ready, so the first real
|
|
18
|
+
request does not pay load/compile cost. Controlled by `--eager-load` /
|
|
19
|
+
`--no-eager-load` (default: on), `--eager-load-prompt`, and
|
|
20
|
+
`--eager-load-max-tokens`, with `HIPENGINE_EAGER_LOAD`,
|
|
21
|
+
`HIPENGINE_EAGER_LOAD_PROMPT`, and `HIPENGINE_EAGER_LOAD_MAX_TOKENS`
|
|
22
|
+
environment variable equivalents.
|
|
23
|
+
- `LLM.stream()` method for single-prompt token-by-token generation when
|
|
24
|
+
the underlying text generator supports it.
|
|
25
|
+
- Reasoning-content splitting for chat completions: `<think>…</think>`
|
|
26
|
+
spans (Qwen/DeepSeek-style) are now separated into
|
|
27
|
+
`message.reasoning_content` (non-streaming) or `delta.reasoning_content`
|
|
28
|
+
chunks (streaming), matching the OpenAI reasoning-content convention.
|
|
29
|
+
|
|
30
|
+
### Changed
|
|
31
|
+
|
|
32
|
+
- PARO text generators and their resident sessions are now cached on the
|
|
33
|
+
`LLM` instance and reused across requests. Session capacity is bucketed
|
|
34
|
+
(floor 4 Ki tokens, configurable via `HIPENGINE_SESSION_MIN_TOKENS` and
|
|
35
|
+
`HIPENGINE_SESSION_BUCKET_TOKENS`) so normal chat-history growth does not
|
|
36
|
+
force reallocation every turn.
|
|
37
|
+
- Chat `stream=true` now yields token-level SSE chunks from the resident
|
|
38
|
+
decode loop instead of buffering the full response and wrapping it in a
|
|
39
|
+
single SSE frame.
|
|
40
|
+
- Chat completions default `max_tokens` raised from 16 to 8192 so clients
|
|
41
|
+
that omit the field get usable reply lengths, including verbose
|
|
42
|
+
chain-of-thought reasoning.
|
|
43
|
+
|
|
44
|
+
### Fixed
|
|
45
|
+
|
|
46
|
+
- Fixed `LLM.generate()` re-resolving the generation factory on every call,
|
|
47
|
+
which discarded generator-local caches and caused the PARO resident
|
|
48
|
+
session (layer weights, KV buffers) to be allocated and freed per request.
|
|
49
|
+
|
|
9
50
|
## v0.2.0 - 2026-05-25
|
|
10
51
|
|
|
11
52
|
Minor release for the GGUF runtime path and W7900 benchmark refresh. GGUF is a
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: hipengine
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: ROCm-native local LLM inference engine with a torch-free runtime hot path
|
|
5
5
|
Project-URL: Homepage, https://github.com/shisa-ai/hipEngine
|
|
6
6
|
Project-URL: Repository, https://github.com/shisa-ai/hipEngine
|
|
@@ -75,7 +75,7 @@ supported GPUs and models.
|
|
|
75
75
|
|
|
76
76
|
## Status
|
|
77
77
|
|
|
78
|
-
**v0.2.
|
|
78
|
+
**v0.2.1 alpha.** The runtime hot path is torch-free by construction, and the
|
|
79
79
|
first two 35B-class model-loading surfaces are now available on gfx1100:
|
|
80
80
|
[shisa-ai/Qwen3.6-35B-A3B-PARO-full4096-e5-packed](https://huggingface.co/shisa-ai/Qwen3.6-35B-A3B-PARO-full4096-e5-packed)
|
|
81
81
|
(19.07 GiB, 4.68 bpw) in packed
|
|
@@ -336,13 +336,20 @@ Install the optional server extra and run the FastAPI layer:
|
|
|
336
336
|
```bash
|
|
337
337
|
pip install -e ".[server]"
|
|
338
338
|
python -m hipengine.server \
|
|
339
|
-
--model /
|
|
339
|
+
--model shisa-ai/Qwen3.6-35B-A3B-PARO-full4096-e5-packed \
|
|
340
340
|
--quant w4_paro \
|
|
341
341
|
--served-model-name qwen-paro
|
|
342
342
|
```
|
|
343
343
|
|
|
344
|
-
|
|
345
|
-
|
|
344
|
+
`--model` accepts either a local filesystem path or a Hugging Face model ID
|
|
345
|
+
already present in the local HF cache; hipEngine resolves IDs locally and does
|
|
346
|
+
not download weights during startup.
|
|
347
|
+
|
|
348
|
+
Supported endpoints: `GET /v1/models`, `POST /v1/completions`, and
|
|
349
|
+
`POST /v1/chat/completions` with token-level SSE streaming. Chat responses
|
|
350
|
+
separate `<think>` reasoning into `reasoning_content` (matching the OpenAI
|
|
351
|
+
reasoning-content convention). The server eagerly warms the model on startup
|
|
352
|
+
by default so the first request does not pay load/compile cost. See
|
|
346
353
|
[`docs/API.md`](docs/API.md) for request examples, bearer-token auth, and
|
|
347
354
|
current limitations.
|
|
348
355
|
|
|
@@ -37,7 +37,7 @@ supported GPUs and models.
|
|
|
37
37
|
|
|
38
38
|
## Status
|
|
39
39
|
|
|
40
|
-
**v0.2.
|
|
40
|
+
**v0.2.1 alpha.** The runtime hot path is torch-free by construction, and the
|
|
41
41
|
first two 35B-class model-loading surfaces are now available on gfx1100:
|
|
42
42
|
[shisa-ai/Qwen3.6-35B-A3B-PARO-full4096-e5-packed](https://huggingface.co/shisa-ai/Qwen3.6-35B-A3B-PARO-full4096-e5-packed)
|
|
43
43
|
(19.07 GiB, 4.68 bpw) in packed
|
|
@@ -298,13 +298,20 @@ Install the optional server extra and run the FastAPI layer:
|
|
|
298
298
|
```bash
|
|
299
299
|
pip install -e ".[server]"
|
|
300
300
|
python -m hipengine.server \
|
|
301
|
-
--model /
|
|
301
|
+
--model shisa-ai/Qwen3.6-35B-A3B-PARO-full4096-e5-packed \
|
|
302
302
|
--quant w4_paro \
|
|
303
303
|
--served-model-name qwen-paro
|
|
304
304
|
```
|
|
305
305
|
|
|
306
|
-
|
|
307
|
-
|
|
306
|
+
`--model` accepts either a local filesystem path or a Hugging Face model ID
|
|
307
|
+
already present in the local HF cache; hipEngine resolves IDs locally and does
|
|
308
|
+
not download weights during startup.
|
|
309
|
+
|
|
310
|
+
Supported endpoints: `GET /v1/models`, `POST /v1/completions`, and
|
|
311
|
+
`POST /v1/chat/completions` with token-level SSE streaming. Chat responses
|
|
312
|
+
separate `<think>` reasoning into `reasoning_content` (matching the OpenAI
|
|
313
|
+
reasoning-content convention). The server eagerly warms the model on startup
|
|
314
|
+
by default so the first request does not pay load/compile cost. See
|
|
308
315
|
[`docs/API.md`](docs/API.md) for request examples, bearer-token auth, and
|
|
309
316
|
current limitations.
|
|
310
317
|
|
|
@@ -26555,3 +26555,82 @@ uv run --extra dev python -m pytest -q
|
|
|
26555
26555
|
```
|
|
26556
26556
|
|
|
26557
26557
|
Next: commit, push `main`, move annotated tag `v0.2.0` to the fix commit, force-push the tag per user approval, and watch the publish workflow complete via trusted publishing.
|
|
26558
|
+
|
|
26559
|
+
### v0.2.0 publish workflow result
|
|
26560
|
+
|
|
26561
|
+
Committed and pushed the HIP guard fix (`36e3516`), force-moved annotated tag `v0.2.0` to that commit per user approval, and re-pushed the tag. The `Publish to PyPI` workflow reran as `26370452261` and completed successfully: build validation passed, artifact attestation was generated, and `Publish to PyPI (trusted publishing)` succeeded. Verified published install path:
|
|
26562
|
+
|
|
26563
|
+
```bash
|
|
26564
|
+
uvx --refresh --from "hipengine[server]==0.2.0" hipengine-server --help
|
|
26565
|
+
# downloaded hipengine 0.2.0 from PyPI and printed CLI help
|
|
26566
|
+
```
|
|
26567
|
+
|
|
26568
|
+
Final release pointers: GitHub release `https://github.com/shisa-ai/hipEngine/releases/tag/v0.2.0`, publish workflow `https://github.com/shisa-ai/hipEngine/actions/runs/26370452261`, release tag commit `36e351607955158e821b51f443bac042b140624f`.
|
|
26569
|
+
|
|
26570
|
+
## 2026-05-25 — HF cache model id resolution for server/LLM
|
|
26571
|
+
|
|
26572
|
+
Added local Hugging Face cache resolution for public model references so `hipengine-server --model` and `LLM(model=...)` can take either a filesystem path or a cached HF model id such as `shisa-ai/Qwen3.6-35B-A3B-PARO-full4096-e5-packed`. Resolution remains local-only: existing paths win, then `huggingface_hub.snapshot_download(..., local_files_only=True)` is used if available, then the standard HF cache layout is inspected directly; no weights are downloaded during startup. GGUF discovery now resolves the model reference first, so cached GGUF repos/directories work as well as direct `.gguf` file paths.
|
|
26573
|
+
|
|
26574
|
+
Also updated README/API server examples to show the HF model id form and note that cache resolution is local-only. User's `uv run --extra server ...` command is the right shape; the `--no-build` error indicates a uv no-build mode/config/environment for the local editable package, not a hipEngine server argument issue. Workaround is to unset no-build for local editable runs or use the published wheel via `uvx --from "hipengine[server]==0.2.0"`.
|
|
26575
|
+
|
|
26576
|
+
Validation:
|
|
26577
|
+
|
|
26578
|
+
```bash
|
|
26579
|
+
python -m py_compile hipengine/loading/hf_cache.py hipengine/loading/safetensors.py hipengine/loading/gguf.py hipengine/llm.py
|
|
26580
|
+
python -m pytest -q tests/test_hf_cache.py tests/test_llm_generate.py tests/test_loading_safetensors.py tests/test_server_api.py
|
|
26581
|
+
# 19 passed
|
|
26582
|
+
uv run --extra dev python -m pytest -q
|
|
26583
|
+
# passed
|
|
26584
|
+
```
|
|
26585
|
+
## 2026-05-23 — llama.cpp wrapper metadata fix for RX 7900 XTX 5-run sweep
|
|
26586
|
+
|
|
26587
|
+
Before rerunning `benchmarks/7900XTX.md` with `--repetitions 5`, fixed `scripts/llamacpp_bench_with_peak.py` artifact metadata so it no longer hardcodes W7900 or a single-shot `--repetitions 1` note. The artifact now records the selected amdgpu card name/PCI/VRAM in `hardware` and emits a dynamic repetitions note. Validation: `python3 -m py_compile scripts/llamacpp_bench_with_peak.py`.
|
|
26588
|
+
|
|
26589
|
+
## 2026-05-24 — server resident-session reuse smoke
|
|
26590
|
+
|
|
26591
|
+
Debugged the local OpenAI-compatible PARO server after LAN chat showed short `<think>` replies and apparent reloads. Root causes:
|
|
26592
|
+
|
|
26593
|
+
- `LLM.generate()` re-resolved the generation factory and constructed a new text generator on every call, discarding generator-local caches.
|
|
26594
|
+
- `Qwen35ParoOneTokenGenerator` then constructed `Qwen35ParoResidentSession` inside each prompt call and closed it immediately, so resident layer weights/KV buffers were materialized and freed per request.
|
|
26595
|
+
- The OpenAI chat server default `max_tokens` was 16; clients that omit `max_tokens` saw very short replies. The Qwen/PARO model config advertises `max_position_embeddings=262144`, but the runtime session capacity was allocated as `len(prompt_ids) + max_tokens + 1` per request.
|
|
26596
|
+
- `stream=true` only returned an SSE wrapper around the completed response; it did not yield tokens while generation was running.
|
|
26597
|
+
|
|
26598
|
+
Fixes made in-tree:
|
|
26599
|
+
|
|
26600
|
+
- Cache the resolved text generator on `hipengine.LLM` so server `app.state.hipengine_llm` keeps its backend/model/quant generator across requests.
|
|
26601
|
+
- Cache/reuse `Qwen35ParoResidentSession` inside the PARO generator when the existing session capacity and KV policy cover the next request; reset it between prompts instead of closing it. Session capacity now floors/buckets at 4096 tokens by default (`HIPENGINE_SESSION_MIN_TOKENS`, `HIPENGINE_SESSION_BUCKET_TOKENS`) so normal chat history growth does not force reallocation every turn.
|
|
26602
|
+
- Add a single-prompt streaming path using resident `step()` per token, and route chat `stream=true` through it. Chat default `max_tokens` is now 256.
|
|
26603
|
+
|
|
26604
|
+
Validation:
|
|
26605
|
+
|
|
26606
|
+
```bash
|
|
26607
|
+
uv run --extra dev python -m pytest -q tests/test_llm_generate.py tests/test_generation_qwen35_paro.py tests/test_server_api.py
|
|
26608
|
+
# 17 passed
|
|
26609
|
+
```
|
|
26610
|
+
|
|
26611
|
+
Restarted the running LAN server from the local checkout with TheRock ROCm library paths:
|
|
26612
|
+
|
|
26613
|
+
```bash
|
|
26614
|
+
uv run --extra server hipengine-server --model shisa-ai/Qwen3.6-35B-A3B-PARO-full4096-e5-packed --quant w4_paro --served-model-name qwen-paro --host 0.0.0.0 --port 8000
|
|
26615
|
+
```
|
|
26616
|
+
|
|
26617
|
+
Smoke evidence on the running server after restart from the local checkout: first streamed `max_tokens=4` request sent the role chunk immediately, then took 30.25s to warm/materialize and yielded token chunks (`<think>`, newline, `Here`, `'s`); VRAM reached 19.54GB. A second chat-history request reused the resident 4096-token session, yielded the first content chunk at 0.46s, completed at 0.49s, and VRAM stayed resident at 19.55GB. Server PID 9305, wrapper PID recorded in `/tmp/hipengine-server-8000.pid`, log `/tmp/hipengine-server-8000.log`.
|
|
26618
|
+
|
|
26619
|
+
## 2026-05-24 — eager server warmup and reasoning-channel split
|
|
26620
|
+
|
|
26621
|
+
Implemented eager server warmup and OpenAI-compatible reasoning segregation for the local PARO server.
|
|
26622
|
+
|
|
26623
|
+
Changes:
|
|
26624
|
+
|
|
26625
|
+
- Added `ServerConfig.eager_load` plus CLI `--eager-load/--no-eager-load` (default on), `--eager-load-prompt`, and `--eager-load-max-tokens`. Startup now constructs `LLM` and runs a one-token warmup before uvicorn reports startup complete, so `/v1/models` is only reachable after the model/session is resident.
|
|
26626
|
+
- Default warmup prompt is `one two three four`; the initial `hello` default failed native prefill because Qwen35/PARO requires at least `linear_conv_kernel_dim` (4) prompt tokens.
|
|
26627
|
+
- Split Qwen/DeepSeek-style `<think>...</think>` spans in chat responses. Non-stream responses put visible answer text in `message.content` and hidden reasoning text in `message.reasoning_content`. Streaming responses emit `delta.reasoning_content` for reasoning chunks and `delta.content` for final answer chunks.
|
|
26628
|
+
|
|
26629
|
+
Validation:
|
|
26630
|
+
|
|
26631
|
+
```bash
|
|
26632
|
+
uv run --extra dev python -m pytest -q tests/test_server_api.py tests/test_llm_generate.py tests/test_generation_qwen35_paro.py
|
|
26633
|
+
# 19 passed
|
|
26634
|
+
```
|
|
26635
|
+
|
|
26636
|
+
Restarted the LAN server from the local checkout with TheRock ROCm library paths. Eager startup completed after 31s, `/v1/models` returned `qwen-paro`, and VRAM was already resident at 19.54GB before serving requests. A subsequent streamed chat request returned the role chunk at 0.01s and reasoning chunks as `delta.reasoning_content` from 0.49s onward, with VRAM staying resident at 19.54GB. Server PID 12107, wrapper PID 12082, log `/tmp/hipengine-server-8000.log`.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# OpenAI-Compatible Server API
|
|
2
2
|
|
|
3
|
-
Last updated: 2026-05-
|
|
3
|
+
Last updated: 2026-05-25
|
|
4
4
|
|
|
5
5
|
hipEngine ships a thin optional FastAPI layer that adapts OpenAI-style requests
|
|
6
6
|
to the torch-free `hipengine.LLM.generate()` library API. It is installed only
|
|
@@ -17,13 +17,17 @@ pip install -e ".[server]"
|
|
|
17
17
|
|
|
18
18
|
```bash
|
|
19
19
|
python -m hipengine.server \
|
|
20
|
-
--model /
|
|
20
|
+
--model shisa-ai/Qwen3.6-35B-A3B-PARO-full4096-e5-packed \
|
|
21
21
|
--quant w4_paro \
|
|
22
22
|
--served-model-name qwen-paro \
|
|
23
23
|
--host 127.0.0.1 \
|
|
24
24
|
--port 8000
|
|
25
25
|
```
|
|
26
26
|
|
|
27
|
+
`--model` accepts a local filesystem path or a Hugging Face model ID that is
|
|
28
|
+
already present in the local HF cache. hipEngine resolves IDs with local cache
|
|
29
|
+
lookups only; it does not download weights during server startup.
|
|
30
|
+
|
|
27
31
|
After installation, the console script is equivalent:
|
|
28
32
|
|
|
29
33
|
```bash
|
|
@@ -36,6 +40,12 @@ select `cpu_reference` where a CPU implementation exists; nearby targets such as
|
|
|
36
40
|
`gfx1101`/`gfx1102` can force a backend with `--backend hip_gfx1100` or
|
|
37
41
|
`HIPENGINE_BACKEND=hip_gfx1100` after local validation.
|
|
38
42
|
|
|
43
|
+
By default the server eagerly loads the model and runs a short warmup
|
|
44
|
+
generation at startup so the first real request does not pay load/compile cost.
|
|
45
|
+
Disable with `--no-eager-load` or `HIPENGINE_EAGER_LOAD=0`. The warmup prompt
|
|
46
|
+
and token count are configurable via `--eager-load-prompt` and
|
|
47
|
+
`--eager-load-max-tokens`.
|
|
48
|
+
|
|
39
49
|
Set `HIPENGINE_API_KEY` or pass `--api-key` to require OpenAI-style bearer
|
|
40
50
|
authentication:
|
|
41
51
|
|
|
@@ -50,8 +60,8 @@ curl -H 'Authorization: Bearer local-secret' http://127.0.0.1:8000/v1/models
|
|
|
50
60
|
| --- | --- | --- |
|
|
51
61
|
| `GET /health` | Built in | Unauthenticated health/model probe. |
|
|
52
62
|
| `GET /v1/models` | Built in | Returns the single served model id. |
|
|
53
|
-
| `POST /v1/completions` | Built in | Text prompt(s) to `LLM.generate()`. Supports `stream=true`
|
|
54
|
-
| `POST /v1/chat/completions` | Built in | Renders text-only messages to a Qwen-style prompt and calls `LLM.generate()`. Supports `stream=true`
|
|
63
|
+
| `POST /v1/completions` | Built in | Text prompt(s) to `LLM.generate()`. Supports `stream=true` (one SSE chunk plus `[DONE]`). |
|
|
64
|
+
| `POST /v1/chat/completions` | Built in | Renders text-only messages to a Qwen-style prompt and calls `LLM.generate()`. Supports token-level `stream=true` SSE. `<think>` spans are separated into `reasoning_content` (non-streaming) or `delta.reasoning_content` chunks (streaming). |
|
|
55
65
|
|
|
56
66
|
## Examples
|
|
57
67
|
|
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
"""Qwen3.5/PARO text generation bring-up path."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Iterator
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
import os
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from hipengine.generation.registry import GenerationRequest, register_text_generator
|
|
12
|
+
from hipengine.kvcache import resolve_kv_policy
|
|
13
|
+
from hipengine.loading import WeightIndex
|
|
14
|
+
from hipengine.runtime.qwen35_paro_runner import (
|
|
15
|
+
Qwen35ParoNextTokenRunner,
|
|
16
|
+
Qwen35ParoResidentSession,
|
|
17
|
+
_decode_token_cached,
|
|
18
|
+
_select_token,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class Qwen35ParoOneTokenGenerator:
|
|
24
|
+
"""Greedy Qwen3.5/PARO generator backed by resident c=1 execution.
|
|
25
|
+
|
|
26
|
+
The implementation is still serial across prompts, but each prompt uses the
|
|
27
|
+
resident single-request native prefill path followed by multi-token
|
|
28
|
+
autoregressive decode using the resident HIP layer chain.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
model_path: str | Path
|
|
32
|
+
weight_index: WeightIndex
|
|
33
|
+
model_plugin: Any
|
|
34
|
+
backend: str = "auto"
|
|
35
|
+
lm_head_chunk: int = 4096
|
|
36
|
+
_runner: Qwen35ParoNextTokenRunner | None = field(default=None, init=False, repr=False)
|
|
37
|
+
_session: Qwen35ParoResidentSession | None = field(default=None, init=False, repr=False)
|
|
38
|
+
_session_capacity: int = field(default=0, init=False, repr=False)
|
|
39
|
+
_session_kv_key: tuple[str, str, str, int] | None = field(default=None, init=False, repr=False)
|
|
40
|
+
|
|
41
|
+
def generate(self, request: GenerationRequest) -> list[str]:
|
|
42
|
+
if request.max_tokens < 0:
|
|
43
|
+
raise ValueError("max_tokens must be non-negative")
|
|
44
|
+
if request.temperature != 0.0 or request.top_p != 1.0:
|
|
45
|
+
raise NotImplementedError(
|
|
46
|
+
"Qwen3.5/PARO generator currently supports greedy sampling only"
|
|
47
|
+
)
|
|
48
|
+
if request.max_tokens == 0:
|
|
49
|
+
return ["" for _ in request.prompts]
|
|
50
|
+
runner = self._get_runner()
|
|
51
|
+
kv_policy = resolve_kv_policy(
|
|
52
|
+
request.kv_storage,
|
|
53
|
+
scale_dtype=request.kv_scale_dtype,
|
|
54
|
+
scale_granularity=request.kv_scale_granularity,
|
|
55
|
+
)
|
|
56
|
+
return [
|
|
57
|
+
self._generate_one(
|
|
58
|
+
runner,
|
|
59
|
+
prompt,
|
|
60
|
+
request.max_tokens,
|
|
61
|
+
ignore_eos=request.ignore_eos,
|
|
62
|
+
kv_policy=kv_policy,
|
|
63
|
+
)
|
|
64
|
+
for prompt in request.prompts
|
|
65
|
+
]
|
|
66
|
+
|
|
67
|
+
def stream(self, request: GenerationRequest) -> Iterator[str]:
|
|
68
|
+
if len(request.prompts) != 1:
|
|
69
|
+
raise ValueError("streaming currently supports exactly one prompt")
|
|
70
|
+
if request.max_tokens < 0:
|
|
71
|
+
raise ValueError("max_tokens must be non-negative")
|
|
72
|
+
if request.temperature != 0.0 or request.top_p != 1.0:
|
|
73
|
+
raise NotImplementedError(
|
|
74
|
+
"Qwen3.5/PARO generator currently supports greedy sampling only"
|
|
75
|
+
)
|
|
76
|
+
if request.max_tokens == 0:
|
|
77
|
+
return
|
|
78
|
+
runner = self._get_runner()
|
|
79
|
+
kv_policy = resolve_kv_policy(
|
|
80
|
+
request.kv_storage,
|
|
81
|
+
scale_dtype=request.kv_scale_dtype,
|
|
82
|
+
scale_granularity=request.kv_scale_granularity,
|
|
83
|
+
)
|
|
84
|
+
yield from self._stream_one(
|
|
85
|
+
runner,
|
|
86
|
+
request.prompts[0],
|
|
87
|
+
request.max_tokens,
|
|
88
|
+
ignore_eos=request.ignore_eos,
|
|
89
|
+
kv_policy=kv_policy,
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
def _generate_one(
|
|
93
|
+
self,
|
|
94
|
+
runner: Qwen35ParoNextTokenRunner,
|
|
95
|
+
prompt: str,
|
|
96
|
+
max_tokens: int,
|
|
97
|
+
*,
|
|
98
|
+
ignore_eos: bool,
|
|
99
|
+
kv_policy,
|
|
100
|
+
) -> str:
|
|
101
|
+
_last_token_id, prompt_ids = _select_token(Path(self.model_path), prompt, None)
|
|
102
|
+
if not prompt_ids:
|
|
103
|
+
raise ValueError("prompt produced no tokens")
|
|
104
|
+
required_sequence_length = len(prompt_ids) + max_tokens + 1
|
|
105
|
+
session_capacity = _session_capacity_for(required_sequence_length)
|
|
106
|
+
generated_text: list[str] = []
|
|
107
|
+
session = self._get_session(
|
|
108
|
+
runner,
|
|
109
|
+
max_sequence_length=session_capacity,
|
|
110
|
+
kv_policy=kv_policy,
|
|
111
|
+
)
|
|
112
|
+
next_result = session.prefill_native(prompt_ids, sample=True)
|
|
113
|
+
if next_result is None:
|
|
114
|
+
raise RuntimeError("native prefill did not produce next-token logits")
|
|
115
|
+
generated_text.append(next_result.token_text)
|
|
116
|
+
if not ignore_eos and _is_eos(session.tokenizer, next_result.token_id):
|
|
117
|
+
return "".join(generated_text)
|
|
118
|
+
|
|
119
|
+
remaining = max_tokens - 1
|
|
120
|
+
if remaining:
|
|
121
|
+
with session.capture_decode_graph(
|
|
122
|
+
position=len(prompt_ids),
|
|
123
|
+
steps_per_replay=1,
|
|
124
|
+
max_replay_steps=remaining,
|
|
125
|
+
record_steps=remaining,
|
|
126
|
+
) as graph:
|
|
127
|
+
graph.replay(remaining)
|
|
128
|
+
token_ids = graph.read_generated_token_ids(remaining)
|
|
129
|
+
for token_id in token_ids:
|
|
130
|
+
generated_text.append(_decode_token_cached(session.tokenizer, token_id))
|
|
131
|
+
if not ignore_eos and _is_eos(session.tokenizer, token_id):
|
|
132
|
+
break
|
|
133
|
+
return "".join(generated_text)
|
|
134
|
+
|
|
135
|
+
def _stream_one(
|
|
136
|
+
self,
|
|
137
|
+
runner: Qwen35ParoNextTokenRunner,
|
|
138
|
+
prompt: str,
|
|
139
|
+
max_tokens: int,
|
|
140
|
+
*,
|
|
141
|
+
ignore_eos: bool,
|
|
142
|
+
kv_policy,
|
|
143
|
+
) -> Iterator[str]:
|
|
144
|
+
_last_token_id, prompt_ids = _select_token(Path(self.model_path), prompt, None)
|
|
145
|
+
if not prompt_ids:
|
|
146
|
+
raise ValueError("prompt produced no tokens")
|
|
147
|
+
required_sequence_length = len(prompt_ids) + max_tokens + 1
|
|
148
|
+
session_capacity = _session_capacity_for(required_sequence_length)
|
|
149
|
+
session = self._get_session(
|
|
150
|
+
runner,
|
|
151
|
+
max_sequence_length=session_capacity,
|
|
152
|
+
kv_policy=kv_policy,
|
|
153
|
+
)
|
|
154
|
+
next_result = session.prefill_native(prompt_ids, sample=True)
|
|
155
|
+
if next_result is None:
|
|
156
|
+
raise RuntimeError("native prefill did not produce next-token logits")
|
|
157
|
+
yield next_result.token_text
|
|
158
|
+
if not ignore_eos and _is_eos(session.tokenizer, next_result.token_id):
|
|
159
|
+
return
|
|
160
|
+
|
|
161
|
+
current_token_id = next_result.token_id
|
|
162
|
+
for position in range(len(prompt_ids), len(prompt_ids) + max_tokens - 1):
|
|
163
|
+
result = session.step(current_token_id, position=position, sample=True)
|
|
164
|
+
if result is None:
|
|
165
|
+
raise RuntimeError("decode step did not produce next-token logits")
|
|
166
|
+
yield result.token_text
|
|
167
|
+
current_token_id = result.token_id
|
|
168
|
+
if not ignore_eos and _is_eos(session.tokenizer, result.token_id):
|
|
169
|
+
return
|
|
170
|
+
|
|
171
|
+
def _get_runner(self) -> Qwen35ParoNextTokenRunner:
|
|
172
|
+
if self._runner is None:
|
|
173
|
+
self._runner = Qwen35ParoNextTokenRunner(
|
|
174
|
+
self.model_path,
|
|
175
|
+
index=self.weight_index,
|
|
176
|
+
backend=self.backend,
|
|
177
|
+
)
|
|
178
|
+
return self._runner
|
|
179
|
+
|
|
180
|
+
def _get_session(
|
|
181
|
+
self,
|
|
182
|
+
runner: Qwen35ParoNextTokenRunner,
|
|
183
|
+
*,
|
|
184
|
+
max_sequence_length: int,
|
|
185
|
+
kv_policy,
|
|
186
|
+
) -> Qwen35ParoResidentSession:
|
|
187
|
+
kv_key = (
|
|
188
|
+
kv_policy.storage_dtype.value,
|
|
189
|
+
kv_policy.scale_dtype.value,
|
|
190
|
+
kv_policy.scale_granularity,
|
|
191
|
+
int(kv_policy.block_size),
|
|
192
|
+
)
|
|
193
|
+
if (
|
|
194
|
+
self._session is None
|
|
195
|
+
or self._session_capacity < max_sequence_length
|
|
196
|
+
or self._session_kv_key != kv_key
|
|
197
|
+
):
|
|
198
|
+
self.close()
|
|
199
|
+
self._session = Qwen35ParoResidentSession(
|
|
200
|
+
runner,
|
|
201
|
+
max_sequence_length=max_sequence_length,
|
|
202
|
+
kv_policy=kv_policy.create_policy(),
|
|
203
|
+
kv_scale_dtype=kv_policy.scale_dtype,
|
|
204
|
+
kv_scale_granularity=kv_policy.scale_granularity,
|
|
205
|
+
)
|
|
206
|
+
self._session_capacity = max_sequence_length
|
|
207
|
+
self._session_kv_key = kv_key
|
|
208
|
+
else:
|
|
209
|
+
self._session.reset()
|
|
210
|
+
return self._session
|
|
211
|
+
|
|
212
|
+
def close(self) -> None:
|
|
213
|
+
if self._session is not None:
|
|
214
|
+
self._session.close()
|
|
215
|
+
self._session = None
|
|
216
|
+
self._session_capacity = 0
|
|
217
|
+
self._session_kv_key = None
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _session_capacity_for(required_sequence_length: int) -> int:
|
|
221
|
+
"""Return a reusable session capacity for a request.
|
|
222
|
+
|
|
223
|
+
Chat prompts grow after every turn, so allocating exactly the current
|
|
224
|
+
prompt+decode length forces resident weights/KV buffers to be torn down and
|
|
225
|
+
rebuilt on each request. Keep a modest floor and bucket growth to preserve
|
|
226
|
+
the resident session across normal local chat turns while still allowing
|
|
227
|
+
larger explicit contexts to expand on demand.
|
|
228
|
+
"""
|
|
229
|
+
|
|
230
|
+
required = int(required_sequence_length)
|
|
231
|
+
if required <= 0:
|
|
232
|
+
raise ValueError("required_sequence_length must be positive")
|
|
233
|
+
floor = max(1, _env_int("HIPENGINE_SESSION_MIN_TOKENS", 4096))
|
|
234
|
+
bucket = max(1, _env_int("HIPENGINE_SESSION_BUCKET_TOKENS", 1024))
|
|
235
|
+
capacity = max(required, floor)
|
|
236
|
+
return ((capacity + bucket - 1) // bucket) * bucket
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _env_int(name: str, default: int) -> int:
|
|
240
|
+
raw = os.environ.get(name)
|
|
241
|
+
if raw is None or raw == "":
|
|
242
|
+
return default
|
|
243
|
+
try:
|
|
244
|
+
return int(raw)
|
|
245
|
+
except ValueError:
|
|
246
|
+
return default
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _is_eos(tokenizer: Any | None, token_id: int) -> bool:
|
|
250
|
+
if tokenizer is None:
|
|
251
|
+
return False
|
|
252
|
+
try:
|
|
253
|
+
eos_id = getattr(tokenizer, "token_to_id")("<|endoftext|>")
|
|
254
|
+
except Exception:
|
|
255
|
+
eos_id = None
|
|
256
|
+
return eos_id is not None and int(token_id) == int(eos_id)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def make_qwen35_paro_one_token_generator(
|
|
260
|
+
*,
|
|
261
|
+
model_path: str | Path,
|
|
262
|
+
weight_index: WeightIndex,
|
|
263
|
+
model_plugin: Any,
|
|
264
|
+
) -> Qwen35ParoOneTokenGenerator:
|
|
265
|
+
return Qwen35ParoOneTokenGenerator(
|
|
266
|
+
model_path=model_path,
|
|
267
|
+
weight_index=weight_index,
|
|
268
|
+
model_plugin=model_plugin,
|
|
269
|
+
backend="hip_gfx1100",
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def make_qwen35_paro_one_token_generator_gfx1151(
|
|
274
|
+
*,
|
|
275
|
+
model_path: str | Path,
|
|
276
|
+
weight_index: WeightIndex,
|
|
277
|
+
model_plugin: Any,
|
|
278
|
+
) -> Qwen35ParoOneTokenGenerator:
|
|
279
|
+
return Qwen35ParoOneTokenGenerator(
|
|
280
|
+
model_path=model_path,
|
|
281
|
+
weight_index=weight_index,
|
|
282
|
+
model_plugin=model_plugin,
|
|
283
|
+
backend="hip_gfx1151",
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
register_text_generator(
|
|
288
|
+
model="qwen3_5_moe_paro",
|
|
289
|
+
backend="hip_gfx1100",
|
|
290
|
+
quant="w4_paro",
|
|
291
|
+
factory=make_qwen35_paro_one_token_generator,
|
|
292
|
+
)
|
|
293
|
+
register_text_generator(
|
|
294
|
+
model="qwen3_5_moe_paro",
|
|
295
|
+
backend="hip_gfx1151",
|
|
296
|
+
quant="w4_paro",
|
|
297
|
+
factory=make_qwen35_paro_one_token_generator_gfx1151,
|
|
298
|
+
)
|
|
@@ -6,7 +6,7 @@ through a registry at call time so backend/quant choices do not become engine br
|
|
|
6
6
|
|
|
7
7
|
from __future__ import annotations
|
|
8
8
|
|
|
9
|
-
from collections.abc import Iterable
|
|
9
|
+
from collections.abc import Iterable, Iterator
|
|
10
10
|
from dataclasses import dataclass
|
|
11
11
|
from pathlib import Path
|
|
12
12
|
from typing import Any
|
|
@@ -41,6 +41,7 @@ class LLM:
|
|
|
41
41
|
self._resolved_backend: str | None = None
|
|
42
42
|
self._weight_index: Any | None = None
|
|
43
43
|
self._model_plugin: Any | None = None
|
|
44
|
+
self._text_generator: Any | None = None
|
|
44
45
|
|
|
45
46
|
def generate(
|
|
46
47
|
self,
|
|
@@ -50,13 +51,31 @@ class LLM:
|
|
|
50
51
|
prompt_tuple = _normalize_prompts(prompts)
|
|
51
52
|
if not prompt_tuple:
|
|
52
53
|
return []
|
|
53
|
-
|
|
54
|
+
generator = self._get_text_generator()
|
|
55
|
+
request = _generation_request(prompt_tuple, sampling_params or SamplingParams())
|
|
56
|
+
return generator.generate(request)
|
|
54
57
|
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
58
|
+
def stream(
|
|
59
|
+
self,
|
|
60
|
+
prompt: str,
|
|
61
|
+
sampling_params: SamplingParams | None = None,
|
|
62
|
+
) -> Iterator[str]:
|
|
63
|
+
"""Yield generated text chunks for a single prompt when supported."""
|
|
64
|
+
|
|
65
|
+
generator = self._get_text_generator()
|
|
66
|
+
request = _generation_request((str(prompt),), sampling_params or SamplingParams())
|
|
67
|
+
streamer = getattr(generator, "stream", None)
|
|
68
|
+
if callable(streamer):
|
|
69
|
+
yield from streamer(request)
|
|
70
|
+
return
|
|
71
|
+
for text in generator.generate(request):
|
|
72
|
+
yield text
|
|
73
|
+
|
|
74
|
+
def _get_text_generator(self) -> Any:
|
|
75
|
+
if self._text_generator is not None:
|
|
76
|
+
return self._text_generator
|
|
77
|
+
|
|
78
|
+
from hipengine.generation import register_builtin_generators, resolve_text_generator
|
|
60
79
|
|
|
61
80
|
register_builtin_generators()
|
|
62
81
|
weight_index, model_plugin = self._load_model_metadata()
|
|
@@ -66,23 +85,12 @@ class LLM:
|
|
|
66
85
|
backend=backend,
|
|
67
86
|
quant=self.quant,
|
|
68
87
|
)
|
|
69
|
-
|
|
88
|
+
self._text_generator = factory(
|
|
70
89
|
model_path=self.model,
|
|
71
90
|
weight_index=weight_index,
|
|
72
91
|
model_plugin=model_plugin,
|
|
73
92
|
)
|
|
74
|
-
return
|
|
75
|
-
GenerationRequest(
|
|
76
|
-
prompts=prompt_tuple,
|
|
77
|
-
max_tokens=params.max_tokens,
|
|
78
|
-
temperature=params.temperature,
|
|
79
|
-
top_p=params.top_p,
|
|
80
|
-
ignore_eos=params.ignore_eos,
|
|
81
|
-
kv_storage=params.kv_storage,
|
|
82
|
-
kv_scale_dtype=params.kv_scale_dtype,
|
|
83
|
-
kv_scale_granularity=params.kv_scale_granularity,
|
|
84
|
-
)
|
|
85
|
-
)
|
|
93
|
+
return self._text_generator
|
|
86
94
|
|
|
87
95
|
def _resolve_backend(self) -> str:
|
|
88
96
|
if self._resolved_backend is not None:
|
|
@@ -97,10 +105,10 @@ class LLM:
|
|
|
97
105
|
if self._weight_index is not None and self._model_plugin is not None:
|
|
98
106
|
return self._weight_index, self._model_plugin
|
|
99
107
|
|
|
100
|
-
from hipengine.loading import discover_gguf_files, load_gguf_index, load_weight_index
|
|
108
|
+
from hipengine.loading import discover_gguf_files, load_gguf_index, load_weight_index, resolve_model_path
|
|
101
109
|
from hipengine.models import resolve_model
|
|
102
110
|
|
|
103
|
-
model_path =
|
|
111
|
+
model_path = resolve_model_path(self.model)
|
|
104
112
|
if _looks_like_gguf_path(model_path):
|
|
105
113
|
index = load_gguf_index(discover_gguf_files(model_path)[0])
|
|
106
114
|
self.model = str(index.path)
|
|
@@ -116,6 +124,21 @@ class LLM:
|
|
|
116
124
|
return index, plugin
|
|
117
125
|
|
|
118
126
|
|
|
127
|
+
def _generation_request(prompt_tuple: tuple[str, ...], params: SamplingParams):
|
|
128
|
+
from hipengine.generation import GenerationRequest
|
|
129
|
+
|
|
130
|
+
return GenerationRequest(
|
|
131
|
+
prompts=prompt_tuple,
|
|
132
|
+
max_tokens=params.max_tokens,
|
|
133
|
+
temperature=params.temperature,
|
|
134
|
+
top_p=params.top_p,
|
|
135
|
+
ignore_eos=params.ignore_eos,
|
|
136
|
+
kv_storage=params.kv_storage,
|
|
137
|
+
kv_scale_dtype=params.kv_scale_dtype,
|
|
138
|
+
kv_scale_granularity=params.kv_scale_granularity,
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
|
|
119
142
|
def _looks_like_gguf_path(path: Path) -> bool:
|
|
120
143
|
if path.is_file():
|
|
121
144
|
return path.suffix.lower() == ".gguf"
|