gputex 0.3.3 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -67
- package/dist/index.d.ts +16 -38
- package/dist/index.js +589 -350
- package/dist/three.d.ts +3 -15
- package/dist/three.js +595 -360
- package/package.json +1 -1
package/dist/three.js
CHANGED
|
@@ -99,17 +99,29 @@ var Encoder = class {
|
|
|
99
99
|
adapter;
|
|
100
100
|
ownsDevice;
|
|
101
101
|
disableF16;
|
|
102
|
-
//
|
|
103
|
-
//
|
|
104
|
-
|
|
105
|
-
// f16 'fast' module — built only when the device supports shader-f16 and the
|
|
106
|
-
// subclass provides an f16 source. null otherwise (falls back to _module).
|
|
107
|
-
_moduleF16 = null;
|
|
108
|
-
_pipelineF16 = null;
|
|
109
|
-
// Default pipeline (fast). Kept as a field for back-compat; the per-quality
|
|
110
|
-
// cache below holds the specialised pipelines for encoders that support it.
|
|
102
|
+
// The single compute pipeline: built from the f16 module when the device
|
|
103
|
+
// supports shader-f16 and the subclass provides an f16 source, from the
|
|
104
|
+
// f32 module otherwise. Both implement the same algorithm.
|
|
111
105
|
_pipeline;
|
|
112
|
-
|
|
106
|
+
// -------------------------------------------------------------------- //
|
|
107
|
+
// Per-encoder GPU resource cache. Creating the source texture, output/
|
|
108
|
+
// staging buffers and bind group on every encode costs ~1ms of host time
|
|
109
|
+
// per call — for small/medium images that overhead dominates the encode
|
|
110
|
+
// (the compute pass itself is tens of µs at 512²). Sequential encodes
|
|
111
|
+
// (the common case: one texture after another, or a mip chain) reuse
|
|
112
|
+
// these; concurrent encodes on the same encoder see `_resourcesBusy` and
|
|
113
|
+
// fall back to transient resources, keeping the API contract unchanged.
|
|
114
|
+
// Buffers are grow-only, the texture is recreated on size change, and the
|
|
115
|
+
// bind group is kept until any bound resource is recreated.
|
|
116
|
+
_cachedSrcTex = null;
|
|
117
|
+
_cachedSrcW = 0;
|
|
118
|
+
_cachedSrcH = 0;
|
|
119
|
+
_cachedDst = null;
|
|
120
|
+
_cachedStaging = null;
|
|
121
|
+
_cachedParams = null;
|
|
122
|
+
_lastParams = null;
|
|
123
|
+
_cachedBindGroup = null;
|
|
124
|
+
_resourcesBusy = false;
|
|
113
125
|
constructor({ device, adapter, ownsDevice = false, disableF16 = false }) {
|
|
114
126
|
this.device = device;
|
|
115
127
|
this.adapter = adapter;
|
|
@@ -119,59 +131,27 @@ var Encoder = class {
|
|
|
119
131
|
}
|
|
120
132
|
_buildPipeline() {
|
|
121
133
|
const device = this.device;
|
|
122
|
-
const
|
|
123
|
-
|
|
124
|
-
label: `${this.label}-encoder`,
|
|
125
|
-
code
|
|
134
|
+
const useF16 = this._useF16;
|
|
135
|
+
const module = device.createShaderModule({
|
|
136
|
+
label: `${this.label}-encoder${useF16 ? "-f16" : ""}`,
|
|
137
|
+
code: useF16 ? this.wgslSourceFastF16() : this.wgslSource()
|
|
126
138
|
});
|
|
127
|
-
|
|
128
|
-
this.
|
|
129
|
-
label: `${this.label}-encoder-f16`,
|
|
130
|
-
code: this.wgslSourceFastF16()
|
|
131
|
-
});
|
|
132
|
-
}
|
|
133
|
-
if (this.supportsQuality) {
|
|
134
|
-
this._pipeline = this._getPipeline("fast");
|
|
135
|
-
} else {
|
|
136
|
-
this._pipeline = device.createComputePipeline({
|
|
137
|
-
label: `${this.label}-encoder-pipeline`,
|
|
138
|
-
layout: "auto",
|
|
139
|
-
compute: { module: this._module, entryPoint: "encode" }
|
|
140
|
-
});
|
|
141
|
-
}
|
|
142
|
-
}
|
|
143
|
-
/**
|
|
144
|
-
* Pipeline for a given quality level. Encoders that don't declare a
|
|
145
|
-
* `QUALITY_HIGH` override (`supportsQuality === false`, e.g. BC1) ignore the
|
|
146
|
-
* argument and reuse the single pipeline. Specialised pipelines are cached.
|
|
147
|
-
*/
|
|
148
|
-
_getPipeline(quality) {
|
|
149
|
-
if (!this.supportsQuality) return this._pipeline;
|
|
150
|
-
if (quality === "fast" && this._moduleF16) {
|
|
151
|
-
if (!this._pipelineF16) {
|
|
152
|
-
this._pipelineF16 = this.device.createComputePipeline({
|
|
153
|
-
label: `${this.label}-encoder-pipeline-fast-f16`,
|
|
154
|
-
layout: "auto",
|
|
155
|
-
compute: { module: this._moduleF16, entryPoint: "encode" }
|
|
156
|
-
});
|
|
157
|
-
}
|
|
158
|
-
return this._pipelineF16;
|
|
159
|
-
}
|
|
160
|
-
const cached = this._pipelineCache.get(quality);
|
|
161
|
-
if (cached) return cached;
|
|
162
|
-
const pipeline = this.device.createComputePipeline({
|
|
163
|
-
label: `${this.label}-encoder-pipeline-${quality}`,
|
|
139
|
+
this._pipeline = device.createComputePipeline({
|
|
140
|
+
label: `${this.label}-encoder-pipeline${useF16 ? "-f16" : ""}`,
|
|
164
141
|
layout: "auto",
|
|
165
|
-
compute: {
|
|
166
|
-
module: this._module,
|
|
167
|
-
entryPoint: "encode",
|
|
168
|
-
constants: { QUALITY_HIGH: quality === "high" ? 1 : 0 }
|
|
169
|
-
}
|
|
142
|
+
compute: { module, entryPoint: "encode" }
|
|
170
143
|
});
|
|
171
|
-
this._pipelineCache.set(quality, pipeline);
|
|
172
|
-
return pipeline;
|
|
173
144
|
}
|
|
174
145
|
destroy() {
|
|
146
|
+
this._cachedSrcTex?.destroy();
|
|
147
|
+
this._cachedDst?.destroy();
|
|
148
|
+
this._cachedStaging?.destroy();
|
|
149
|
+
this._cachedParams?.destroy();
|
|
150
|
+
this._cachedSrcTex = null;
|
|
151
|
+
this._cachedDst = null;
|
|
152
|
+
this._cachedStaging = null;
|
|
153
|
+
this._cachedParams = null;
|
|
154
|
+
this._cachedBindGroup = null;
|
|
175
155
|
if (this.ownsDevice) this.device.destroy();
|
|
176
156
|
}
|
|
177
157
|
/** WGSL `@workgroup_size` dimensions. Default 8×8×1. */
|
|
@@ -183,22 +163,14 @@ var Encoder = class {
|
|
|
183
163
|
return true;
|
|
184
164
|
}
|
|
185
165
|
/**
|
|
186
|
-
*
|
|
187
|
-
*
|
|
188
|
-
*
|
|
189
|
-
*/
|
|
190
|
-
get supportsQuality() {
|
|
191
|
-
return false;
|
|
192
|
-
}
|
|
193
|
-
/**
|
|
194
|
-
* Optional f16 WGSL for the 'fast' path. Used only when the device reports the
|
|
195
|
-
* `shader-f16` feature; the format's f32 `wgslSource()` is the fallback and
|
|
196
|
-
* `'high'` always uses it. Returns null when there's no f16 variant.
|
|
166
|
+
* Optional f16 WGSL variant. Used only when the device reports the
|
|
167
|
+
* `shader-f16` feature; the format's f32 `wgslSource()` is the automatic
|
|
168
|
+
* fallback. Returns null when there's no f16 variant.
|
|
197
169
|
*/
|
|
198
170
|
wgslSourceFastF16() {
|
|
199
171
|
return null;
|
|
200
172
|
}
|
|
201
|
-
/** Whether the f16
|
|
173
|
+
/** Whether the f16 shader is both available and supported on this device. */
|
|
202
174
|
get _useF16() {
|
|
203
175
|
return !this.disableF16 && this.wgslSourceFastF16() !== null && this.device.features.has("shader-f16");
|
|
204
176
|
}
|
|
@@ -221,11 +193,7 @@ var Encoder = class {
|
|
|
221
193
|
* bytes into a `CompressedTexture`; callers targeting another engine feed
|
|
222
194
|
* `data` into that engine's compressed-texture upload directly.
|
|
223
195
|
*/
|
|
224
|
-
async encodeToBytes(source, {
|
|
225
|
-
flipY = false,
|
|
226
|
-
quality = "fast",
|
|
227
|
-
withGpuTime = false
|
|
228
|
-
} = {}) {
|
|
196
|
+
async encodeToBytes(source, { flipY = false, withGpuTime = false } = {}) {
|
|
229
197
|
const device = this.device;
|
|
230
198
|
const width = source.width;
|
|
231
199
|
const height = source.height;
|
|
@@ -238,93 +206,158 @@ var Encoder = class {
|
|
|
238
206
|
const blocksY = paddedHeight >> 2;
|
|
239
207
|
const blockCount = blocksX * blocksY;
|
|
240
208
|
const outByteLen = blockCount * this.bytesPerBlock;
|
|
241
|
-
const
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
209
|
+
const useCache = !this._resourcesBusy;
|
|
210
|
+
if (useCache) this._resourcesBusy = true;
|
|
211
|
+
let srcTex;
|
|
212
|
+
let dstBuffer;
|
|
213
|
+
let paramsBuffer;
|
|
214
|
+
let staging;
|
|
215
|
+
try {
|
|
216
|
+
let srcTexIsNew = true;
|
|
217
|
+
if (useCache && this._cachedSrcTex && this._cachedSrcW === paddedWidth && this._cachedSrcH === paddedHeight) {
|
|
218
|
+
srcTex = this._cachedSrcTex;
|
|
219
|
+
srcTexIsNew = false;
|
|
220
|
+
} else {
|
|
221
|
+
srcTex = device.createTexture({
|
|
222
|
+
label: `${this.label}-src`,
|
|
223
|
+
size: [paddedWidth, paddedHeight, 1],
|
|
224
|
+
format: "rgba8unorm",
|
|
225
|
+
// RENDER_ATTACHMENT is required by copyExternalImageToTexture
|
|
226
|
+
// (internally a blit) even though we never render into this texture.
|
|
227
|
+
usage: GPUTextureUsage.COPY_DST | GPUTextureUsage.TEXTURE_BINDING | GPUTextureUsage.RENDER_ATTACHMENT
|
|
228
|
+
});
|
|
229
|
+
if (useCache) {
|
|
230
|
+
this._cachedSrcTex?.destroy();
|
|
231
|
+
this._cachedSrcTex = srcTex;
|
|
232
|
+
this._cachedSrcW = paddedWidth;
|
|
233
|
+
this._cachedSrcH = paddedHeight;
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
uploadSourceTexture(device, srcTex, source, width, height, flipY, source instanceof ImageData);
|
|
237
|
+
let dstIsNew = true;
|
|
238
|
+
if (useCache && this._cachedDst && this._cachedDst.size >= outByteLen) {
|
|
239
|
+
dstBuffer = this._cachedDst;
|
|
240
|
+
dstIsNew = false;
|
|
241
|
+
} else {
|
|
242
|
+
dstBuffer = device.createBuffer({
|
|
243
|
+
label: `${this.label}-dst`,
|
|
244
|
+
size: outByteLen,
|
|
245
|
+
usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_SRC
|
|
246
|
+
});
|
|
247
|
+
if (useCache) {
|
|
248
|
+
this._cachedDst?.destroy();
|
|
249
|
+
this._cachedDst = dstBuffer;
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
if (useCache && this._cachedStaging && this._cachedStaging.size >= outByteLen) {
|
|
253
|
+
staging = this._cachedStaging;
|
|
254
|
+
} else {
|
|
255
|
+
staging = device.createBuffer({
|
|
256
|
+
label: `${this.label}-staging`,
|
|
257
|
+
size: outByteLen,
|
|
258
|
+
usage: GPUBufferUsage.COPY_DST | GPUBufferUsage.MAP_READ
|
|
259
|
+
});
|
|
260
|
+
if (useCache) {
|
|
261
|
+
this._cachedStaging?.destroy();
|
|
262
|
+
this._cachedStaging = staging;
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
if (useCache) {
|
|
266
|
+
if (!this._cachedParams) {
|
|
267
|
+
this._cachedParams = device.createBuffer({
|
|
268
|
+
label: `${this.label}-params`,
|
|
269
|
+
size: 16,
|
|
270
|
+
usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST
|
|
271
|
+
});
|
|
272
|
+
this._lastParams = null;
|
|
273
|
+
}
|
|
274
|
+
paramsBuffer = this._cachedParams;
|
|
275
|
+
const lp = this._lastParams;
|
|
276
|
+
if (!lp || lp[0] !== blocksX || lp[1] !== blocksY || lp[2] !== width || lp[3] !== height) {
|
|
277
|
+
device.queue.writeBuffer(paramsBuffer, 0, new Uint32Array([blocksX, blocksY, width, height]));
|
|
278
|
+
this._lastParams = [blocksX, blocksY, width, height];
|
|
279
|
+
}
|
|
280
|
+
} else {
|
|
281
|
+
paramsBuffer = device.createBuffer({
|
|
282
|
+
label: `${this.label}-params`,
|
|
283
|
+
size: 16,
|
|
284
|
+
usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST
|
|
285
|
+
});
|
|
286
|
+
device.queue.writeBuffer(paramsBuffer, 0, new Uint32Array([blocksX, blocksY, width, height]));
|
|
287
|
+
}
|
|
288
|
+
const pipeline = this._pipeline;
|
|
289
|
+
if (useCache && (srcTexIsNew || dstIsNew)) this._cachedBindGroup = null;
|
|
290
|
+
let bindGroup = useCache ? this._cachedBindGroup : null;
|
|
291
|
+
if (!bindGroup) {
|
|
292
|
+
bindGroup = device.createBindGroup({
|
|
293
|
+
label: `${this.label}-bg`,
|
|
294
|
+
layout: pipeline.getBindGroupLayout(0),
|
|
295
|
+
entries: [
|
|
296
|
+
{ binding: 0, resource: srcTex.createView() },
|
|
297
|
+
{ binding: 1, resource: { buffer: dstBuffer } },
|
|
298
|
+
{ binding: 2, resource: { buffer: paramsBuffer } }
|
|
299
|
+
]
|
|
300
|
+
});
|
|
301
|
+
if (useCache) this._cachedBindGroup = bindGroup;
|
|
302
|
+
}
|
|
303
|
+
const useTimestamps = withGpuTime && device.features.has("timestamp-query");
|
|
304
|
+
const querySet = useTimestamps ? device.createQuerySet({ type: "timestamp", count: 2 }) : null;
|
|
305
|
+
const queryBuffer = useTimestamps ? device.createBuffer({
|
|
306
|
+
label: `${this.label}-ts-resolve`,
|
|
307
|
+
size: 16,
|
|
308
|
+
usage: GPUBufferUsage.QUERY_RESOLVE | GPUBufferUsage.COPY_SRC
|
|
309
|
+
}) : null;
|
|
310
|
+
const [wgX, wgY] = this.workgroupSize;
|
|
311
|
+
const t0 = performance.now();
|
|
312
|
+
const enc = device.createCommandEncoder({ label: `${this.label}-encode` });
|
|
313
|
+
const pass = enc.beginComputePass(
|
|
314
|
+
querySet ? { timestampWrites: { querySet, beginningOfPassWriteIndex: 0, endOfPassWriteIndex: 1 } } : void 0
|
|
315
|
+
);
|
|
316
|
+
pass.setPipeline(pipeline);
|
|
317
|
+
pass.setBindGroup(0, bindGroup);
|
|
318
|
+
pass.dispatchWorkgroups(Math.ceil(blocksX / wgX), Math.ceil(blocksY / wgY), 1);
|
|
319
|
+
pass.end();
|
|
320
|
+
if (querySet && queryBuffer) enc.resolveQuerySet(querySet, 0, 2, queryBuffer, 0);
|
|
321
|
+
enc.copyBufferToBuffer(dstBuffer, 0, staging, 0, outByteLen);
|
|
322
|
+
const tsStaging = querySet && queryBuffer ? device.createBuffer({
|
|
323
|
+
label: `${this.label}-ts-staging`,
|
|
324
|
+
size: 16,
|
|
325
|
+
usage: GPUBufferUsage.COPY_DST | GPUBufferUsage.MAP_READ
|
|
326
|
+
}) : null;
|
|
327
|
+
if (tsStaging && queryBuffer) enc.copyBufferToBuffer(queryBuffer, 0, tsStaging, 0, 16);
|
|
328
|
+
device.queue.submit([enc.finish()]);
|
|
329
|
+
await staging.mapAsync(GPUMapMode.READ, 0, outByteLen);
|
|
330
|
+
const data = new Uint8Array(staging.getMappedRange(0, outByteLen).slice(0));
|
|
331
|
+
staging.unmap();
|
|
332
|
+
const encodeMs = performance.now() - t0;
|
|
333
|
+
let gpuMs;
|
|
334
|
+
if (tsStaging) {
|
|
335
|
+
await tsStaging.mapAsync(GPUMapMode.READ);
|
|
336
|
+
const [begin, end] = new BigUint64Array(tsStaging.getMappedRange().slice(0));
|
|
337
|
+
tsStaging.unmap();
|
|
338
|
+
tsStaging.destroy();
|
|
339
|
+
if (end !== void 0 && begin !== void 0 && end > begin) {
|
|
340
|
+
gpuMs = Number(end - begin) / 1e6;
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
querySet?.destroy();
|
|
344
|
+
queryBuffer?.destroy();
|
|
345
|
+
return { width, height, paddedWidth, paddedHeight, data, encodeMs, gpuMs };
|
|
346
|
+
} finally {
|
|
347
|
+
if (useCache) {
|
|
348
|
+
this._resourcesBusy = false;
|
|
349
|
+
} else {
|
|
350
|
+
srcTex?.destroy();
|
|
351
|
+
dstBuffer?.destroy();
|
|
352
|
+
staging?.destroy();
|
|
353
|
+
paramsBuffer?.destroy();
|
|
314
354
|
}
|
|
315
355
|
}
|
|
316
|
-
querySet?.destroy();
|
|
317
|
-
queryBuffer?.destroy();
|
|
318
|
-
srcTex.destroy();
|
|
319
|
-
dstBuffer.destroy();
|
|
320
|
-
staging.destroy();
|
|
321
|
-
paramsBuffer.destroy();
|
|
322
|
-
return { width, height, paddedWidth, paddedHeight, data, encodeMs, gpuMs };
|
|
323
356
|
}
|
|
324
357
|
};
|
|
325
358
|
|
|
326
359
|
// src/bc1.wgsl
|
|
327
|
-
var bc1_default = "// BC1 (DXT1) compute shader encoder.\n//\n// Each invocation encodes one 4x4 pixel block into an 8-byte BC1 block\n// written as 2 x u32 into the destination storage buffer.\n//\n// BC1 block layout (little-endian):\n// u32[0]: color0 (low 16) | color1 (high 16) both in RGB565\n// u32[1]: 16 x 2-bit indices, pixel 0 = bits 0..1, pixel 15 = bits 30..31\n//\n// We always force the 4-color mode (color0 > color1, numeric 16-bit):\n// idx 0 -> color0\n// idx 1 -> color1\n// idx 2 -> (2*color0 + color1) / 3\n// idx 3 -> ( color0 + 2*color1) / 3\n//\n// QUALITY LEVELS (pipeline-overridable constant `QUALITY_HIGH`)\n// fast (0, default): bounding-box endpoints inset by ~half a 565 cell, then\n// ONE fused pass that projects every pixel onto the decoded-endpoint line\n// (the 4 palette entries are colinear and evenly spaced, so the nearest\n// entry is the rounded projection \u2014 no 4-entry search) while accumulating\n// the least-squares refit sums; the refit endpoints are re-quantised and a\n// final projection pass assigns the indices, packed on the fly.\n// high (1): endpoints are seeded from the block's principal colour axis\n// (covariance power-iteration) as well as the bbox diagonal, each refined by\n// several least-squares passes with full 4-entry searches; the lower-error\n// family wins. Mirrors bc1_ref.ts.\n\n// 0 = fast (default), 1 = high-quality. Set via pipeline constants.\noverride QUALITY_HIGH: u32 = 0u;\n\nstruct Params {\n blocks_x: u32,\n blocks_y: u32,\n width: u32,\n height: u32,\n};\n\n@group(0) @binding(0) var src_tex: texture_2d<f32>;\n@group(0) @binding(1) var<storage, read_write> dst: array<u32>;\n@group(0) @binding(2) var<uniform> params: Params;\n\nfn to565(c: vec3<f32>) -> u32 {\n // Round-to-nearest quantization into 5-6-5.\n let r = u32(clamp(floor(c.r * 31.0 + 0.5), 0.0, 31.0));\n let g = u32(clamp(floor(c.g * 63.0 + 0.5), 0.0, 63.0));\n let b = u32(clamp(floor(c.b * 31.0 + 0.5), 0.0, 31.0));\n return (r << 11u) | (g << 5u) | b;\n}\n\nfn from565(c: u32) -> vec3<f32> {\n let r = (c >> 11u) & 31u;\n let g = (c >> 5u) & 63u;\n let b = c & 31u;\n // 5/6-bit -> 8-bit: (x*527+23)>>6 (6-bit: 259/33) \u2014 round-to-nearest\n // scaling, matching bc1_ref.ts and typical hardware decoders (white ->\n // 255). Integer u32 math is exact. NOTE: this is NOT plain bit-replication\n // ((x<<3)|(x>>2)) \u2014 they differ for some codes (e.g. 5-bit 3 -> 25 vs 24).\n // Selecting indices against this palette is what makes the encoder agree\n // with what the GPU will actually sample.\n let r8 = (r * 527u + 23u) >> 6u;\n let g8 = (g * 259u + 33u) >> 6u;\n let b8 = (b * 527u + 23u) >> 6u;\n return vec3<f32>(vec3<u32>(r8, g8, b8)) / 255.0;\n}\n\n// 4-color-mode interpolation weights: palette[j] = wa(j)*c0 + wb(j)*c1.\nfn wa(j: u32) -> f32 {\n switch j {\n case 0u: { return 1.0; }\n case 1u: { return 0.0; }\n case 2u: { return 2.0 / 3.0; }\n default: { return 1.0 / 3.0; } // case 3u\n }\n}\nfn wb(j: u32) -> f32 {\n switch j {\n case 0u: { return 0.0; }\n case 1u: { return 1.0; }\n case 2u: { return 1.0 / 3.0; }\n default: { return 2.0 / 3.0; } // case 3u\n }\n}\n\nfn build_palette(c0: u32, c1: u32, pal: ptr<function, array<vec3<f32>, 4>>) {\n let p0 = from565(c0);\n let p1 = from565(c1);\n for (var j: u32 = 0u; j < 4u; j = j + 1u) {\n (*pal)[j] = wa(j) * p0 + wb(j) * p1;\n }\n}\n\n// Assign each of the 16 pixels its nearest palette entry (full 4-entry L2),\n// writing indices into `out_idx` and returning the total squared error.\nfn assign_indices(\n pixels: ptr<function, array<vec3<f32>, 16>>,\n pal: ptr<function, array<vec3<f32>, 4>>,\n out_idx: ptr<function, array<u32, 16>>,\n) -> f32 {\n var err: f32 = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let c = (*pixels)[k];\n var best_j: u32 = 0u;\n var best_d: f32 = 1e30;\n for (var j: u32 = 0u; j < 4u; j = j + 1u) {\n let d = (*pal)[j] - c;\n let d2 = dot(d, d);\n if (d2 < best_d) {\n best_d = d2;\n best_j = j;\n }\n }\n (*out_idx)[k] = best_j;\n err = err + best_d;\n }\n return err;\n}\n\n// One least-squares refit pass: solve the 2x2 normal equations for the endpoint\n// colours that minimise \u03A3\u2016wa\xB7e0 + wb\xB7e1 \u2212 c\u2016\xB2 under the current indices. The\n// three channels share the scalar sums, so it's one 2x2 solve with vec3 RHS.\nstruct RefitResult { e0: vec3<f32>, e1: vec3<f32>, valid: bool };\nfn refit(\n pixels: ptr<function, array<vec3<f32>, 16>>,\n indices: ptr<function, array<u32, 16>>,\n) -> RefitResult {\n var sAA: f32 = 0.0; var sBB: f32 = 0.0; var sAB: f32 = 0.0;\n var sAV: vec3<f32> = vec3<f32>(0.0);\n var sBV: vec3<f32> = vec3<f32>(0.0);\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let a = wa((*indices)[k]);\n let b = wb((*indices)[k]);\n let v = (*pixels)[k];\n sAA = sAA + a * a;\n sBB = sBB + b * b;\n sAB = sAB + a * b;\n sAV = sAV + a * v;\n sBV = sBV + b * v;\n }\n var out: RefitResult;\n let det = sAA * sBB - sAB * sAB;\n if (abs(det) < 1e-9) {\n out.valid = false;\n return out;\n }\n out.e0 = clamp((sBB * sAV - sAB * sBV) / det, vec3<f32>(0.0), vec3<f32>(1.0));\n out.e1 = clamp((sAA * sBV - sAB * sAV) / det, vec3<f32>(0.0), vec3<f32>(1.0));\n out.valid = true;\n return out;\n}\n\n// Candidate solution tracked across endpoint seeds / refit passes.\nstruct Best { c0: u32, c1: u32, indices: array<u32, 16>, err: f32 };\n\n// Quantize (hi, lo) to 565, force 4-color mode, assign indices, then refine with\n// up to `max_refits` least-squares passes. Commits to `*best` only on strict\n// improvement.\nfn fit_from_endpoints(\n pixels: ptr<function, array<vec3<f32>, 16>>,\n hi: vec3<f32>,\n lo: vec3<f32>,\n max_refits: u32,\n best: ptr<function, Best>,\n) {\n var c0 = to565(hi);\n var c1 = to565(lo);\n // 4-color mode requires color0 > color1.\n if (c0 == c1) {\n if (c1 > 0u) { c1 = c1 - 1u; } else { c0 = c0 + 1u; }\n } else if (c0 < c1) {\n let t = c0; c0 = c1; c1 = t;\n }\n\n var pal: array<vec3<f32>, 4>;\n var idx: array<u32, 16>;\n build_palette(c0, c1, &pal);\n var err = assign_indices(pixels, &pal, &idx);\n if (err < (*best).err) {\n (*best).c0 = c0; (*best).c1 = c1; (*best).indices = idx; (*best).err = err;\n }\n\n for (var rp: u32 = 0u; rp < max_refits; rp = rp + 1u) {\n let r = refit(pixels, &idx);\n if (!r.valid) { break; }\n var nc0 = to565(r.e0);\n var nc1 = to565(r.e1);\n // A refit that flips/equalises the endpoints would change decode mode;\n // keep 4-color mode, and stop once it stops moving.\n if (nc0 < nc1) { let t = nc0; nc0 = nc1; nc1 = t; }\n if (nc0 == nc1) { break; }\n if (nc0 == c0 && nc1 == c1) { break; }\n build_palette(nc0, nc1, &pal);\n let nerr = assign_indices(pixels, &pal, &idx);\n c0 = nc0; c1 = nc1; err = nerr;\n if (nerr < (*best).err) {\n (*best).c0 = nc0; (*best).c1 = nc1; (*best).indices = idx; (*best).err = nerr;\n }\n }\n}\n\n// Principal colour axis via covariance power-iteration, seeded with the bbox\n// diagonal. Returns a unit axis, or vec3(0) for a degenerate (constant) block.\nfn principal_axis(\n pixels: ptr<function, array<vec3<f32>, 16>>,\n mean: vec3<f32>,\n seed: vec3<f32>,\n) -> vec3<f32> {\n // Symmetric 3x3 covariance, stored as its three rows.\n var c0v = vec3<f32>(0.0);\n var c1v = vec3<f32>(0.0);\n var c2v = vec3<f32>(0.0);\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let d = (*pixels)[k] - mean;\n c0v = c0v + d.x * d;\n c1v = c1v + d.y * d;\n c2v = c2v + d.z * d;\n }\n var v = seed;\n var len = length(v);\n if (len < 1e-9) { return vec3<f32>(0.0); }\n v = v / len;\n for (var iter: u32 = 0u; iter < 8u; iter = iter + 1u) {\n let nv = vec3<f32>(dot(c0v, v), dot(c1v, v), dot(c2v, v));\n len = length(nv);\n if (len < 1e-12) { return vec3<f32>(0.0); }\n v = nv / len;\n }\n return v;\n}\n\n@compute @workgroup_size(8, 8, 1)\nfn encode(@builtin(global_invocation_id) gid: vec3<u32>) {\n if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) {\n return;\n }\n\n let bx = gid.x;\n let by = gid.y;\n let block_index = by * params.blocks_x + bx;\n\n let base = vec2<i32>(i32(bx) * 4, i32(by) * 4);\n let max_xy = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);\n\n var pixels: array<vec3<f32>, 16>;\n var bb_min = vec3<f32>(1.0, 1.0, 1.0);\n var bb_max = vec3<f32>(0.0, 0.0, 0.0);\n var mean = vec3<f32>(0.0);\n\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let lx = i32(i & 3u);\n let ly = i32(i >> 2u);\n // Clamp to edge for non-multiple-of-4 textures.\n let p = clamp(base + vec2<i32>(lx, ly), vec2<i32>(0, 0), max_xy);\n let c = textureLoad(src_tex, p, 0).rgb;\n pixels[i] = c;\n bb_min = min(bb_min, c);\n bb_max = max(bb_max, c);\n mean = mean + c;\n }\n mean = mean * (1.0 / 16.0);\n\n // Inset the bounding box by ~half an RGB565 cell (1/16) so the quantized\n // 4-color palette covers the real data range more tightly (stb_dxt heuristic).\n let inset = (bb_max - bb_min) / 16.0;\n let bbox_hi = clamp(bb_max - inset, vec3<f32>(0.0), vec3<f32>(1.0));\n let bbox_lo = clamp(bb_min + inset, vec3<f32>(0.0), vec3<f32>(1.0));\n\n if (QUALITY_HIGH == 0u) {\n // -------- fast: projection + fused LSQ refit + reprojection --------\n var c0 = to565(bbox_hi);\n var c1 = to565(bbox_lo);\n if (c0 == c1) {\n if (c1 > 0u) { c1 = c1 - 1u; } else { c0 = c0 + 1u; }\n } else if (c0 < c1) {\n let t = c0; c0 = c1; c1 = t;\n }\n let p0 = from565(c0);\n let p1 = from565(c1);\n\n // Fused pass: projection assignment + LSQ sums + the seed solution's\n // packed indices and squared error. Level \u2192 BC1 index: 0\u21920 (c0), 1\u21922,\n // 2\u21923, 3\u21921 (c1); as a packed LUT: (0x78 >> 2L) & 3.\n var idx_bits: u32 = 0u;\n let dir = p1 - p0;\n let dd = dot(dir, dir);\n if (dd > 0.0) {\n let inv = 3.0 / dd;\n var sAA: f32 = 0.0; var sBB: f32 = 0.0; var sAB: f32 = 0.0;\n var sAV = vec3<f32>(0.0); var sBV = vec3<f32>(0.0);\n var s_min = 3.0; var s_max = 0.0;\n var seed_err: f32 = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let v = pixels[k];\n let s = clamp(floor(dot(v - p0, dir) * inv + 0.5), 0.0, 3.0);\n s_min = min(s_min, s); s_max = max(s_max, s);\n let b = s * (1.0 / 3.0); let a = 1.0 - b;\n sAA = sAA + a * a; sBB = sBB + b * b; sAB = sAB + a * b;\n sAV = sAV + a * v; sBV = sBV + b * v;\n let e = v - (p0 + b * dir);\n seed_err = seed_err + dot(e, e);\n idx_bits = idx_bits | (((0x78u >> (u32(s) * 2u)) & 3u) << (k * 2u));\n }\n let det = sAA * sBB - sAB * sAB;\n // Refit only on a well-conditioned system: when every pixel lands on\n // ONE level (flat blocks \u2014 the 4-colour nudge forces c0 \u2260 c1 even\n // then) the system is rank-1 and det/numerators are pure float noise;\n // the solve would return garbage endpoints. With \u22652 levels\n // det = \u03A3_i<j (b_j \u2212 b_i)\xB2 \u2265 ~1.67, so 1e-3 is a safe guard.\n if (s_min < s_max && abs(det) > 1e-3) {\n // Clamp the refit to the block bbox (not [0,1]): on multi-cluster\n // blocks the unconstrained solve extrapolates far outside the block's\n // colours and the per-channel clamp then bends the hue \u2014 fringe pixels\n // decode to colours that exist nowhere in the block. Constraining to\n // the bbox also measures better in plain SSE (+1.6 dB on the colour\n // test card), so the accept-if-better guard below keeps more refits.\n let e0 = clamp((sBB * sAV - sAB * sBV) / det, bb_min, bb_max);\n let e1 = clamp((sAA * sBV - sAB * sAV) / det, bb_min, bb_max);\n var nc0 = to565(e0);\n var nc1 = to565(e1);\n if (nc0 == nc1) {\n if (nc1 > 0u) { nc1 = nc1 - 1u; } else { nc0 = nc0 + 1u; }\n } else if (nc0 < nc1) {\n let t = nc0; nc0 = nc1; nc1 = t;\n }\n let np0 = from565(nc0);\n let np1 = from565(nc1);\n let ndir = np1 - np0;\n let ndd = dot(ndir, ndir);\n if (ndd > 0.0 && !(nc0 == c0 && nc1 == c1)) {\n // Reproject against the refit endpoints and accept them only if\n // the block error actually decreases (the refit minimises a\n // continuous objective; after 565 quantisation it can lose).\n let ninv = 3.0 / ndd;\n var refit_err: f32 = 0.0;\n var nidx_bits: u32 = 0u;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let v = pixels[k];\n let s = clamp(floor(dot(v - np0, ndir) * ninv + 0.5), 0.0, 3.0);\n let e = v - (np0 + s * (1.0 / 3.0) * ndir);\n refit_err = refit_err + dot(e, e);\n nidx_bits = nidx_bits | (((0x78u >> (u32(s) * 2u)) & 3u) << (k * 2u));\n }\n if (refit_err < seed_err) {\n c0 = nc0; c1 = nc1;\n idx_bits = nidx_bits;\n }\n }\n }\n }\n\n let out = block_index * 2u;\n dst[out] = c0 | (c1 << 16u);\n dst[out + 1u] = idx_bits;\n return;\n }\n\n // ------------------------------ high --------------------------------- //\n var best: Best;\n best.err = 1e30;\n\n // Seed from the principal colour axis: project all texels onto it, take the\n // extreme projections as endpoints, inset along the axis. Then also try the\n // bbox seed and keep whichever family yields the lower error.\n let axis = principal_axis(&pixels, mean, bb_max - bb_min);\n if (dot(axis, axis) > 0.0) {\n var t_min: f32 = 1e30;\n var t_max: f32 = -1e30;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let t = dot(pixels[k] - mean, axis);\n t_min = min(t_min, t);\n t_max = max(t_max, t);\n }\n let pad = (t_max - t_min) / 16.0;\n let pca_hi = clamp(mean + (t_max - pad) * axis, vec3<f32>(0.0), vec3<f32>(1.0));\n let pca_lo = clamp(mean + (t_min + pad) * axis, vec3<f32>(0.0), vec3<f32>(1.0));\n fit_from_endpoints(&pixels, pca_hi, pca_lo, 3u, &best);\n }\n fit_from_endpoints(&pixels, bbox_hi, bbox_lo, 3u, &best);\n\n var indices: u32 = 0u;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n indices = indices | ((best.indices[k] & 3u) << (k * 2u));\n }\n\n let out = block_index * 2u;\n dst[out] = best.c0 | (best.c1 << 16u);\n dst[out + 1u] = indices;\n}\n";
|
|
360
|
+
var bc1_default = "// BC1 (DXT1) compute shader encoder.\n//\n// Each invocation encodes one 4x4 pixel block into an 8-byte BC1 block\n// written as 2 x u32 into the destination storage buffer. This is the f32\n// fallback; bc1_fast_f16.wgsl is the same algorithm and is preferred when\n// the device reports shader-f16.\n//\n// BC1 block layout (little-endian):\n// u32[0]: color0 (low 16) | color1 (high 16) both in RGB565\n// u32[1]: 16 x 2-bit indices, pixel 0 = bits 0..1, pixel 15 = bits 30..31\n//\n// We always force the 4-color mode (color0 > color1, numeric 16-bit):\n// idx 0 -> color0\n// idx 1 -> color1\n// idx 2 -> (2*color0 + color1) / 3\n// idx 3 -> ( color0 + 2*color1) / 3\n//\n// ALGORITHM: principal-axis endpoint seed (covariance power-iteration; inset\n// bbox on degenerate blocks), inset by ~half a 565 cell along the axis, then\n// a fused pass that projects every pixel onto the decoded-endpoint line (the\n// 4 palette entries are colinear and evenly spaced, so the nearest entry is\n// the rounded projection \u2014 no 4-entry search) while accumulating the\n// least-squares refit sums, followed by up to TWO refit rounds (re-quantise,\n// reproject with indices packed on the fly, accept only on lower block\n// error).\n\nstruct Params {\n blocks_x: u32,\n blocks_y: u32,\n width: u32,\n height: u32,\n};\n\n@group(0) @binding(0) var src_tex: texture_2d<f32>;\n@group(0) @binding(1) var<storage, read_write> dst: array<u32>;\n@group(0) @binding(2) var<uniform> params: Params;\n\nfn to565(c: vec3<f32>) -> u32 {\n // Round-to-nearest quantization into 5-6-5.\n let r = u32(clamp(floor(c.r * 31.0 + 0.5), 0.0, 31.0));\n let g = u32(clamp(floor(c.g * 63.0 + 0.5), 0.0, 63.0));\n let b = u32(clamp(floor(c.b * 31.0 + 0.5), 0.0, 31.0));\n return (r << 11u) | (g << 5u) | b;\n}\n\nfn from565(c: u32) -> vec3<f32> {\n let r = (c >> 11u) & 31u;\n let g = (c >> 5u) & 63u;\n let b = c & 31u;\n // 5/6-bit -> 8-bit: (x*527+23)>>6 (6-bit: 259/33) \u2014 round-to-nearest\n // scaling, matching bc1_ref.ts and typical hardware decoders (white ->\n // 255). Integer u32 math is exact. NOTE: this is NOT plain bit-replication\n // ((x<<3)|(x>>2)) \u2014 they differ for some codes (e.g. 5-bit 3 -> 25 vs 24).\n // Selecting indices against this palette is what makes the encoder agree\n // with what the GPU will actually sample.\n let r8 = (r * 527u + 23u) >> 6u;\n let g8 = (g * 259u + 33u) >> 6u;\n let b8 = (b * 527u + 23u) >> 6u;\n return vec3<f32>(vec3<u32>(r8, g8, b8)) / 255.0;\n}\n\n// One projection pass against the decoded endpoints of (c0,c1): the packed\n// 2-bit indices, the block's squared error, and the LSQ normal-equation sums\n// of the resulting assignment \u2014 so an accepted refit can seed the next\n// round. Levels s run 0..3 along p0\u2192p1 (palette = p0, p0+\u2153d, p0+\u2154d, p1 \u2014\n// colinear, evenly spaced, so rounding the projection IS the nearest-entry\n// search). Level \u2192 BC1 index: 0\u21920 (c0), 1\u21922 (\u2154c0+\u2153c1), 2\u21923, 3\u21921 (c1); as a\n// packed LUT: (0x78 >> 2L) & 3.\nstruct ProjStats {\n indices: u32,\n err: f32,\n sAA: f32, sBB: f32, sAB: f32,\n sAV: vec3<f32>, sBV: vec3<f32>,\n s_min: f32, s_max: f32,\n};\nfn project_stats(pix: ptr<function, array<vec3<f32>, 16>>, c0: u32, c1: u32) -> ProjStats {\n var out: ProjStats;\n out.indices = 0u;\n out.err = 0.0;\n out.sAA = 0.0; out.sBB = 0.0; out.sAB = 0.0;\n out.sAV = vec3<f32>(0.0); out.sBV = vec3<f32>(0.0);\n out.s_min = 3.0; out.s_max = 0.0;\n let p0 = from565(c0);\n let p1 = from565(c1);\n let dir = p1 - p0;\n let dd = dot(dir, dir);\n if (dd == 0.0) {\n // Unreachable for distinct 565 codes (the decode is injective); kept so\n // a degenerate call still returns a consistent error.\n out.s_min = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let e = (*pix)[k] - p0;\n out.err = out.err + dot(e, e);\n }\n return out;\n }\n let inv = 3.0 / dd;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let v = (*pix)[k];\n let s = clamp(floor(dot(v - p0, dir) * inv + 0.5), 0.0, 3.0);\n out.s_min = min(out.s_min, s); out.s_max = max(out.s_max, s);\n let b = s * (1.0 / 3.0); let a = 1.0 - b;\n out.sAA = out.sAA + a * a; out.sBB = out.sBB + b * b; out.sAB = out.sAB + a * b;\n out.sAV = out.sAV + a * v; out.sBV = out.sBV + b * v;\n let e = v - (p0 + b * dir);\n out.err = out.err + dot(e, e);\n out.indices = out.indices | (((0x78u >> (u32(s) * 2u)) & 3u) << (k * 2u));\n }\n return out;\n}\n\n// Principal colour axis via covariance power-iteration, seeded with the bbox\n// diagonal. Returns a unit axis, or vec3(0) for a degenerate (constant)\n// block. The bbox diagonal alone is sign-blind and points across\n// anti-correlated data (normal maps, hue edges) instead of along it.\nfn principal_axis(\n pixels: ptr<function, array<vec3<f32>, 16>>,\n mean: vec3<f32>,\n seed: vec3<f32>,\n) -> vec3<f32> {\n // Symmetric 3x3 covariance, stored as its three rows.\n var c0v = vec3<f32>(0.0);\n var c1v = vec3<f32>(0.0);\n var c2v = vec3<f32>(0.0);\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let d = (*pixels)[k] - mean;\n c0v = c0v + d.x * d;\n c1v = c1v + d.y * d;\n c2v = c2v + d.z * d;\n }\n var v = seed;\n var len = length(v);\n if (len < 1e-9) { return vec3<f32>(0.0); }\n v = v / len;\n for (var iter: u32 = 0u; iter < 8u; iter = iter + 1u) {\n let nv = vec3<f32>(dot(c0v, v), dot(c1v, v), dot(c2v, v));\n len = length(nv);\n if (len < 1e-12) { return vec3<f32>(0.0); }\n v = nv / len;\n }\n return v;\n}\n\n@compute @workgroup_size(8, 8, 1)\nfn encode(@builtin(global_invocation_id) gid: vec3<u32>) {\n if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) {\n return;\n }\n\n let bx = gid.x;\n let by = gid.y;\n let block_index = by * params.blocks_x + bx;\n\n let base = vec2<i32>(i32(bx) * 4, i32(by) * 4);\n let max_xy = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);\n\n var pixels: array<vec3<f32>, 16>;\n var bb_min = vec3<f32>(1.0, 1.0, 1.0);\n var bb_max = vec3<f32>(0.0, 0.0, 0.0);\n var mean = vec3<f32>(0.0);\n\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let lx = i32(i & 3u);\n let ly = i32(i >> 2u);\n // Clamp to edge for non-multiple-of-4 textures.\n let p = clamp(base + vec2<i32>(lx, ly), vec2<i32>(0, 0), max_xy);\n let c = textureLoad(src_tex, p, 0).rgb;\n pixels[i] = c;\n bb_min = min(bb_min, c);\n bb_max = max(bb_max, c);\n mean = mean + c;\n }\n mean = mean * (1.0 / 16.0);\n\n // Seed endpoints from the block's principal colour axis at the exact\n // projection extents, inset by ~half a 565 cell along the axis (stb_dxt\n // heuristic). Degenerate (near-flat) blocks keep the inset-bbox seed.\n var seed_hi: vec3<f32>;\n var seed_lo: vec3<f32>;\n let axis = principal_axis(&pixels, mean, bb_max - bb_min);\n if (dot(axis, axis) > 0.0) {\n var t_min: f32 = 1e30;\n var t_max: f32 = -1e30;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let t = dot(pixels[k] - mean, axis);\n t_min = min(t_min, t);\n t_max = max(t_max, t);\n }\n let pad = (t_max - t_min) / 16.0;\n seed_hi = clamp(mean + (t_max - pad) * axis, vec3<f32>(0.0), vec3<f32>(1.0));\n seed_lo = clamp(mean + (t_min + pad) * axis, vec3<f32>(0.0), vec3<f32>(1.0));\n } else {\n let inset = (bb_max - bb_min) / 16.0;\n seed_hi = clamp(bb_max - inset, vec3<f32>(0.0), vec3<f32>(1.0));\n seed_lo = clamp(bb_min + inset, vec3<f32>(0.0), vec3<f32>(1.0));\n }\n var c0 = to565(seed_hi);\n var c1 = to565(seed_lo);\n if (c0 == c1) {\n if (c1 > 0u) { c1 = c1 - 1u; } else { c0 = c0 + 1u; }\n } else if (c0 < c1) {\n let t = c0; c0 = c1; c1 = t;\n }\n\n // Fused seed pass, then up to TWO least-squares refit rounds, each\n // accepted only if the block's squared error actually decreases \u2014 the\n // refit minimises a continuous objective and can lose after 565\n // quantisation. Every pass re-accumulates the normal-equation sums, so an\n // accepted round seeds the next.\n var cur = project_stats(&pixels, c0, c1);\n for (var it: u32 = 0u; it < 2u; it = it + 1u) {\n // Refit only on a well-conditioned system: when every pixel lands on\n // ONE level (flat blocks \u2014 the 4-colour nudge forces c0 \u2260 c1 even\n // then) the system is rank-1 and det/numerators are pure float noise;\n // the solve would return garbage endpoints. With \u22652 levels\n // det = \u03A3_i<j (b_j \u2212 b_i)\xB2 \u2265 ~1.67, so 1e-3 is a safe guard.\n if (cur.s_min >= cur.s_max) { break; }\n let det = cur.sAA * cur.sBB - cur.sAB * cur.sAB;\n if (abs(det) <= 1e-3) { break; }\n // Clamp the refit to the block bbox (not [0,1]): on multi-cluster\n // blocks the unconstrained solve extrapolates far outside the block's\n // colours and the per-channel clamp then bends the hue \u2014 fringe pixels\n // decode to colours that exist nowhere in the block. Constraining to\n // the bbox also measures better in plain SSE (+1.6 dB on the colour\n // test card), so the accept-if-better guard below keeps more refits.\n let e0 = clamp((cur.sBB * cur.sAV - cur.sAB * cur.sBV) / det, bb_min, bb_max);\n let e1 = clamp((cur.sAA * cur.sBV - cur.sAB * cur.sAV) / det, bb_min, bb_max);\n var nc0 = to565(e0);\n var nc1 = to565(e1);\n if (nc0 == nc1) {\n if (nc1 > 0u) { nc1 = nc1 - 1u; } else { nc0 = nc0 + 1u; }\n } else if (nc0 < nc1) {\n let t = nc0; nc0 = nc1; nc1 = t;\n }\n if (nc0 == c0 && nc1 == c1) { break; }\n let nxt = project_stats(&pixels, nc0, nc1);\n if (nxt.err >= cur.err) { break; }\n c0 = nc0;\n c1 = nc1;\n cur = nxt;\n }\n\n let out = block_index * 2u;\n dst[out] = c0 | (c1 << 16u);\n dst[out + 1u] = cur.indices;\n}\n";
|
|
328
361
|
|
|
329
362
|
// src/bc1_fast_f16.wgsl
|
|
330
363
|
var bc1_fast_f16_default = `// bc1 "fast" encoder \u2014 f16 variant (requires the shader-f16 feature).
|
|
@@ -334,16 +367,19 @@ var bc1_fast_f16_default = `// bc1 "fast" encoder \u2014 f16 variant (requires t
|
|
|
334
367
|
// f16 ([0,1] domain). The algorithm is the same family as the BC7/ASTC fast
|
|
335
368
|
// paths rather than a port of bc1.wgsl's fast branch:
|
|
336
369
|
//
|
|
337
|
-
// 1.
|
|
370
|
+
// 1. principal-axis endpoint seed (covariance power-iteration; inset bbox
|
|
371
|
+
// on degenerate blocks), inset by ~half a 565 cell (stb_dxt heuristic)
|
|
338
372
|
// 2. quantise to 565, force 4-colour mode (c0 > c1)
|
|
339
373
|
// 3. ONE fused pass: project every pixel onto the decoded-endpoint line
|
|
340
374
|
// (the 4 palette entries are colinear and evenly spaced, so the nearest
|
|
341
375
|
// entry is the rounded projection \u2014 no 4-entry search) while
|
|
342
|
-
// accumulating the least-squares refit sums, the
|
|
343
|
-
//
|
|
344
|
-
// 4.
|
|
345
|
-
//
|
|
346
|
-
//
|
|
376
|
+
// accumulating the least-squares refit sums, the packed indices and the
|
|
377
|
+
// squared error
|
|
378
|
+
// 4. up to TWO refit rounds (mirroring the high path's iterated refits):
|
|
379
|
+
// re-quantise the refit endpoints, reproject (indices packed on the
|
|
380
|
+
// fly, sums re-accumulated to seed the next round), and accept each
|
|
381
|
+
// round only if the block error decreases \u2014 flat/single-level blocks
|
|
382
|
+
// skip these passes entirely
|
|
347
383
|
//
|
|
348
384
|
// vs the pre-projection fast branch (build palette + full 4-entry search \xD7 3
|
|
349
385
|
// passes + refit sums pass) this does roughly half the ALU per block. The
|
|
@@ -391,6 +427,56 @@ fn order565(a: u32, b: u32) -> vec2<u32> {
|
|
|
391
427
|
return vec2<u32>(c0, c1);
|
|
392
428
|
}
|
|
393
429
|
|
|
430
|
+
// One projection pass against the decoded endpoints of (c0,c1): the packed
|
|
431
|
+
// 2-bit indices, the block's squared error, and the LSQ normal-equation sums
|
|
432
|
+
// of the resulting assignment \u2014 so an accepted refit can seed the next
|
|
433
|
+
// round. Levels s run 0..3 along p0\u2192p1 (palette = p0, p0+\u2153d, p0+\u2154d, p1 \u2014
|
|
434
|
+
// colinear, evenly spaced, so rounding the projection IS the nearest-entry
|
|
435
|
+
// search). Level \u2192 BC1 index: 0\u21920 (c0), 1\u21922 (\u2154c0+\u2153c1), 2\u21923, 3\u21921 (c1); as a
|
|
436
|
+
// packed LUT: (0x78 >> 2L) & 3.
|
|
437
|
+
struct Proj {
|
|
438
|
+
indices: u32,
|
|
439
|
+
err: h,
|
|
440
|
+
sAA: h, sBB: h, sAB: h,
|
|
441
|
+
sAV: h3, sBV: h3,
|
|
442
|
+
s_min: h, s_max: h,
|
|
443
|
+
};
|
|
444
|
+
fn project_stats(pix: ptr<function, array<h3, 16>>, c0: u32, c1: u32) -> Proj {
|
|
445
|
+
var out: Proj;
|
|
446
|
+
out.indices = 0u;
|
|
447
|
+
out.err = h(0.0);
|
|
448
|
+
out.sAA = h(0.0); out.sBB = h(0.0); out.sAB = h(0.0);
|
|
449
|
+
out.sAV = h3(0.0); out.sBV = h3(0.0);
|
|
450
|
+
out.s_min = h(3.0); out.s_max = h(0.0);
|
|
451
|
+
let p0 = from565(c0);
|
|
452
|
+
let p1 = from565(c1);
|
|
453
|
+
let dir = p1 - p0;
|
|
454
|
+
let dd = dot(dir, dir);
|
|
455
|
+
if (dd == h(0.0)) {
|
|
456
|
+
// Unreachable for distinct 565 codes (the decode is injective); kept so
|
|
457
|
+
// a degenerate call still returns a consistent error.
|
|
458
|
+
out.s_min = h(0.0);
|
|
459
|
+
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
460
|
+
let e = (*pix)[k] - p0;
|
|
461
|
+
out.err = out.err + dot(e, e);
|
|
462
|
+
}
|
|
463
|
+
return out;
|
|
464
|
+
}
|
|
465
|
+
let inv = h(3.0) / dd;
|
|
466
|
+
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
467
|
+
let v = (*pix)[k];
|
|
468
|
+
let s = clamp(floor(dot(v - p0, dir) * inv + h(0.5)), h(0.0), h(3.0));
|
|
469
|
+
out.s_min = min(out.s_min, s); out.s_max = max(out.s_max, s);
|
|
470
|
+
let b = s * h(1.0 / 3.0); let a = h(1.0) - b;
|
|
471
|
+
out.sAA = out.sAA + a * a; out.sBB = out.sBB + b * b; out.sAB = out.sAB + a * b;
|
|
472
|
+
out.sAV = out.sAV + a * v; out.sBV = out.sBV + b * v;
|
|
473
|
+
let e = v - (p0 + b * dir);
|
|
474
|
+
out.err = out.err + dot(e, e);
|
|
475
|
+
out.indices = out.indices | (((0x78u >> (u32(s) * 2u)) & 3u) << (k * 2u));
|
|
476
|
+
}
|
|
477
|
+
return out;
|
|
478
|
+
}
|
|
479
|
+
|
|
394
480
|
@compute @workgroup_size(8, 8, 1)
|
|
395
481
|
fn encode(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
396
482
|
if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) { return; }
|
|
@@ -401,46 +487,77 @@ fn encode(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
|
401
487
|
var pix: array<h3, 16>;
|
|
402
488
|
var mn = h3(1.0);
|
|
403
489
|
var mxv = h3(0.0);
|
|
490
|
+
var mean = h3(0.0);
|
|
404
491
|
for (var i: u32 = 0u; i < 16u; i = i + 1u) {
|
|
405
492
|
let p = clamp(base + vec2<i32>(i32(i & 3u), i32(i >> 2u)), vec2<i32>(0), mx);
|
|
406
493
|
let px = h3(textureLoad(src_tex, p, 0).rgb);
|
|
407
494
|
pix[i] = px; mn = min(mn, px); mxv = max(mxv, px);
|
|
495
|
+
mean = mean + px;
|
|
496
|
+
}
|
|
497
|
+
mean = mean * h(1.0 / 16.0);
|
|
498
|
+
|
|
499
|
+
// Seed endpoints from the block's principal colour axis (covariance
|
|
500
|
+
// power-iteration, seeded with the bbox diagonal \u2014 same family as the
|
|
501
|
+
// 'high' path). The bbox diagonal is sign-blind: on anti-correlated
|
|
502
|
+
// channels (normal maps, hue edges) it points across the data instead of
|
|
503
|
+
// along it, the projection indices come out garbage, and the LSQ refit \u2014
|
|
504
|
+
// which fits endpoints GIVEN those indices \u2014 can't recover. Deviations are
|
|
505
|
+
// pre-scaled \xD716 so covariance entries for shallow blocks stay in f16's
|
|
506
|
+
// normal range (span ~1/255 \u2192 d\xB2 \u2248 1e-3) while full-range sums stay \u22644096;
|
|
507
|
+
// the iteration renormalises by the max component (a plain length() of the
|
|
508
|
+
// matvec output could overflow f16), so only the direction survives \u2014 the
|
|
509
|
+
// \xD7256 covariance scale is irrelevant.
|
|
510
|
+
var seed_hi: h3;
|
|
511
|
+
var seed_lo: h3;
|
|
512
|
+
var c0v = h3(0.0);
|
|
513
|
+
var c1v = h3(0.0);
|
|
514
|
+
var c2v = h3(0.0);
|
|
515
|
+
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
516
|
+
let d = (pix[k] - mean) * h(16.0);
|
|
517
|
+
c0v = c0v + d.x * d;
|
|
518
|
+
c1v = c1v + d.y * d;
|
|
519
|
+
c2v = c2v + d.z * d;
|
|
520
|
+
}
|
|
521
|
+
var axis = mxv - mn;
|
|
522
|
+
var axis_ok = true;
|
|
523
|
+
for (var it: u32 = 0u; it < 4u; it = it + 1u) {
|
|
524
|
+
let nv = h3(dot(c0v, axis), dot(c1v, axis), dot(c2v, axis));
|
|
525
|
+
let m = max(max(abs(nv.x), abs(nv.y)), abs(nv.z));
|
|
526
|
+
if (m < h(1e-4)) { axis_ok = false; break; }
|
|
527
|
+
axis = nv / m;
|
|
528
|
+
}
|
|
529
|
+
if (axis_ok) {
|
|
530
|
+
axis = axis / length(axis);
|
|
531
|
+
var t_min = h(4.0);
|
|
532
|
+
var t_max = h(-4.0);
|
|
533
|
+
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
534
|
+
let t = dot(pix[k] - mean, axis);
|
|
535
|
+
t_min = min(t_min, t);
|
|
536
|
+
t_max = max(t_max, t);
|
|
537
|
+
}
|
|
538
|
+
// Inset along the axis by ~half a 565 cell (stb_dxt heuristic, matching
|
|
539
|
+
// the degenerate-case bbox inset below).
|
|
540
|
+
let pad = (t_max - t_min) * h(1.0 / 16.0);
|
|
541
|
+
seed_hi = clamp(mean + (t_max - pad) * axis, h3(0.0), h3(1.0));
|
|
542
|
+
seed_lo = clamp(mean + (t_min + pad) * axis, h3(0.0), h3(1.0));
|
|
543
|
+
} else {
|
|
544
|
+
// Degenerate (near-flat) block: inset bbox seed, as before.
|
|
545
|
+
let inset = (mxv - mn) * h(1.0 / 16.0);
|
|
546
|
+
seed_hi = clamp(mxv - inset, h3(0.0), h3(1.0));
|
|
547
|
+
seed_lo = clamp(mn + inset, h3(0.0), h3(1.0));
|
|
408
548
|
}
|
|
409
|
-
|
|
410
|
-
// Inset bbox by ~half a 565 cell so the quantised palette hugs the data.
|
|
411
|
-
let inset = (mxv - mn) * h(1.0 / 16.0);
|
|
412
|
-
let seed = order565(to565(clamp(mxv - inset, h3(0.0), h3(1.0))), to565(clamp(mn + inset, h3(0.0), h3(1.0))));
|
|
549
|
+
let seed = order565(to565(seed_hi), to565(seed_lo));
|
|
413
550
|
var c0 = seed.x;
|
|
414
551
|
var c1 = seed.y;
|
|
415
|
-
let p0 = from565(c0);
|
|
416
|
-
let p1 = from565(c1);
|
|
417
552
|
|
|
418
|
-
// Fused pass
|
|
419
|
-
//
|
|
420
|
-
//
|
|
421
|
-
//
|
|
422
|
-
//
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
if (dd > h(0.0)) {
|
|
427
|
-
let inv = h(3.0) / dd;
|
|
428
|
-
var sAA = h(0.0); var sBB = h(0.0); var sAB = h(0.0);
|
|
429
|
-
var sAV = h3(0.0); var sBV = h3(0.0);
|
|
430
|
-
var s_min = h(3.0); var s_max = h(0.0);
|
|
431
|
-
var seed_err = h(0.0);
|
|
432
|
-
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
433
|
-
let v = pix[k];
|
|
434
|
-
let s = clamp(floor(dot(v - p0, dir) * inv + h(0.5)), h(0.0), h(3.0));
|
|
435
|
-
s_min = min(s_min, s); s_max = max(s_max, s);
|
|
436
|
-
let b = s * h(1.0 / 3.0); let a = h(1.0) - b;
|
|
437
|
-
sAA = sAA + a * a; sBB = sBB + b * b; sAB = sAB + a * b;
|
|
438
|
-
sAV = sAV + a * v; sBV = sBV + b * v;
|
|
439
|
-
let e = v - (p0 + b * dir);
|
|
440
|
-
seed_err = seed_err + dot(e, e);
|
|
441
|
-
indices = indices | (((0x78u >> (u32(s) * 2u)) & 3u) << (k * 2u));
|
|
442
|
-
}
|
|
443
|
-
let det = sAA * sBB - sAB * sAB;
|
|
553
|
+
// Fused seed pass, then up to TWO least-squares refit rounds (mirroring
|
|
554
|
+
// the high path's iterated refits, at projection cost), each accepted only
|
|
555
|
+
// if the block's squared error actually decreases \u2014 the refit minimises a
|
|
556
|
+
// continuous objective and can lose after 565 quantisation. Every pass
|
|
557
|
+
// re-accumulates the normal-equation sums, so an accepted round seeds the
|
|
558
|
+
// next.
|
|
559
|
+
var cur = project_stats(&pix, c0, c1);
|
|
560
|
+
for (var it: u32 = 0u; it < 2u; it = it + 1u) {
|
|
444
561
|
// Refit only on a well-conditioned system. When every pixel lands on ONE
|
|
445
562
|
// level (flat / near-flat blocks \u2014 note the 4-colour-mode nudge forces
|
|
446
563
|
// c0 \u2260 c1 even for perfectly flat blocks) the system is rank-1: det is 0
|
|
@@ -448,45 +565,29 @@ fn encode(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
|
448
565
|
// noise, so the solve returns garbage endpoints. With \u22652 distinct levels
|
|
449
566
|
// det = \u03A3_i<j (b_j \u2212 b_i)\xB2 \u2265 15\xB7(1/3)\xB2 \u2248 1.67, far above the ~0.05 f16
|
|
450
567
|
// noise floor \u2014 0.5 separates the two regimes cleanly.
|
|
451
|
-
if (s_min
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
let ninv = h(3.0) / ndd;
|
|
470
|
-
var refit_err = h(0.0);
|
|
471
|
-
var nindices: u32 = 0u;
|
|
472
|
-
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
473
|
-
let v = pix[k];
|
|
474
|
-
let s = clamp(floor(dot(v - np0, ndir) * ninv + h(0.5)), h(0.0), h(3.0));
|
|
475
|
-
let e = v - (np0 + s * h(1.0 / 3.0) * ndir);
|
|
476
|
-
refit_err = refit_err + dot(e, e);
|
|
477
|
-
nindices = nindices | (((0x78u >> (u32(s) * 2u)) & 3u) << (k * 2u));
|
|
478
|
-
}
|
|
479
|
-
if (refit_err < seed_err) {
|
|
480
|
-
c0 = refit.x; c1 = refit.y;
|
|
481
|
-
indices = nindices;
|
|
482
|
-
}
|
|
483
|
-
}
|
|
484
|
-
}
|
|
568
|
+
if (cur.s_min >= cur.s_max) { break; }
|
|
569
|
+
let det = cur.sAA * cur.sBB - cur.sAB * cur.sAB;
|
|
570
|
+
if (abs(det) <= h(0.5)) { break; }
|
|
571
|
+
// Clamp the refit to the block bbox (not [0,1]): on multi-cluster blocks
|
|
572
|
+
// the unconstrained solve extrapolates far outside the block's colours
|
|
573
|
+
// and the per-channel clamp then bends the hue \u2014 fringe pixels decode to
|
|
574
|
+
// colours that exist nowhere in the block. Constraining to the bbox also
|
|
575
|
+
// measures better in plain SSE (+1.6 dB on the colour test card), so the
|
|
576
|
+
// accept-if-better guard below keeps more refits.
|
|
577
|
+
let e0 = clamp((cur.sBB * cur.sAV - cur.sAB * cur.sBV) / det, mn, mxv);
|
|
578
|
+
let e1 = clamp((cur.sAA * cur.sBV - cur.sAB * cur.sAV) / det, mn, mxv);
|
|
579
|
+
let rq = order565(to565(e0), to565(e1));
|
|
580
|
+
if (rq.x == c0 && rq.y == c1) { break; }
|
|
581
|
+
let nxt = project_stats(&pix, rq.x, rq.y);
|
|
582
|
+
if (nxt.err >= cur.err) { break; }
|
|
583
|
+
c0 = rq.x;
|
|
584
|
+
c1 = rq.y;
|
|
585
|
+
cur = nxt;
|
|
485
586
|
}
|
|
486
587
|
|
|
487
588
|
let o = bi * 2u;
|
|
488
589
|
dst[o] = c0 | (c1 << 16u);
|
|
489
|
-
dst[o + 1u] = indices;
|
|
590
|
+
dst[o + 1u] = cur.indices;
|
|
490
591
|
}
|
|
491
592
|
`;
|
|
492
593
|
|
|
@@ -503,9 +604,6 @@ var BC1Encoder = class extends Encoder {
|
|
|
503
604
|
get supportsSrgb() {
|
|
504
605
|
return true;
|
|
505
606
|
}
|
|
506
|
-
get supportsQuality() {
|
|
507
|
-
return true;
|
|
508
|
-
}
|
|
509
607
|
wgslSource() {
|
|
510
608
|
return bc1_default;
|
|
511
609
|
}
|
|
@@ -518,12 +616,12 @@ var BC1Encoder = class extends Encoder {
|
|
|
518
616
|
};
|
|
519
617
|
|
|
520
618
|
// src/bc5.wgsl
|
|
521
|
-
var bc5_default = "// BC5 (RGTC2) compute shader encoder.\n//\n// Each invocation encodes one 4x4 pixel block into a 16-byte BC5 block\n// written as 4 x u32 into the destination storage buffer.\n//\n// BC5 = two BC4 blocks concatenated:\n// block bytes 0..7 : BC4 of R channel (normal.x for tangent-space normals)\n// block bytes 8..15 : BC4 of G channel (normal.y)\n//\n// Each BC4 half-block (8 bytes):\n// byte 0 : red0 (8-bit endpoint)\n// byte 1 : red1 (8-bit endpoint)\n// bytes 2..7 : 16 \xD7 3-bit indices, LSB-first, pixel 0 at bit 0\n//\n// We always produce the 6-interpolation mode (red0 > red1). See\n// `bc4_ref.js` for the reasoning and the CPU reference this shader is\n// ported from \u2014 the algorithm and edge cases mirror it line-for-line.\n//\n// Pipeline per channel:\n// 1. Load 16 single-channel values, find min/max \u2192 initial endpoints.\n// 2. Quantize to 8-bit. Nudge apart if equal (forces 6-interp mode).\n// 3. Build palette, assign each texel its nearest entry (full L2).\n// 4. One-pass least-squares refinement: solve the 2\xD72 normal equations\n// for the (r0, r1) that minimizes \u03A3(palette[i_k] \u2212 v_k)\xB2. Accept\n// only if quantized endpoints still satisfy r0 > r1 AND total\n// squared error decreased.\n// 5. Pack 2 endpoint bytes + 48 bits of indices into the 8-byte block.\n//\n// The candidate endpoints/indices/error are tracked in place \u2014 the refit\n// overwrites them only when accepted \u2014 so no 16-entry index array is ever\n// copied across a function return.\n//\n// QUALITY LEVELS (pipeline-overridable constant `QUALITY_HIGH`)\n// fast (0, default): bbox endpoints + O(1) projection assignment per texel.\n// The 8-entry palette in 6-interp mode is colinear and EVENLY spaced from\n// r0 to r1 (levels 0..7 in palette order 0,2,3,4,5,6,7,1), so the nearest\n// entry is the rounded projection onto the r0\u2192r1 axis \u2014 no 8-entry\n// search, and the 3-bit indices are packed on the fly. The LSQ refit is\n// skipped (buys only ~0.36 dB).\n// high (1): full nearest search + refit, matches bc4_ref/bc5_ref.\n\n// 0 = fast (default), 1 = exhaustive/high-quality. Set via pipeline constants.\noverride QUALITY_HIGH: u32 = 0u;\n\nstruct Params {\n blocks_x: u32,\n blocks_y: u32,\n width: u32,\n height: u32,\n};\n\n@group(0) @binding(0) var src_tex: texture_2d<f32>;\n@group(0) @binding(1) var<storage, read_write> dst: array<u32>;\n@group(0) @binding(2) var<uniform> params: Params;\n\n// 6-interpolation-mode palette weights. palette[j] = W0_6[j]*r0 + W1_6[j]*r1.\n// Expressed as a switch so we don't rely on module-scope const arrays.\nfn w0_6(j: u32) -> f32 {\n switch j {\n case 0u: { return 1.0; }\n case 1u: { return 0.0; }\n case 2u: { return 6.0 / 7.0; }\n case 3u: { return 5.0 / 7.0; }\n case 4u: { return 4.0 / 7.0; }\n case 5u: { return 3.0 / 7.0; }\n case 6u: { return 2.0 / 7.0; }\n default: { return 1.0 / 7.0; } // case 7u\n }\n}\n\nfn w1_6(j: u32) -> f32 {\n switch j {\n case 0u: { return 0.0; }\n case 1u: { return 1.0; }\n case 2u: { return 1.0 / 7.0; }\n case 3u: { return 2.0 / 7.0; }\n case 4u: { return 3.0 / 7.0; }\n case 5u: { return 4.0 / 7.0; }\n case 6u: { return 5.0 / 7.0; }\n default: { return 6.0 / 7.0; } // case 7u\n }\n}\n\nfn quantize8(v: f32) -> u32 {\n // Round-to-nearest, clamp to [0, 255]. floor(x + 0.5) is the same\n // rounding rule the CPU reference uses.\n return u32(clamp(floor(v * 255.0 + 0.5), 0.0, 255.0));\n}\n\n// Build the 8-entry palette for endpoints (r0f, r1f) in normalised space.\nfn build_pal(r0f: f32, r1f: f32, pal: ptr<function, array<f32, 8>>) {\n for (var j: u32 = 0u; j < 8u; j = j + 1u) {\n (*pal)[j] = w0_6(j) * r0f + w1_6(j) * r1f;\n }\n}\n\n// Assign each of the 16 values its nearest palette entry (full 8-entry L2),\n// writing indices into `out_idx` and returning the total squared error.\nfn assign_all(\n values: ptr<function, array<f32, 16>>,\n pal: ptr<function, array<f32, 8>>,\n out_idx: ptr<function, array<u32, 16>>,\n) -> f32 {\n var err: f32 = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let v = (*values)[k];\n var best_j: u32 = 0u;\n var best_d: f32 = 1e20;\n for (var j: u32 = 0u; j < 8u; j = j + 1u) {\n let d = (*pal)[j] - v;\n let d2 = d * d;\n if (d2 < best_d) {\n best_d = d2;\n best_j = j;\n }\n }\n (*out_idx)[k] = best_j;\n err = err + best_d;\n }\n return err;\n}\n\n// Encode 16 single-channel values into an 8-byte BC4 block, packed as\n// two little-endian u32s (u32[0] = bytes 0..3, u32[1] = bytes 4..7).\nfn encode_bc4(values: ptr<function, array<f32, 16>>) -> vec2<u32> {\n // ---------------- 1. Initial endpoints: bbox of input ----------------\n var vmin: f32 = 1.0;\n var vmax: f32 = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n vmin = min(vmin, (*values)[k]);\n vmax = max(vmax, (*values)[k]);\n }\n var r0: u32 = quantize8(vmax);\n var r1: u32 = quantize8(vmin);\n // Force 6-interp mode: red0 > red1 strictly.\n if (r0 == r1) {\n if (r1 > 0u) { r1 = r1 - 1u; }\n else { r0 = r0 + 1u; }\n }\n\n if (QUALITY_HIGH == 0u) {\n // -------- fast: projection assignment, indices packed on the fly ----\n // level = round(7\xB7(v \u2212 r0)/(r1 \u2212 r0)); level \u2192 BC4 index LUT (0,2,3,4,\n // 5,6,7,1) packed as 3-bit entries in 0x3F58D0. Pixel k's 3 bits start\n // at bit 3k+16 of the (w0,w1) pair (bytes 0..1 are the endpoints).\n let r0f = f32(r0) / 255.0;\n let scale = 7.0 / (f32(r1) / 255.0 - r0f);\n var w0 = r0 | (r1 << 8u);\n var w1 = 0u;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let L = u32(clamp(floor(((*values)[k] - r0f) * scale + 0.5), 0.0, 7.0));\n let idx = (0x3F58D0u >> (L * 3u)) & 7u;\n let bit = 3u * k + 16u;\n if (bit <= 29u) {\n w0 = w0 | (idx << bit);\n } else if (bit >= 32u) {\n w1 = w1 | (idx << (bit - 32u));\n } else {\n // k = 5 straddles the word boundary (bits 31..33).\n w0 = w0 | (idx << bit);\n w1 = w1 | (idx >> (32u - bit));\n }\n }\n return vec2<u32>(w0, w1);\n }\n\n // ---------------- 2. Initial palette + indices + error --------------\n var pal: array<f32, 8>;\n build_pal(f32(r0) / 255.0, f32(r1) / 255.0, &pal);\n var indices: array<u32, 16>;\n var err = assign_all(values, &pal, &indices);\n\n // ---------------- 3. Refinement: least-squares on (r0, r1) ----------\n // High-quality only \u2014 the refit is the bulk of the per-channel cost and the\n // branch is resolved at pipeline-compile time, so the fast path skips all of\n // it (the sums loop included), not just the acceptance test.\n if (QUALITY_HIGH != 0u) {\n // Normal equations for palette[j] = a_j * r0 + b_j * r1:\n // [\u03A3AA \u03A3AB] [r0] [\u03A3AV]\n // [\u03A3AB \u03A3BB] [r1] = [\u03A3BV]\n var sAA: f32 = 0.0; var sBB: f32 = 0.0; var sAB: f32 = 0.0;\n var sAV: f32 = 0.0; var sBV: f32 = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let a = w0_6(indices[k]);\n let b = w1_6(indices[k]);\n let v = (*values)[k];\n sAA = sAA + a * a;\n sBB = sBB + b * b;\n sAB = sAB + a * b;\n sAV = sAV + a * v;\n sBV = sBV + b * v;\n }\n let det = sAA * sBB - sAB * sAB;\n // Degenerate system \u2192 skip refinement.\n if (abs(det) > 1e-9) {\n let new_r0 = clamp((sBB * sAV - sAB * sBV) / det, 0.0, 1.0);\n let new_r1 = clamp((sAA * sBV - sAB * sAV) / det, 0.0, 1.0);\n let qR0 = quantize8(new_r0);\n let qR1 = quantize8(new_r1);\n // Only accept refinements that stay in 6-interp mode. A refinement\n // that flips or equalizes the endpoints would change decode mode.\n if (qR0 > qR1) {\n build_pal(f32(qR0) / 255.0, f32(qR1) / 255.0, &pal);\n var idx2: array<u32, 16>;\n let err2 = assign_all(values, &pal, &idx2);\n if (err2 < err) {\n r0 = qR0;\n r1 = qR1;\n indices = idx2;\n err = err2;\n }\n }\n }\n }\n\n // ---------------- 4. Pack 48-bit index field + 2 endpoint bytes -----\n // The 48-bit index field spans block bytes 2..7. Split into idx_lo\n // (low 32 bits of the field) and idx_hi (high 16 bits). An index at\n // bit position 3k straddles the 32-bit boundary iff 3k < 32 < 3k+3\n // (only k = 10, 11 straddle: bits 30..32 and 33..35; actually k=10\n // is bits 30..32, k=11 is 33..35 \u2014 so k=10 straddles). We handle\n // straddles by writing to both halves.\n var idx_lo: u32 = 0u;\n var idx_hi: u32 = 0u;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let bit = 3u * k;\n let v = indices[k] & 7u;\n if (bit + 3u <= 32u) {\n idx_lo = idx_lo | (v << bit);\n } else if (bit >= 32u) {\n idx_hi = idx_hi | (v << (bit - 32u));\n } else {\n // Straddle: low part into idx_lo's top, high part into idx_hi's bottom.\n idx_lo = idx_lo | (v << bit);\n idx_hi = idx_hi | (v >> (32u - bit));\n }\n }\n\n // Final u32s, both little-endian:\n // u32[0] bytes = red0, red1, idx_lo[7:0], idx_lo[15:8]\n // u32[1] bytes = idx_lo[23:16], idx_lo[31:24], idx_hi[7:0], idx_hi[15:8]\n let out_lo = r0 | (r1 << 8u) | ((idx_lo & 0xFFFFu) << 16u);\n let out_hi = (idx_lo >> 16u) | (idx_hi << 16u);\n\n return vec2<u32>(out_lo, out_hi);\n}\n\n@compute @workgroup_size(8, 8, 1)\nfn encode(@builtin(global_invocation_id) gid: vec3<u32>) {\n if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) {\n return;\n }\n\n let bx = gid.x;\n let by = gid.y;\n let block_index = by * params.blocks_x + bx;\n\n let base = vec2<i32>(i32(bx) * 4, i32(by) * 4);\n let max_xy = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);\n\n // Load 4\xD74 RG values, splitting into per-channel arrays so each can\n // be handed to encode_bc4 independently.\n var r_values: array<f32, 16>;\n var g_values: array<f32, 16>;\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let lx = i32(i & 3u);\n let ly = i32(i >> 2u);\n // Clamp to edge for non-multiple-of-4 input sizes.\n let p = clamp(base + vec2<i32>(lx, ly), vec2<i32>(0, 0), max_xy);\n let c = textureLoad(src_tex, p, 0);\n r_values[i] = c.r;\n g_values[i] = c.g;\n }\n\n let r_block = encode_bc4(&r_values);\n let g_block = encode_bc4(&g_values);\n\n // BC5 block = R half (bytes 0..7) || G half (bytes 8..15) = 4 u32s.\n let out = block_index * 4u;\n dst[out + 0u] = r_block.x;\n dst[out + 1u] = r_block.y;\n dst[out + 2u] = g_block.x;\n dst[out + 3u] = g_block.y;\n}\n";
|
|
619
|
+
var bc5_default = "// BC5 (RGTC2) compute shader encoder.\n//\n// Each invocation encodes one 4x4 pixel block into a 16-byte BC5 block\n// written as 4 x u32 into the destination storage buffer. This is the f32\n// fallback; bc5_fast_f16.wgsl is the same algorithm and is preferred when\n// the device reports shader-f16.\n//\n// BC5 = two BC4 blocks concatenated:\n// block bytes 0..7 : BC4 of R channel (normal.x for tangent-space normals)\n// block bytes 8..15 : BC4 of G channel (normal.y)\n//\n// Each BC4 half-block (8 bytes):\n// byte 0 : red0 (8-bit endpoint)\n// byte 1 : red1 (8-bit endpoint)\n// bytes 2..7 : 16 \xD7 3-bit indices, LSB-first, pixel 0 at bit 0\n//\n// We always produce the 6-interpolation mode (red0 > red1). See `bc4_ref.ts`\n// for the reference this encoder is validated against.\n//\n// Pipeline per channel: bbox endpoints + O(1) projection assignment per\n// texel. The 8-entry palette in 6-interp mode is COLINEAR and EVENLY spaced\n// from r0 to r1 (levels 0..7 in palette order 0,2,3,4,5,6,7,1), so the\n// nearest entry is the rounded projection onto the r0\u2192r1 axis \u2014 no 8-entry\n// search, and the 3-bit indices are packed on the fly. The LSQ refit sums\n// are accumulated in the same fused pass; the requantised refit is accepted\n// only if it lowers the block error (worth ~1.3 dB on the normal-map card).\n\nstruct Params {\n blocks_x: u32,\n blocks_y: u32,\n width: u32,\n height: u32,\n};\n\n@group(0) @binding(0) var src_tex: texture_2d<f32>;\n@group(0) @binding(1) var<storage, read_write> dst: array<u32>;\n@group(0) @binding(2) var<uniform> params: Params;\n\nfn quantize8(v: f32) -> u32 {\n // Round-to-nearest, clamp to [0, 255]. floor(x + 0.5) is the same\n // rounding rule the CPU reference uses.\n return u32(clamp(floor(v * 255.0 + 0.5), 0.0, 255.0));\n}\n\n// Encode 16 single-channel values into an 8-byte BC4 block, packed as\n// two little-endian u32s (u32[0] = bytes 0..3, u32[1] = bytes 4..7).\n// vmin/vmax are the channel's min/max, computed in the caller's load loop \u2014\n// fusing that scan there saves a 16-value pass per channel.\nfn encode_bc4(values: ptr<function, array<f32, 16>>, vmin: f32, vmax: f32) -> vec2<u32> {\n var r0: u32 = quantize8(vmax);\n var r1: u32 = quantize8(vmin);\n // Force 6-interp mode: red0 > red1 strictly.\n if (r0 == r1) {\n if (r1 > 0u) { r1 = r1 - 1u; }\n else { r0 = r0 + 1u; }\n }\n\n // ONE fused pass: level = round(7\xB7(v \u2212 r0)/(r1 \u2212 r0)) projection\n // assignment (the 8-entry 6-interp palette is colinear and evenly\n // spaced, so the rounded projection IS the nearest-entry search), the\n // seed solution's packed indices and squared error, and the least-\n // squares normal-equation sums. The refit endpoints are re-quantised,\n // reprojected, and accepted only if the block error decreases.\n // Level \u2192 BC4 index LUT (0,2,3,4,5,6,7,1) packed as 3-bit entries in\n // 0x3F58D0. Pixel k's 3 bits start at bit 3k+16 of the (w0,w1) pair\n // (bytes 0..1 are the endpoints); k = 5 straddles the word boundary.\n let r0f = f32(r0) / 255.0;\n let dir = f32(r1) / 255.0 - r0f;\n let scale = 7.0 / dir;\n var w0 = r0 | (r1 << 8u);\n var w1 = 0u;\n var sAA = 0.0; var sBB = 0.0; var sAB = 0.0;\n var sAV = 0.0; var sBV = 0.0;\n var s_min = 7.0; var s_max = 0.0;\n var seed_err = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let vr = (*values)[k] - r0f;\n let L = clamp(floor(vr * scale + 0.5), 0.0, 7.0);\n s_min = min(s_min, L); s_max = max(s_max, L);\n let b = L * (1.0 / 7.0); let a = 1.0 - b;\n sAA = sAA + a * a; sBB = sBB + b * b; sAB = sAB + a * b;\n sAV = sAV + a * vr; sBV = sBV + b * vr;\n let e = vr - b * dir;\n seed_err = seed_err + e * e;\n let idx = (0x3F58D0u >> (u32(L) * 3u)) & 7u;\n let bit = 3u * k + 16u;\n if (bit <= 29u) {\n w0 = w0 | (idx << bit);\n } else if (bit >= 32u) {\n w1 = w1 | (idx << (bit - 32u));\n } else {\n w0 = w0 | (idx << bit);\n w1 = w1 | (idx >> (32u - bit));\n }\n }\n\n // Rank-1 guard: with every pixel on ONE level the system is singular\n // (det is float rounding noise); with \u22652 distinct levels\n // det = \u03A3_i<j (b_j \u2212 b_i)\xB2 \u2265 15/49 \u2248 0.306.\n if (s_min < s_max) {\n let det = sAA * sBB - sAB * sAB;\n if (abs(det) > 1e-3) {\n // Clamp to [0,1], NOT the block's value range: for a scalar channel,\n // endpoints beyond the data range are often genuinely optimal (they\n // centre the palette levels on the data) and there is no colour axis\n // to bend \u2014 the bbox clamp the colour formats need costs ~0.3 dB\n // here. The accept-if-better guard still protects against a refit\n // that loses after quantisation.\n let e0 = clamp(r0f + (sBB * sAV - sAB * sBV) / det, 0.0, 1.0);\n let e1 = clamp(r0f + (sAA * sBV - sAB * sAV) / det, 0.0, 1.0);\n let n0 = quantize8(e0);\n let n1 = quantize8(e1);\n // Keep 6-interp mode (r0 > r1 strictly); skip the no-op refit.\n if (n0 > n1 && !(n0 == r0 && n1 == r1)) {\n let n0f = f32(n0) / 255.0;\n let ndir = f32(n1) / 255.0 - n0f;\n let nscale = 7.0 / ndir;\n var nw0 = n0 | (n1 << 8u);\n var nw1 = 0u;\n var refit_err = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let vr = (*values)[k] - n0f;\n let L = clamp(floor(vr * nscale + 0.5), 0.0, 7.0);\n let e = vr - L * (1.0 / 7.0) * ndir;\n refit_err = refit_err + e * e;\n let idx = (0x3F58D0u >> (u32(L) * 3u)) & 7u;\n let bit = 3u * k + 16u;\n if (bit <= 29u) {\n nw0 = nw0 | (idx << bit);\n } else if (bit >= 32u) {\n nw1 = nw1 | (idx << (bit - 32u));\n } else {\n nw0 = nw0 | (idx << bit);\n nw1 = nw1 | (idx >> (32u - bit));\n }\n }\n if (refit_err < seed_err) {\n return vec2<u32>(nw0, nw1);\n }\n }\n }\n }\n return vec2<u32>(w0, w1);\n}\n\n@compute @workgroup_size(8, 8, 1)\nfn encode(@builtin(global_invocation_id) gid: vec3<u32>) {\n if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) {\n return;\n }\n\n let bx = gid.x;\n let by = gid.y;\n let block_index = by * params.blocks_x + bx;\n\n let base = vec2<i32>(i32(bx) * 4, i32(by) * 4);\n let max_xy = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);\n\n // Load 4\xD74 RG values, splitting into per-channel arrays so each can\n // be handed to encode_bc4 independently; the per-channel min/max scan is\n // fused into the same loop.\n var r_values: array<f32, 16>;\n var g_values: array<f32, 16>;\n var r_min: f32 = 1.0; var r_max: f32 = 0.0;\n var g_min: f32 = 1.0; var g_max: f32 = 0.0;\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let lx = i32(i & 3u);\n let ly = i32(i >> 2u);\n // Clamp to edge for non-multiple-of-4 input sizes.\n let p = clamp(base + vec2<i32>(lx, ly), vec2<i32>(0, 0), max_xy);\n let c = textureLoad(src_tex, p, 0);\n r_values[i] = c.r;\n g_values[i] = c.g;\n r_min = min(r_min, c.r); r_max = max(r_max, c.r);\n g_min = min(g_min, c.g); g_max = max(g_max, c.g);\n }\n\n let r_block = encode_bc4(&r_values, r_min, r_max);\n let g_block = encode_bc4(&g_values, g_min, g_max);\n\n // BC5 block = R half (bytes 0..7) || G half (bytes 8..15) = 4 u32s.\n let out = block_index * 4u;\n dst[out + 0u] = r_block.x;\n dst[out + 1u] = r_block.y;\n dst[out + 2u] = g_block.x;\n dst[out + 3u] = g_block.y;\n}\n";
|
|
522
620
|
|
|
523
621
|
// src/bc5_fast_f16.wgsl
|
|
524
622
|
var bc5_fast_f16_default = `// bc5 "fast" encoder \u2014 f16 variant (requires the shader-f16 feature).
|
|
525
|
-
// Two BC4 halves (R and G)
|
|
526
|
-
//
|
|
623
|
+
// Two BC4 halves (R and G) \u2014 same output family as bc5.wgsl's fast branch,
|
|
624
|
+
// tuned for throughput:
|
|
527
625
|
//
|
|
528
626
|
// \u2022 The 8-entry palette in 6-interpolation mode is COLINEAR and EVENLY
|
|
529
627
|
// spaced from r0 to r1 (levels 0..7 in palette order 0,2,3,4,5,6,7,1),
|
|
@@ -532,9 +630,21 @@ var bc5_fast_f16_default = `// bc5 "fast" encoder \u2014 f16 variant (requires t
|
|
|
532
630
|
// \u2022 Math runs in the exact-integer [0,255] f16 domain: endpoints and pixel
|
|
533
631
|
// values are whole numbers \u2264 255 (exact in f16), so the only rounding is
|
|
534
632
|
// the single 1/(r1\u2212r0) division.
|
|
633
|
+
// \u2022 ONE fused pass per channel: projection assignment + the least-squares
|
|
634
|
+
// refit sums + the seed solution's packed indices and squared error. The
|
|
635
|
+
// refit endpoints are re-quantised, reprojected, and accepted only if
|
|
636
|
+
// the block error decreases (same accept-if-better family as the BC1
|
|
637
|
+
// fast path) \u2014 worth ~1.3 dB on the normal-map card.
|
|
535
638
|
// \u2022 3-bit indices are packed into the 48-bit field on the fly \u2014 no
|
|
536
639
|
// array<u32,16> private array and no separate packing loop.
|
|
537
640
|
//
|
|
641
|
+
// f16 range notes: value sums accumulate v \u2212 r0 (the affine-basis shift trick
|
|
642
|
+
// from the BC7/ASTC fast paths) scaled by 1/16, and error residuals are
|
|
643
|
+
// scaled by 1/16 before squaring \u2014 worst-case magnitudes stay \u22724k, well
|
|
644
|
+
// inside f16's 65504 max, with rounding a small fraction of a level. The
|
|
645
|
+
// accept-if-better guard makes any residual f16 noise fail-safe (worst case:
|
|
646
|
+
// the refit is rejected and the seed solution ships).
|
|
647
|
+
//
|
|
538
648
|
// Level \u2192 BC4 index (0\u2192r0 ... 7\u2192r1): 0,2,3,4,5,6,7,1 \u2014 packed 3-bit LUT
|
|
539
649
|
// 0x3F58D0 = sum(idx[L] << 3L).
|
|
540
650
|
//
|
|
@@ -547,14 +657,43 @@ struct Params { blocks_x: u32, blocks_y: u32, width: u32, height: u32, };
|
|
|
547
657
|
@group(0) @binding(1) var<storage, read_write> dst: array<u32>;
|
|
548
658
|
@group(0) @binding(2) var<uniform> params: Params;
|
|
549
659
|
|
|
550
|
-
//
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
660
|
+
// Project the 16 values onto the r0\u2192r1 axis, packing the 3-bit indices on the
|
|
661
|
+
// fly (pixel k's index starts at bit 3k of the 48-bit field, i.e. bit 3k+16
|
|
662
|
+
// of the (w0,w1) pair; k = 5 straddles the word boundary) and accumulating
|
|
663
|
+
// the squared error (residuals scaled by 1/16 before squaring). Returns the
|
|
664
|
+
// packed words with the endpoint bytes already in place.
|
|
665
|
+
struct Proj { w0: u32, w1: u32, err: h };
|
|
666
|
+
fn project_pack(values: ptr<function, array<h, 16>>, r0: u32, r1: u32) -> Proj {
|
|
667
|
+
let r0f = h(f32(r0));
|
|
668
|
+
let dir = h(f32(r1)) - r0f;
|
|
669
|
+
let scale = h(7.0) / dir;
|
|
670
|
+
var out: Proj;
|
|
671
|
+
out.w0 = r0 | (r1 << 8u);
|
|
672
|
+
out.w1 = 0u;
|
|
673
|
+
out.err = h(0.0);
|
|
554
674
|
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
555
|
-
|
|
556
|
-
|
|
675
|
+
let vr = (*values)[k] - r0f;
|
|
676
|
+
let L = clamp(floor(vr * scale + h(0.5)), h(0.0), h(7.0));
|
|
677
|
+
let e = (vr - L * h(1.0 / 7.0) * dir) * h(1.0 / 16.0);
|
|
678
|
+
out.err = out.err + e * e;
|
|
679
|
+
let idx = (0x3F58D0u >> (u32(L) * 3u)) & 7u;
|
|
680
|
+
let bit = 3u * k + 16u;
|
|
681
|
+
if (bit <= 29u) {
|
|
682
|
+
out.w0 = out.w0 | (idx << bit);
|
|
683
|
+
} else if (bit >= 32u) {
|
|
684
|
+
out.w1 = out.w1 | (idx << (bit - 32u));
|
|
685
|
+
} else {
|
|
686
|
+
out.w0 = out.w0 | (idx << bit);
|
|
687
|
+
out.w1 = out.w1 | (idx >> (32u - bit));
|
|
688
|
+
}
|
|
557
689
|
}
|
|
690
|
+
return out;
|
|
691
|
+
}
|
|
692
|
+
|
|
693
|
+
// Encode one channel (16 values in exact-integer [0,255] f16) to a BC4 half.
|
|
694
|
+
// vmin/vmax are the channel's min/max, computed in the caller's load loop \u2014
|
|
695
|
+
// fusing that scan there saves a 16-value pass per channel.
|
|
696
|
+
fn encode_bc4(values: ptr<function, array<h, 16>>, vmin: h, vmax: h) -> vec2<u32> {
|
|
558
697
|
var r0 = u32(vmax); // values are exact integers \u2014 no rounding needed
|
|
559
698
|
var r1 = u32(vmin);
|
|
560
699
|
if (r0 == r1) {
|
|
@@ -562,31 +701,71 @@ fn encode_bc4(values: ptr<function, array<h, 16>>) -> vec2<u32> {
|
|
|
562
701
|
if (r1 > 0u) { r1 = r1 - 1u; } else { r0 = r0 + 1u; }
|
|
563
702
|
}
|
|
564
703
|
|
|
565
|
-
//
|
|
566
|
-
// |v \u2212 r0| \u2264 r0 \u2212 r1 for
|
|
704
|
+
// Fused seed pass: projection assignment (level = round(7\xB7(v \u2212 r0)/
|
|
705
|
+
// (r1 \u2212 r0)), clamped \u2014 |v \u2212 r0| \u2264 r0 \u2212 r1 for in-block values) + packed
|
|
706
|
+
// indices + seed error + the LSQ normal-equation sums. Value sums
|
|
707
|
+
// accumulate (v \u2212 r0)/16: the shift keeps the accumulators proportional to
|
|
708
|
+
// the block's span, the exact power-of-two scale keeps products \u2264 4080.
|
|
567
709
|
let r0f = h(f32(r0));
|
|
568
|
-
let
|
|
569
|
-
|
|
570
|
-
var
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
var
|
|
710
|
+
let dir = h(f32(r1)) - r0f;
|
|
711
|
+
let scale = h(7.0) / dir;
|
|
712
|
+
var seed: Proj;
|
|
713
|
+
seed.w0 = r0 | (r1 << 8u);
|
|
714
|
+
seed.w1 = 0u;
|
|
715
|
+
seed.err = h(0.0);
|
|
716
|
+
var sAA = h(0.0); var sBB = h(0.0); var sAB = h(0.0);
|
|
717
|
+
var sAV = h(0.0); var sBV = h(0.0);
|
|
718
|
+
var s_min = h(7.0); var s_max = h(0.0);
|
|
575
719
|
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
576
|
-
let
|
|
577
|
-
let
|
|
720
|
+
let vr = (*values)[k] - r0f;
|
|
721
|
+
let L = clamp(floor(vr * scale + h(0.5)), h(0.0), h(7.0));
|
|
722
|
+
s_min = min(s_min, L); s_max = max(s_max, L);
|
|
723
|
+
let b = L * h(1.0 / 7.0); let a = h(1.0) - b;
|
|
724
|
+
let vr16 = vr * h(1.0 / 16.0);
|
|
725
|
+
sAA = sAA + a * a; sBB = sBB + b * b; sAB = sAB + a * b;
|
|
726
|
+
sAV = sAV + a * vr16; sBV = sBV + b * vr16;
|
|
727
|
+
let e = vr16 - b * dir * h(1.0 / 16.0);
|
|
728
|
+
seed.err = seed.err + e * e;
|
|
729
|
+
let idx = (0x3F58D0u >> (u32(L) * 3u)) & 7u;
|
|
578
730
|
let bit = 3u * k + 16u;
|
|
579
731
|
if (bit <= 29u) {
|
|
580
|
-
w0 = w0 | (idx << bit);
|
|
732
|
+
seed.w0 = seed.w0 | (idx << bit);
|
|
581
733
|
} else if (bit >= 32u) {
|
|
582
|
-
w1 = w1 | (idx << (bit - 32u));
|
|
734
|
+
seed.w1 = seed.w1 | (idx << (bit - 32u));
|
|
583
735
|
} else {
|
|
584
736
|
// k = 5 straddles the word boundary (bits 31..33).
|
|
585
|
-
w0 = w0 | (idx << bit);
|
|
586
|
-
w1 = w1 | (idx >> (32u - bit));
|
|
737
|
+
seed.w0 = seed.w0 | (idx << bit);
|
|
738
|
+
seed.w1 = seed.w1 | (idx >> (32u - bit));
|
|
739
|
+
}
|
|
740
|
+
}
|
|
741
|
+
|
|
742
|
+
// LSQ refit, accepted only if the requantised endpoints lower the block
|
|
743
|
+
// error. Rank-1 guard: with every pixel on ONE level the system is
|
|
744
|
+
// singular and det is pure f16 rounding noise (\u22720.03); with \u22652 distinct
|
|
745
|
+
// levels det = \u03A3_i<j (b_j \u2212 b_i)\xB2 \u2265 15/49 \u2248 0.306 \u2014 0.1 separates cleanly.
|
|
746
|
+
if (s_min < s_max) {
|
|
747
|
+
let det = sAA * sBB - sAB * sAB;
|
|
748
|
+
if (abs(det) > h(0.1)) {
|
|
749
|
+
// \xD716 undoes the accumulator scale. Clamp to [0,255], NOT the block's
|
|
750
|
+
// value range: for a scalar channel, endpoints beyond the data range
|
|
751
|
+
// are often genuinely optimal (they centre the palette levels on the
|
|
752
|
+
// data) and there is no colour axis to bend \u2014 the bbox clamp the
|
|
753
|
+
// colour formats need costs ~0.3 dB here. The accept-if-better guard
|
|
754
|
+
// still protects against a refit that loses after quantisation.
|
|
755
|
+
let e0 = clamp(r0f + (sBB * sAV - sAB * sBV) * h(16.0) / det, h(0.0), h(255.0));
|
|
756
|
+
let e1 = clamp(r0f + (sAA * sBV - sAB * sAV) * h(16.0) / det, h(0.0), h(255.0));
|
|
757
|
+
let n0 = u32(floor(e0 + h(0.5)));
|
|
758
|
+
let n1 = u32(floor(e1 + h(0.5)));
|
|
759
|
+
// Keep 6-interp mode (r0 > r1 strictly); skip the no-op refit.
|
|
760
|
+
if (n0 > n1 && !(n0 == r0 && n1 == r1)) {
|
|
761
|
+
let refit = project_pack(values, n0, n1);
|
|
762
|
+
if (refit.err < seed.err) {
|
|
763
|
+
return vec2<u32>(refit.w0, refit.w1);
|
|
764
|
+
}
|
|
765
|
+
}
|
|
587
766
|
}
|
|
588
767
|
}
|
|
589
|
-
return vec2<u32>(w0, w1);
|
|
768
|
+
return vec2<u32>(seed.w0, seed.w1);
|
|
590
769
|
}
|
|
591
770
|
|
|
592
771
|
@compute @workgroup_size(8, 8, 1)
|
|
@@ -597,14 +776,18 @@ fn encode(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
|
597
776
|
let mx = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);
|
|
598
777
|
var rv: array<h, 16>;
|
|
599
778
|
var gv: array<h, 16>;
|
|
779
|
+
var rmin = h(255.0); var rmax = h(0.0);
|
|
780
|
+
var gmin = h(255.0); var gmax = h(0.0);
|
|
600
781
|
for (var i: u32 = 0u; i < 16u; i = i + 1u) {
|
|
601
782
|
let p = clamp(base + vec2<i32>(i32(i & 3u), i32(i >> 2u)), vec2<i32>(0), mx);
|
|
602
783
|
let c = textureLoad(src_tex, p, 0);
|
|
603
|
-
|
|
604
|
-
|
|
784
|
+
let r = h(c.r * 255.0);
|
|
785
|
+
let g = h(c.g * 255.0);
|
|
786
|
+
rv[i] = r; rmin = min(rmin, r); rmax = max(rmax, r);
|
|
787
|
+
gv[i] = g; gmin = min(gmin, g); gmax = max(gmax, g);
|
|
605
788
|
}
|
|
606
|
-
let rb = encode_bc4(&rv);
|
|
607
|
-
let gb = encode_bc4(&gv);
|
|
789
|
+
let rb = encode_bc4(&rv, rmin, rmax);
|
|
790
|
+
let gb = encode_bc4(&gv, gmin, gmax);
|
|
608
791
|
let o = bi * 4u;
|
|
609
792
|
dst[o] = rb.x; dst[o + 1u] = rb.y; dst[o + 2u] = gb.x; dst[o + 3u] = gb.y;
|
|
610
793
|
}
|
|
@@ -623,9 +806,6 @@ var BC5Encoder = class extends Encoder {
|
|
|
623
806
|
get supportsSrgb() {
|
|
624
807
|
return false;
|
|
625
808
|
}
|
|
626
|
-
get supportsQuality() {
|
|
627
|
-
return true;
|
|
628
|
-
}
|
|
629
809
|
wgslSource() {
|
|
630
810
|
return bc5_default;
|
|
631
811
|
}
|
|
@@ -638,28 +818,31 @@ var BC5Encoder = class extends Encoder {
|
|
|
638
818
|
};
|
|
639
819
|
|
|
640
820
|
// src/bc7.wgsl
|
|
641
|
-
var bc7_default = "// BC7 (BPTC) mode 6 compute shader encoder.\n//\n// One invocation per 4\xD74 block. Emits 16 bytes = 4 u32s into the storage\n// buffer at `dst[block_index * 4 .. + 3]`.\n//\n// QUALITY LEVELS (pipeline-overridable constant `QUALITY_HIGH`)\n// fast (0, default): O(N) bounding-box seed \u2192 one fused pass that projects\n// each pixel onto the endpoint line (the 16 palette entries are colinear,\n// so the nearest index is the rounded projection \u2014 no palette build, no\n// 16-entry search) while accumulating the least-squares refit sums, then\n// a reprojection against the quantised refit endpoints for the final\n// indices, packed on the fly into two nibble words.\n// high (1): farthest-pair seed, exhaustive p-bit search over all four\n// (p0,p1) \u2208 {0,1}\xB2 combos, full 16-entry nearest search, one LSQ refit \u2014\n// matches bc7_ref.ts up to FP tie-breaks.\n//\n// The fast/high branch is selected at pipeline-compile time, so the driver\n// eliminates the unused code entirely.\n//\n// MODE 6 LAYOUT (LSB-first, bit 0 = byte 0's bit 0)\n// bits 0..6 mode field (0b0000001 \u2014 only bit 6 is 1)\n// bits 7..13 R0 (7-bit) bits 14..20 R1 bits 21..27 G0 bits 28..34 G1\n// bits 35..41 B0 bits 42..48 B1 bits 49..55 A0 bits 56..62 A1\n// bit 63 P0 bit 64 P1\n// bits 65..67 pixel 0 index (3 bits; anchor, MSB implicit 0)\n// bits 68..71 pixel 1 index (4 bits) ... bits 124..127 pixel 15 index\n//\n// Effective 8-bit endpoint channel = (7_bit_value << 1) | p_bit.\n// Palette[i] = ((64 \u2212 W4[i]) \xD7 e0_8 + W4[i] \xD7 e1_8 + 32) >> 6, integer.\n//\n// The block is assembled with straight-line constant shifts (see the layout\n// summary in bc7_fast_f16.wgsl) \u2014 a generic write_bits() helper's dynamic\n// word indexing keeps the output array out of registers.\n\n// 0 = fast (default), 1 = exhaustive/high-quality. Set via pipeline constants.\noverride QUALITY_HIGH: u32 = 0u;\n\nstruct Params {\n blocks_x: u32,\n blocks_y: u32,\n width: u32,\n height: u32,\n};\n\n@group(0) @binding(0) var src_tex: texture_2d<f32>;\n@group(0) @binding(1) var<storage, read_write> dst: array<u32>;\n@group(0) @binding(2) var<uniform> params: Params;\n\n// Mode 6 interpolation weights (\xD7 1/64), fixed by the spec (`W4` in bc7_ref.ts).\nfn w4(i: u32) -> u32 {\n switch i {\n case 0u: { return 0u; }\n case 1u: { return 4u; }\n case 2u: { return 9u; }\n case 3u: { return 13u; }\n case 4u: { return 17u; }\n case 5u: { return 21u; }\n case 6u: { return 26u; }\n case 7u: { return 30u; }\n case 8u: { return 34u; }\n case 9u: { return 38u; }\n case 10u: { return 43u; }\n case 11u: { return 47u; }\n case 12u: { return 51u; }\n case 13u: { return 55u; }\n case 14u: { return 60u; }\n default: { return 64u; } // case 15u\n }\n}\n\nfn interp4(e0: vec4<i32>, e1: vec4<i32>, w: i32) -> vec4<i32> {\n return ((64 - w) * e0 + w * e1 + vec4<i32>(32)) >> vec4<u32>(6u);\n}\n\nfn to8(v: vec4<f32>) -> vec4<i32> {\n return vec4<i32>(clamp(floor(v * 255.0 + 0.5), vec4<f32>(0.0), vec4<f32>(255.0)));\n}\n\nfn dist2(a: vec4<i32>, b: vec4<i32>) -> i32 {\n let d = a - b;\n let e = d * d;\n return e.x + e.y + e.z + e.w;\n}\n\n// Quantize an 8-bit ideal endpoint to (7-bit value, reconstructed 8-bit) under\n// a fixed p-bit, all four channels at once. q7 = round((ideal8 \u2212 p)/2); used by\n// both paths.\nstruct QuantPair { seven: vec4<i32>, eight: vec4<i32> };\nfn quantize_endpoint(ideal8: vec4<i32>, p: u32) -> QuantPair {\n let q = vec4<i32>(clamp(\n floor((vec4<f32>(ideal8) - f32(p)) / 2.0 + 0.5),\n vec4<f32>(0.0), vec4<f32>(127.0),\n ));\n let eff = (q << vec4<u32>(1u)) | vec4<i32>(i32(p));\n return QuantPair(q, eff);\n}\n\n// ============================ FAST PATH ================================ //\n\n// Endpoint with its chosen p-bit, picked by minimum quantisation error.\nstruct Ep { seven: vec4<i32>, eight: vec4<i32>, p: u32 };\nfn pick_ep(ideal: vec4<i32>) -> Ep {\n let a = quantize_endpoint(ideal, 0u);\n let b = quantize_endpoint(ideal, 1u);\n if (dist2(b.eight, ideal) < dist2(a.eight, ideal)) { return Ep(b.seven, b.eight, 1u); }\n return Ep(a.seven, a.eight, 0u);\n}\n\n// One pass over the block: project every pixel onto the e0\u2192e1 line and\n// accumulate the least-squares normal-equation sums; solve for the refit\n// endpoints (in 8-bit space). Indices are not produced here \u2014 the caller\n// reprojects against the quantised refit endpoints anyway.\nstruct Fit { e0: vec4<i32>, e1: vec4<i32>, valid: bool };\nfn proj_fit(pixels: ptr<function, array<vec4<i32>, 16>>, e0: vec4<i32>, e1: vec4<i32>) -> Fit {\n var out: Fit;\n out.valid = false;\n let dir = vec4<f32>(e1 - e0);\n let dd = dot(dir, dir);\n if (dd == 0.0) { return out; }\n let e0f = vec4<f32>(e0);\n let inv = 15.0 / dd;\n var sAA: f32 = 0.0; var sBB: f32 = 0.0; var sAB: f32 = 0.0;\n var sAV: vec4<f32> = vec4<f32>(0.0); var sBV: vec4<f32> = vec4<f32>(0.0);\n var s_min = 15.0; var s_max = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let v = vec4<f32>((*pixels)[k]);\n let s = clamp(floor(dot(v - e0f, dir) * inv + 0.5), 0.0, 15.0);\n s_min = min(s_min, s); s_max = max(s_max, s);\n let b = s * (1.0 / 15.0); let a = 1.0 - b;\n sAA = sAA + a * a; sBB = sBB + b * b; sAB = sAB + a * b;\n sAV = sAV + a * v; sBV = sBV + b * v;\n }\n // Rank-1 guard: if every pixel projects to ONE level the system is\n // singular \u2014 det and the numerators are pure float rounding noise and the\n // solve returns garbage endpoints. With \u22652 levels det \u2265 15/225 \u2248 0.067.\n if (s_min == s_max) { return out; }\n let det = sAA * sBB - sAB * sAB;\n if (abs(det) < 1e-3) { return out; }\n out.e0 = vec4<i32>(clamp(round((sBB * sAV - sAB * sBV) / det), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.e1 = vec4<i32>(clamp(round((sAA * sBV - sAB * sAV) / det), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.valid = true;\n return out;\n}\n\n// ============================ HIGH PATH ================================ //\n\nstruct Pair { a: vec4<i32>, b: vec4<i32> };\nfn farthest_pair(pixels: ptr<function, array<vec4<i32>, 16>>) -> Pair {\n var best_d: i32 = 0;\n var pa = (*pixels)[0];\n var pb = (*pixels)[1];\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let xi = (*pixels)[i];\n for (var j: u32 = i + 1u; j < 16u; j = j + 1u) {\n let d = dist2(xi, (*pixels)[j]);\n if (d > best_d) { best_d = d; pa = xi; pb = (*pixels)[j]; }\n }\n }\n return Pair(pa, pb);\n}\n\nfn build_palette_6(e0: vec4<i32>, e1: vec4<i32>, pal: ptr<function, array<vec4<i32>, 16>>) {\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n (*pal)[i] = interp4(e0, e1, i32(w4(i)));\n }\n}\n\nfn assign_all(\n pixels: ptr<function, array<vec4<i32>, 16>>,\n pal: ptr<function, array<vec4<i32>, 16>>,\n out_idx: ptr<function, array<u32, 16>>,\n) -> i32 {\n var err: i32 = 0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let px = (*pixels)[k];\n var best_i: u32 = 0u;\n var best_d: i32 = 2147483647;\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let d = dist2(px, (*pal)[i]);\n if (d < best_d) { best_d = d; best_i = i; }\n }\n (*out_idx)[k] = best_i;\n err = err + best_d;\n }\n return err;\n}\n\nstruct BestMode6 {\n e0_7: vec4<i32>, e1_7: vec4<i32>,\n p0: u32, p1: u32,\n indices: array<u32, 16>,\n err: i32,\n};\n\n// Exhaustive p-bit search (high path); commits to `*best` only on improvement.\nfn try_pbit_combos(\n pixels: ptr<function, array<vec4<i32>, 16>>,\n ideal0: vec4<i32>,\n ideal1: vec4<i32>,\n best: ptr<function, BestMode6>,\n) {\n var local_best = (*best).err;\n var pal: array<vec4<i32>, 16>;\n var tmp: array<u32, 16>;\n for (var p0: u32 = 0u; p0 < 2u; p0 = p0 + 1u) {\n let q0 = quantize_endpoint(ideal0, p0);\n for (var p1: u32 = 0u; p1 < 2u; p1 = p1 + 1u) {\n let q1 = quantize_endpoint(ideal1, p1);\n build_palette_6(q0.eight, q1.eight, &pal);\n let err = assign_all(pixels, &pal, &tmp);\n if (err < local_best) {\n local_best = err;\n (*best).e0_7 = q0.seven;\n (*best).e1_7 = q1.seven;\n (*best).p0 = p0;\n (*best).p1 = p1;\n (*best).indices = tmp;\n (*best).err = err;\n }\n }\n }\n}\n\n// Exact-weight LSQ refit (high path); matches bc7_ref.ts `refitEndpointsMode6`.\nstruct RefitResult { e0: vec4<i32>, e1: vec4<i32>, valid: bool };\nfn refit_endpoints(\n pixels: ptr<function, array<vec4<i32>, 16>>,\n indices: ptr<function, array<u32, 16>>,\n) -> RefitResult {\n var sAA: f32 = 0.0;\n var sBB: f32 = 0.0;\n var sAB: f32 = 0.0;\n var sAV: vec4<f32> = vec4<f32>(0.0);\n var sBV: vec4<f32> = vec4<f32>(0.0);\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let i = (*indices)[k];\n let a = f32(64u - w4(i)) / 64.0;\n let b = f32(w4(i)) / 64.0;\n let v = vec4<f32>((*pixels)[k]);\n sAA = sAA + a * a;\n sBB = sBB + b * b;\n sAB = sAB + a * b;\n sAV = sAV + a * v;\n sBV = sBV + b * v;\n }\n let det = sAA * sBB - sAB * sAB;\n var out: RefitResult;\n if (abs(det) < 1e-9) {\n out.valid = false;\n return out;\n }\n let e0f = (sBB * sAV - sAB * sBV) / det;\n let e1f = (sAA * sBV - sAB * sAV) / det;\n out.e0 = vec4<i32>(clamp(round(e0f), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.e1 = vec4<i32>(clamp(round(e1f), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.valid = true;\n return out;\n}\n\n// ------------------------------- Entry --------------------------------- //\n\n@compute @workgroup_size(8, 8, 1)\nfn encode(@builtin(global_invocation_id) gid: vec3<u32>) {\n if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) {\n return;\n }\n\n let bx = gid.x;\n let by = gid.y;\n let block_index = by * params.blocks_x + bx;\n\n let base = vec2<i32>(i32(bx) * 4, i32(by) * 4);\n let max_xy = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);\n\n // Load 16 RGBA pixels (8-bit integer domain) and the per-channel bbox.\n var pixels: array<vec4<i32>, 16>;\n var lo = vec4<i32>(255);\n var hi = vec4<i32>(0);\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let lx = i32(i & 3u);\n let ly = i32(i >> 2u);\n let p = clamp(base + vec2<i32>(lx, ly), vec2<i32>(0, 0), max_xy);\n let px = to8(textureLoad(src_tex, p, 0));\n pixels[i] = px;\n lo = min(lo, px);\n hi = max(hi, px);\n }\n\n // Both branches produce: 7-bit endpoints + p-bits, and the 16 4-bit indices\n // packed LSB-first into two nibble words (pixel k \u2192 bits 4k..4k+3).\n var e0_7: vec4<i32>;\n var e1_7: vec4<i32>;\n var p0: u32;\n var p1: u32;\n var ilo: u32 = 0u;\n var ihi: u32 = 0u;\n\n if (QUALITY_HIGH != 0u) {\n let fp = farthest_pair(&pixels);\n var best: BestMode6;\n best.err = 2147483647;\n try_pbit_combos(&pixels, fp.a, fp.b, &best);\n let refit = refit_endpoints(&pixels, &best.indices);\n if (refit.valid) {\n try_pbit_combos(&pixels, refit.e0, refit.e1, &best);\n }\n e0_7 = best.e0_7; e1_7 = best.e1_7; p0 = best.p0; p1 = best.p1;\n for (var k: u32 = 0u; k < 8u; k = k + 1u) {\n ilo = ilo | (best.indices[k] << (k * 4u));\n }\n for (var k: u32 = 8u; k < 16u; k = k + 1u) {\n ihi = ihi | (best.indices[k] << ((k - 8u) * 4u));\n }\n } else {\n // Seed the fused LSQ fit from the raw bbox, then quantise the refit\n // endpoints and reproject for the final indices.\n // The refit is clamped to the block bbox: on multi-cluster blocks (a hard\n // edge through two-colour noise) the unconstrained solve extrapolates far\n // outside the block's colours and the per-channel [0,255] clamp then bends\n // the hue \u2014 fringe pixels decode to colours that exist nowhere in the\n // block. Constraining to the bbox also measures BETTER in plain SSE\n // (+1.3 dB on the colour test card): the wild endpoints were losing more\n // after quantisation + reassignment than the extrapolation ever bought.\n let r = proj_fit(&pixels, lo, hi);\n var ep0: Ep;\n var ep1: Ep;\n if (r.valid) { ep0 = pick_ep(clamp(r.e0, lo, hi)); ep1 = pick_ep(clamp(r.e1, lo, hi)); }\n else { ep0 = pick_ep(lo); ep1 = pick_ep(hi); }\n let dir = vec4<f32>(ep1.eight - ep0.eight);\n let dd = dot(dir, dir);\n if (dd > 0.0) {\n let e0f = vec4<f32>(ep0.eight);\n let inv = 15.0 / dd;\n for (var k: u32 = 0u; k < 8u; k = k + 1u) {\n let s = clamp(floor(dot(vec4<f32>(pixels[k]) - e0f, dir) * inv + 0.5), 0.0, 15.0);\n ilo = ilo | (u32(s) << (k * 4u));\n }\n for (var k: u32 = 8u; k < 16u; k = k + 1u) {\n let s = clamp(floor(dot(vec4<f32>(pixels[k]) - e0f, dir) * inv + 0.5), 0.0, 15.0);\n ihi = ihi | (u32(s) << ((k - 8u) * 4u));\n }\n }\n e0_7 = ep0.seven; e1_7 = ep1.seven; p0 = ep0.p; p1 = ep1.p;\n }\n\n // Anchor rule \u2014 pixel 0's index MSB must be 0. Swapping endpoints reflects\n // every index (i \u2192 15\u2212i), which on packed nibbles is a bitwise NOT.\n if ((ilo & 0x8u) != 0u) {\n let t7 = e0_7; e0_7 = e1_7; e1_7 = t7;\n let tp = p0; p0 = p1; p1 = tp;\n ilo = ~ilo; ihi = ~ihi;\n }\n\n // Straight-line mode-6 packing (see layout at the top of the file).\n let e0 = vec4<u32>(e0_7);\n let e1 = vec4<u32>(e1_7);\n let w0 = 0x40u | (e0.x << 7u) | (e1.x << 14u) | (e0.y << 21u) | (e1.y << 28u);\n let w1 = (e1.y >> 4u) | (e0.z << 3u) | (e1.z << 10u) | (e0.w << 17u) | (e1.w << 24u) | (p0 << 31u);\n let w2 = p1 | ((ilo & 0x7u) << 1u) | (ilo & 0xFFFFFFF0u);\n let w3 = ihi;\n\n let out = block_index * 4u;\n dst[out + 0u] = w0;\n dst[out + 1u] = w1;\n dst[out + 2u] = w2;\n dst[out + 3u] = w3;\n}\n";
|
|
821
|
+
var bc7_default = "// BC7 (BPTC) mode 6 compute shader encoder.\n//\n// One invocation per 4\xD74 block. Emits 16 bytes = 4 u32s into the storage\n// buffer at `dst[block_index * 4 .. + 3]`. This is the f32 fallback;\n// bc7_fast_f16.wgsl is the same algorithm and is preferred when the device\n// reports shader-f16.\n//\n// ALGORITHM: principal-axis seed (covariance power-iteration; bbox on\n// degenerate blocks) at the exact projection extents, quantised directly \u2014\n// no LSQ refit; with the seed on the principal axis, mode 6's 16-level\n// palette leaves the refit under 0.15 dB, unlike the 4-level BC1/ASTC\n// encoders which keep theirs \u2014 then one pass that projects each pixel onto\n// the endpoint line (the 16 palette entries are colinear, so the nearest\n// index is the rounded projection \u2014 no palette build, no 16-entry search),\n// packed on the fly into two nibble words.\n//\n// MODE 6 LAYOUT (LSB-first, bit 0 = byte 0's bit 0)\n// bits 0..6 mode field (0b0000001 \u2014 only bit 6 is 1)\n// bits 7..13 R0 (7-bit) bits 14..20 R1 bits 21..27 G0 bits 28..34 G1\n// bits 35..41 B0 bits 42..48 B1 bits 49..55 A0 bits 56..62 A1\n// bit 63 P0 bit 64 P1\n// bits 65..67 pixel 0 index (3 bits; anchor, MSB implicit 0)\n// bits 68..71 pixel 1 index (4 bits) ... bits 124..127 pixel 15 index\n//\n// Effective 8-bit endpoint channel = (7_bit_value << 1) | p_bit.\n// Palette[i] = ((64 \u2212 W4[i]) \xD7 e0_8 + W4[i] \xD7 e1_8 + 32) >> 6, integer.\n//\n// The block is assembled with straight-line constant shifts (see the layout\n// summary in bc7_fast_f16.wgsl) \u2014 a generic write_bits() helper's dynamic\n// word indexing keeps the output array out of registers.\n\nstruct Params {\n blocks_x: u32,\n blocks_y: u32,\n width: u32,\n height: u32,\n};\n\n@group(0) @binding(0) var src_tex: texture_2d<f32>;\n@group(0) @binding(1) var<storage, read_write> dst: array<u32>;\n@group(0) @binding(2) var<uniform> params: Params;\n\nfn to8(v: vec4<f32>) -> vec4<i32> {\n return vec4<i32>(clamp(floor(v * 255.0 + 0.5), vec4<f32>(0.0), vec4<f32>(255.0)));\n}\n\nfn dist2(a: vec4<i32>, b: vec4<i32>) -> i32 {\n let d = a - b;\n let e = d * d;\n return e.x + e.y + e.z + e.w;\n}\n\n// Quantize an 8-bit ideal endpoint to (7-bit value, reconstructed 8-bit)\n// under a fixed p-bit, all four channels at once. q7 = round((ideal8 \u2212 p)/2).\nstruct QuantPair { seven: vec4<i32>, eight: vec4<i32> };\nfn quantize_endpoint(ideal8: vec4<i32>, p: u32) -> QuantPair {\n let q = vec4<i32>(clamp(\n floor((vec4<f32>(ideal8) - f32(p)) / 2.0 + 0.5),\n vec4<f32>(0.0), vec4<f32>(127.0),\n ));\n let eff = (q << vec4<u32>(1u)) | vec4<i32>(i32(p));\n return QuantPair(q, eff);\n}\n\n// Endpoint with its chosen p-bit, picked by minimum quantisation error.\nstruct Ep { seven: vec4<i32>, eight: vec4<i32>, p: u32 };\nfn pick_ep(ideal: vec4<i32>) -> Ep {\n let a = quantize_endpoint(ideal, 0u);\n let b = quantize_endpoint(ideal, 1u);\n if (dist2(b.eight, ideal) < dist2(a.eight, ideal)) { return Ep(b.seven, b.eight, 1u); }\n return Ep(a.seven, a.eight, 0u);\n}\n\n// Principal colour axis via power-iteration over precomputed, mean-corrected\n// covariance rows (the moments are accumulated for free in the pixel-load\n// loop), seeded with the bbox diagonal. Returns a unit axis, or vec4(0) for\n// a degenerate (constant) block. Same family as bc1.wgsl's principal_axis \u2014\n// the bbox diagonal alone is sign-blind and points across anti-correlated\n// data (normal maps, hue edges) instead of along it.\nfn principal_axis4(\n c0v: vec4<f32>,\n c1v: vec4<f32>,\n c2v: vec4<f32>,\n c3v: vec4<f32>,\n seed: vec4<f32>,\n) -> vec4<f32> {\n var v = seed;\n var len = length(v);\n if (len < 1e-9) { return vec4<f32>(0.0); }\n v = v / len;\n for (var iter: u32 = 0u; iter < 8u; iter = iter + 1u) {\n let nv = vec4<f32>(dot(c0v, v), dot(c1v, v), dot(c2v, v), dot(c3v, v));\n len = length(nv);\n if (len < 1e-12) { return vec4<f32>(0.0); }\n v = nv / len;\n }\n return v;\n}\n\n// ------------------------------- Entry --------------------------------- //\n\n@compute @workgroup_size(8, 8, 1)\nfn encode(@builtin(global_invocation_id) gid: vec3<u32>) {\n if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) {\n return;\n }\n\n let bx = gid.x;\n let by = gid.y;\n let block_index = by * params.blocks_x + bx;\n\n let base = vec2<i32>(i32(bx) * 4, i32(by) * 4);\n let max_xy = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);\n\n // Load 16 RGBA pixels (8-bit integer domain) and the per-channel bbox,\n // with the covariance moments FUSED in: d = px \u2212 pixel0 (first-pixel-\n // relative, so the sums scale with the block's span; d is integer-valued\n // and \u2264255, exact in f32).\n var pixels: array<vec4<i32>, 16>;\n var lo = vec4<i32>(255);\n var hi = vec4<i32>(0);\n var p0f = vec4<f32>(0.0);\n var sd = vec4<f32>(0.0);\n var c0v = vec4<f32>(0.0);\n var c1v = vec4<f32>(0.0);\n var c2v = vec4<f32>(0.0);\n var c3v = vec4<f32>(0.0);\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let lx = i32(i & 3u);\n let ly = i32(i >> 2u);\n let p = clamp(base + vec2<i32>(lx, ly), vec2<i32>(0, 0), max_xy);\n let px = to8(textureLoad(src_tex, p, 0));\n pixels[i] = px;\n lo = min(lo, px);\n hi = max(hi, px);\n if (i == 0u) { p0f = vec4<f32>(px); }\n let d = vec4<f32>(px) - p0f;\n sd = sd + d;\n c0v = c0v + d.x * d;\n c1v = c1v + d.y * d;\n c2v = c2v + d.z * d;\n c3v = c3v + d.w * d;\n }\n let mean = p0f + sd * (1.0 / 16.0);\n\n // Seed endpoints from the block's principal colour axis at the exact\n // projection extents (see header), quantise, and assign indices in one\n // projection pass.\n // Mean-correct the fused moments: C = \u03A3dd\u1D40 \u2212 (\u03A3d)(\u03A3d)\u1D40/16.\n let sd16 = sd * (1.0 / 16.0);\n let r0v = c0v - sd.x * sd16;\n let r1v = c1v - sd.y * sd16;\n let r2v = c2v - sd.z * sd16;\n let r3v = c3v - sd.w * sd16;\n var seed0 = lo;\n var seed1 = hi;\n let axis = principal_axis4(r0v, r1v, r2v, r3v, vec4<f32>(hi - lo));\n if (dot(axis, axis) > 0.0) {\n // Exact projection extents along the axis. (A Rayleigh-quotient span\n // estimate was tried in place of this pass \u2014 it saves 16 dots but\n // costs 0.1\u20130.8 dB and 4\u201310\xD7 on the worst-easy-block gate: \u03C3\n // misjudges two-cluster and outlier blocks. The pass stays.)\n var t_min: f32 = 1e30;\n var t_max: f32 = -1e30;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let t = dot(vec4<f32>(pixels[k]) - mean, axis);\n t_min = min(t_min, t);\n t_max = max(t_max, t);\n }\n seed0 = vec4<i32>(clamp(round(mean + t_min * axis), vec4<f32>(0.0), vec4<f32>(255.0)));\n seed1 = vec4<i32>(clamp(round(mean + t_max * axis), vec4<f32>(0.0), vec4<f32>(255.0)));\n }\n\n // The 16 4-bit indices, packed LSB-first into two nibble words\n // (pixel k \u2192 bits 4k..4k+3).\n var ilo: u32 = 0u;\n var ihi: u32 = 0u;\n var ep0 = pick_ep(seed0);\n var ep1 = pick_ep(seed1);\n let dir = vec4<f32>(ep1.eight - ep0.eight);\n let dd = dot(dir, dir);\n if (dd > 0.0) {\n let e0f = vec4<f32>(ep0.eight);\n let inv = 15.0 / dd;\n for (var k: u32 = 0u; k < 8u; k = k + 1u) {\n let s = clamp(floor(dot(vec4<f32>(pixels[k]) - e0f, dir) * inv + 0.5), 0.0, 15.0);\n ilo = ilo | (u32(s) << (k * 4u));\n }\n for (var k: u32 = 8u; k < 16u; k = k + 1u) {\n let s = clamp(floor(dot(vec4<f32>(pixels[k]) - e0f, dir) * inv + 0.5), 0.0, 15.0);\n ihi = ihi | (u32(s) << ((k - 8u) * 4u));\n }\n }\n var e0_7 = ep0.seven;\n var e1_7 = ep1.seven;\n var p0 = ep0.p;\n var p1 = ep1.p;\n\n // Anchor rule \u2014 pixel 0's index MSB must be 0. Swapping endpoints reflects\n // every index (i \u2192 15\u2212i), which on packed nibbles is a bitwise NOT.\n if ((ilo & 0x8u) != 0u) {\n let t7 = e0_7; e0_7 = e1_7; e1_7 = t7;\n let tp = p0; p0 = p1; p1 = tp;\n ilo = ~ilo; ihi = ~ihi;\n }\n\n // Straight-line mode-6 packing (see layout at the top of the file).\n let e0 = vec4<u32>(e0_7);\n let e1 = vec4<u32>(e1_7);\n let w0 = 0x40u | (e0.x << 7u) | (e1.x << 14u) | (e0.y << 21u) | (e1.y << 28u);\n let w1 = (e1.y >> 4u) | (e0.z << 3u) | (e1.z << 10u) | (e0.w << 17u) | (e1.w << 24u) | (p0 << 31u);\n let w2 = p1 | ((ilo & 0x7u) << 1u) | (ilo & 0xFFFFFFF0u);\n let w3 = ihi;\n\n let out = block_index * 4u;\n dst[out + 0u] = w0;\n dst[out + 1u] = w1;\n dst[out + 2u] = w2;\n dst[out + 3u] = w3;\n}\n";
|
|
642
822
|
|
|
643
823
|
// src/bc7_fast_f16.wgsl
|
|
644
824
|
var bc7_fast_f16_default = `// bc7 "fast" encoder \u2014 f16 variant (requires the shader-f16 feature).
|
|
645
|
-
// Same algorithm family as the f32 fast path in bc7.wgsl (
|
|
646
|
-
//
|
|
647
|
-
//
|
|
825
|
+
// Same algorithm family as the f32 fast path in bc7.wgsl (principal-axis
|
|
826
|
+
// seed at the exact projection extents \u2192 quantise \u2192 one projection-based
|
|
827
|
+
// index-assignment pass), tuned for throughput:
|
|
648
828
|
//
|
|
649
|
-
// \u2022 All projection
|
|
650
|
-
//
|
|
829
|
+
// \u2022 All projection math in f16 ([0,1] domain). ~2\xD7 ALU throughput on
|
|
830
|
+
// f16-capable GPUs. The projection direction is pre-scaled by 32:
|
|
651
831
|
// a shallow block (endpoints ~1/255 apart) has dd = dot(dir,dir) \u2248 1.5e-5,
|
|
652
832
|
// where 15/dd \u2248 10\u2076 overflows f16 (max 65504) to +inf and the products
|
|
653
|
-
// inside the projection dot are subnormal \u2014 the indices
|
|
654
|
-
//
|
|
655
|
-
//
|
|
656
|
-
//
|
|
657
|
-
//
|
|
658
|
-
// \u2022
|
|
659
|
-
//
|
|
660
|
-
//
|
|
833
|
+
// inside the projection dot are subnormal \u2014 the indices turn to garbage
|
|
834
|
+
// (visible as banding on smooth gradients). Scaling dir by 32 multiplies
|
|
835
|
+
// the dots by 32 and dd by 1024; s = dot\xB7(32\xB715/dd\u2083\u2082) is the same
|
|
836
|
+
// quantity with every intermediate in f16's normal range (worst case
|
|
837
|
+
// inv = 480/0.0157 \u2248 3.0e4 < 65504).
|
|
838
|
+
// \u2022 NO least-squares refit, unlike the BC1/BC5/ASTC fast paths: with the
|
|
839
|
+
// seed already on the principal axis at the exact projection extents,
|
|
840
|
+
// mode 6's fine 16-level palette leaves the refit \u22640.05 dB on the colour
|
|
841
|
+
// card and \u22640.15 dB on the normal card \u2014 not worth its two extra
|
|
842
|
+
// 16-pixel passes. The coarse 4-level formats DO need it (dropping it
|
|
843
|
+
// there costs 0.5\u20131.3 dB).
|
|
661
844
|
// \u2022 Indices are packed into two u32 nibble words ON THE FLY during the
|
|
662
|
-
//
|
|
845
|
+
// projection pass \u2014 no array<u32,16> private array. The BC7 anchor
|
|
663
846
|
// reflection (i \u2192 15\u2212i) is then just a bitwise NOT of both words.
|
|
664
847
|
// \u2022 The 128-bit block is assembled with straight-line constant shifts
|
|
665
848
|
// instead of a generic write_bits() helper (whose dynamic word indexing
|
|
@@ -696,53 +879,6 @@ fn pick_ep(ideal01: h4) -> Ep {
|
|
|
696
879
|
return Ep(vec4<u32>(q0), e0 * h(1.0 / 255.0), 0u);
|
|
697
880
|
}
|
|
698
881
|
|
|
699
|
-
// One pass over the block: project every pixel onto the e0\u2192e1 line and
|
|
700
|
-
// accumulate the least-squares normal-equation sums; solve for the refit
|
|
701
|
-
// endpoints. Indices are NOT produced here \u2014 the caller reprojects against
|
|
702
|
-
// the quantised refit endpoints anyway.
|
|
703
|
-
//
|
|
704
|
-
// The value sums accumulate v \u2212 e0, not v: the basis is affine (a + b = 1),
|
|
705
|
-
// so fitting the shifted data and adding e0 back is the same fit, but the
|
|
706
|
-
// accumulators scale with the block's span instead of its absolute level \u2014
|
|
707
|
-
// on a shallow dark block, f16 rounding of absolute sums (ulp \u2248 0.12 of an
|
|
708
|
-
// 8-bit level per add) drifts the refit endpoints by \xB11 level.
|
|
709
|
-
struct Fit { e0: h4, e1: h4, valid: bool };
|
|
710
|
-
fn proj_fit(pix: ptr<function, array<h4, 16>>, e0: h4, e1: h4) -> Fit {
|
|
711
|
-
var out: Fit;
|
|
712
|
-
out.valid = false;
|
|
713
|
-
// dir pre-scaled by 32 to keep dd and the projection dots in f16's normal
|
|
714
|
-
// range (see header). Spans below ~0.7 of an 8-bit step (dd\u2083\u2082 < 0.008,
|
|
715
|
-
// possible only for non-8-bit sources) are treated as flat \u2014 encoding them
|
|
716
|
-
// flat is under half a level of error, while running the math on them risks
|
|
717
|
-
// inv overflowing to +inf.
|
|
718
|
-
let dir = (e1 - e0) * h(32.0);
|
|
719
|
-
let dd = dot(dir, dir);
|
|
720
|
-
if (dd < h(0.008)) { return out; }
|
|
721
|
-
let inv = h(480.0) / dd; // 32\xB715/dd\u2083\u2082 \u2261 15/dd
|
|
722
|
-
var sAA = h(0.0); var sBB = h(0.0); var sAB = h(0.0);
|
|
723
|
-
var sAV = h4(0.0); var sBV = h4(0.0);
|
|
724
|
-
var s_min = h(15.0); var s_max = h(0.0);
|
|
725
|
-
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
726
|
-
let vr = (*pix)[k] - e0;
|
|
727
|
-
let s = clamp(floor(dot(vr, dir) * inv + h(0.5)), h(0.0), h(15.0));
|
|
728
|
-
s_min = min(s_min, s); s_max = max(s_max, s);
|
|
729
|
-
let b = s * h(1.0 / 15.0); let a = h(1.0) - b;
|
|
730
|
-
sAA = sAA + a * a; sBB = sBB + b * b; sAB = sAB + a * b;
|
|
731
|
-
sAV = sAV + a * vr; sBV = sBV + b * vr;
|
|
732
|
-
}
|
|
733
|
-
// Rank-1 guard: if every pixel projects to ONE level the system is
|
|
734
|
-
// singular \u2014 det/numerators are pure f16 rounding noise and the solve
|
|
735
|
-
// returns garbage endpoints. With \u22652 distinct levels
|
|
736
|
-
// det = \u03A3_i<j (b_j \u2212 b_i)\xB2 \u2265 15/225 \u2248 0.067, so 0.02 is a safe floor.
|
|
737
|
-
if (s_min == s_max) { return out; }
|
|
738
|
-
let det = sAA * sBB - sAB * sAB;
|
|
739
|
-
if (abs(det) < h(0.02)) { return out; }
|
|
740
|
-
out.e0 = clamp(e0 + (sBB * sAV - sAB * sBV) / det, h4(0.0), h4(1.0));
|
|
741
|
-
out.e1 = clamp(e0 + (sAA * sBV - sAB * sAV) / det, h4(0.0), h4(1.0));
|
|
742
|
-
out.valid = true;
|
|
743
|
-
return out;
|
|
744
|
-
}
|
|
745
|
-
|
|
746
882
|
@compute @workgroup_size(8, 8, 1)
|
|
747
883
|
fn encode(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
748
884
|
if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) { return; }
|
|
@@ -750,26 +886,88 @@ fn encode(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
|
750
886
|
let base = vec2<i32>(i32(gid.x) * 4, i32(gid.y) * 4);
|
|
751
887
|
let mx = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);
|
|
752
888
|
|
|
889
|
+
// Load pass, with the covariance moments FUSED in (no separate 16-pixel
|
|
890
|
+
// pass): d = (px \u2212 pixel0)\xB716, relative to the block's first pixel so the
|
|
891
|
+
// accumulators scale with the block's span \u2014 raw \u03A3v\xB7v\u1D40 moments would
|
|
892
|
+
// cancel catastrophically in f16 \u2014 and pre-scaled \xD716 so shallow blocks
|
|
893
|
+
// (span ~1/255 \u2192 d\xB2 \u2248 1e-3) clear the subnormal floor while full-range
|
|
894
|
+
// sums stay \u22644096. C = \u03A3dd\u1D40 \u2212 (\u03A3d)(\u03A3d)\u1D40/16 is the \xD7256-scaled covariance.
|
|
753
895
|
var pix: array<h4, 16>;
|
|
754
896
|
var lo = h4(1.0);
|
|
755
897
|
var hi = h4(0.0);
|
|
898
|
+
var p0v = h4(0.0);
|
|
899
|
+
var sd = h4(0.0);
|
|
900
|
+
var c0v = h4(0.0);
|
|
901
|
+
var c1v = h4(0.0);
|
|
902
|
+
var c2v = h4(0.0);
|
|
903
|
+
var c3v = h4(0.0);
|
|
756
904
|
for (var i: u32 = 0u; i < 16u; i = i + 1u) {
|
|
757
905
|
let p = clamp(base + vec2<i32>(i32(i & 3u), i32(i >> 2u)), vec2<i32>(0), mx);
|
|
758
906
|
let px = h4(textureLoad(src_tex, p, 0));
|
|
759
907
|
pix[i] = px; lo = min(lo, px); hi = max(hi, px);
|
|
908
|
+
if (i == 0u) { p0v = px; }
|
|
909
|
+
let d = (px - p0v) * h(16.0);
|
|
910
|
+
sd = sd + d;
|
|
911
|
+
c0v = c0v + d.x * d;
|
|
912
|
+
c1v = c1v + d.y * d;
|
|
913
|
+
c2v = c2v + d.z * d;
|
|
914
|
+
c3v = c3v + d.w * d;
|
|
915
|
+
}
|
|
916
|
+
let mean = p0v + sd * h(1.0 / 256.0);
|
|
917
|
+
// Mean-correction via sd4\xB7sd4\u1D40 with sd4 = \u03A3d/4: (\u03A3d)(\u03A3d)\u1D40/16 with every
|
|
918
|
+
// product \u22644096 (a direct \u03A3d\xB7\u03A3d\u1D40 could hit 65536 and overflow f16).
|
|
919
|
+
let sd4 = sd * h(0.25);
|
|
920
|
+
c0v = c0v - sd4.x * sd4;
|
|
921
|
+
c1v = c1v - sd4.y * sd4;
|
|
922
|
+
c2v = c2v - sd4.z * sd4;
|
|
923
|
+
c3v = c3v - sd4.w * sd4;
|
|
924
|
+
|
|
925
|
+
// Seed endpoints from the block's principal colour axis (covariance
|
|
926
|
+
// power-iteration, seeded with the bbox diagonal \u2014 same family as the BC1
|
|
927
|
+
// 'high' path). The bbox diagonal is sign-blind: on anti-correlated
|
|
928
|
+
// channels (normal maps, hue edges) it points across the data instead of
|
|
929
|
+
// along it, and the LSQ refit \u2014 which fits endpoints GIVEN the projection
|
|
930
|
+
// indices \u2014 can't recover from a wrong axis. The iteration renormalises by
|
|
931
|
+
// the max component (a plain length() of the matvec output could overflow
|
|
932
|
+
// f16), so only the direction survives.
|
|
933
|
+
var seed_lo = lo;
|
|
934
|
+
var seed_hi = hi;
|
|
935
|
+
var axis = hi - lo;
|
|
936
|
+
var axis_ok = true;
|
|
937
|
+
for (var it: u32 = 0u; it < 4u; it = it + 1u) {
|
|
938
|
+
let nv = h4(dot(c0v, axis), dot(c1v, axis), dot(c2v, axis), dot(c3v, axis));
|
|
939
|
+
let m = max(max(abs(nv.x), abs(nv.y)), max(abs(nv.z), abs(nv.w)));
|
|
940
|
+
if (m < h(1e-4)) { axis_ok = false; break; }
|
|
941
|
+
axis = nv / m;
|
|
942
|
+
}
|
|
943
|
+
if (axis_ok) {
|
|
944
|
+
axis = axis / length(axis);
|
|
945
|
+
// Exact projection extents along the axis. (A Rayleigh-quotient span
|
|
946
|
+
// estimate was tried in place of this pass \u2014 it saves 16 dots but costs
|
|
947
|
+
// 0.1\u20130.8 dB and 4\u201310\xD7 on the worst-easy-block gate: \u03C3 misjudges
|
|
948
|
+
// two-cluster and outlier blocks and the quantised weight grid can't
|
|
949
|
+
// recover. The pass stays.)
|
|
950
|
+
var t_min = h(4.0);
|
|
951
|
+
var t_max = h(-4.0);
|
|
952
|
+
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
953
|
+
let t = dot(pix[k] - mean, axis);
|
|
954
|
+
t_min = min(t_min, t);
|
|
955
|
+
t_max = max(t_max, t);
|
|
956
|
+
}
|
|
957
|
+
seed_lo = clamp(mean + t_min * axis, h4(0.0), h4(1.0));
|
|
958
|
+
seed_hi = clamp(mean + t_max * axis, h4(0.0), h4(1.0));
|
|
760
959
|
}
|
|
761
960
|
|
|
762
|
-
//
|
|
763
|
-
// is clamped to the block bbox: on
|
|
764
|
-
//
|
|
765
|
-
// [0,1] clamp then bends the hue \u2014
|
|
766
|
-
// exist nowhere in the block.
|
|
767
|
-
// in plain SSE (+1.3 dB on
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
var
|
|
771
|
-
|
|
772
|
-
else { ep0 = pick_ep(lo); ep1 = pick_ep(hi); }
|
|
961
|
+
// Fit from the principal-axis seed (bbox on degenerate blocks), then
|
|
962
|
+
// quantise the refit endpoints. The refit is clamped to the block bbox: on
|
|
963
|
+
// multi-cluster blocks the unconstrained solve extrapolates far outside
|
|
964
|
+
// the block's colours and the per-channel [0,1] clamp then bends the hue \u2014
|
|
965
|
+
// fringe pixels decode to colours that exist nowhere in the block.
|
|
966
|
+
// Constraining to the bbox also measures better in plain SSE (+1.3 dB on
|
|
967
|
+
// the colour test card).
|
|
968
|
+
// Quantise the PCA-extents seed directly \u2014 no LSQ refit (see header).
|
|
969
|
+
var ep0 = pick_ep(seed_lo);
|
|
970
|
+
var ep1 = pick_ep(seed_hi);
|
|
773
971
|
|
|
774
972
|
// Final projection against the decoded endpoints, packing the 4-bit indices
|
|
775
973
|
// into two nibble words as we go (pixel k \u2192 bits 4k..4k+3 of ilo/ihi).
|
|
@@ -825,9 +1023,6 @@ var BC7Encoder = class extends Encoder {
|
|
|
825
1023
|
get supportsSrgb() {
|
|
826
1024
|
return true;
|
|
827
1025
|
}
|
|
828
|
-
get supportsQuality() {
|
|
829
|
-
return true;
|
|
830
|
-
}
|
|
831
1026
|
wgslSource() {
|
|
832
1027
|
return bc7_default;
|
|
833
1028
|
}
|
|
@@ -840,12 +1035,12 @@ var BC7Encoder = class extends Encoder {
|
|
|
840
1035
|
};
|
|
841
1036
|
|
|
842
1037
|
// src/astc4x4.wgsl
|
|
843
|
-
var astc4x4_default = "// ASTC 4\xD74 LDR compute shader encoder.\n//\n// One invocation per 4\xD74 block. Emits 16 bytes = 4 u32s into the storage\n// buffer at `dst[block_index * 4 .. + 3]`.\n//\n// QUALITY LEVELS (pipeline-overridable constant `QUALITY_HIGH`)\n// fast (0, default): O(N) bounding-box seed \u2192 one fused pass that projects\n// each pixel onto the endpoint line (the 4 palette entries are colinear,\n// so the nearest is the rounded projection \u2014 no per-entry search) while\n// accumulating the least-squares refit sums, then a reprojection against\n// the quantised refit endpoints with the weights packed on the fly. The\n// endpoint ordering rule is applied before the weight pass, so no\n// reflection is needed.\n// high (1): O(N\xB2) farthest-pair seed, full 4-entry nearest search, one LSQ\n// refit \u2014 matches astc4x4_ref.ts up to FP tie-breaks.\n// The fast branch is selected at pipeline-compile time; the driver eliminates\n// the unused (high) code.\n//\n// RESTRICTED SUBSET: single partition, no dual-plane, CEM 12 (LDR RGBA direct),\n// 4\xD74 weight grid with 2-bit weights (QUANT_4), 8-bit endpoints (QUANT_256).\n//\n// BLOCK LAYOUT (128 bits, LSB-first)\n// bits [10:0] block mode = 0x042\n// bits [12:11] partition count \u2212 1 = 0\n// bits [16:13] CEM = 12\n// bits [80:17] endpoints: R0 R1 G0 G1 B0 B1 A0 A1 (8-bit each)\n// bits [127:96] 16 \xD7 2-bit weights; weight k: bit(127\u22122k)=lsb, bit(126\u22122k)=msb\n//\n// ENDPOINT ORDERING: if sum(e0.rgb) > sum(e1.rgb) swap endpoints and reflect\n// indices (w' = 3 \u2212 w) to keep the decoder out of blue contraction.\n\n// 0 = fast (default), 1 = exhaustive/high-quality. Set via pipeline constants.\noverride QUALITY_HIGH: u32 = 0u;\n\nstruct Params {\n blocks_x: u32,\n blocks_y: u32,\n width: u32,\n height: u32,\n};\n\n@group(0) @binding(0) var src_tex: texture_2d<f32>;\n@group(0) @binding(1) var<storage, read_write> dst: array<u32>;\n@group(0) @binding(2) var<uniform> params: Params;\n\nfn weight_unq(i: u32) -> i32 {\n switch i {\n case 0u: { return 0; }\n case 1u: { return 21; }\n case 2u: { return 43; }\n default: { return 64; } // case 3u\n }\n}\n\nfn interp4(e0: vec4<i32>, e1: vec4<i32>, w: i32) -> vec4<i32> {\n return ((64 - w) * e0 + w * e1 + vec4<i32>(32)) >> vec4<u32>(6u);\n}\n\nfn to8(v: vec4<f32>) -> vec4<i32> {\n return vec4<i32>(clamp(floor(v * 255.0 + 0.5), vec4<f32>(0.0), vec4<f32>(255.0)));\n}\n\nfn dist2(a: vec4<i32>, b: vec4<i32>) -> i32 {\n let d = a - b;\n let e = d * d;\n return e.x + e.y + e.z + e.w;\n}\n\n// ============================ FAST PATH ================================ //\n\n// One pass over the block: project every pixel onto the e0\u2192e1 line (4 levels,\n// QUANT_4 \u2248 thirds) and accumulate the least-squares normal-equation sums;\n// solve for the refit endpoints. Weights are not produced here \u2014 the caller\n// reprojects against the quantised refit endpoints anyway.\nstruct Fit { e0: vec4<i32>, e1: vec4<i32>, valid: bool };\nfn proj_fit(pixels: ptr<function, array<vec4<i32>, 16>>, e0: vec4<i32>, e1: vec4<i32>) -> Fit {\n var out: Fit;\n out.valid = false;\n let dir = vec4<f32>(e1 - e0);\n let dd = dot(dir, dir);\n if (dd == 0.0) { return out; }\n let e0f = vec4<f32>(e0);\n let inv = 3.0 / dd;\n var sAA: f32 = 0.0; var sBB: f32 = 0.0; var sAB: f32 = 0.0;\n var sAV: vec4<f32> = vec4<f32>(0.0); var sBV: vec4<f32> = vec4<f32>(0.0);\n var s_min = 3.0; var s_max = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let v = vec4<f32>((*pixels)[k]);\n let s = clamp(floor(dot(v - e0f, dir) * inv + 0.5), 0.0, 3.0);\n s_min = min(s_min, s); s_max = max(s_max, s);\n let b = s * (1.0 / 3.0); let a = 1.0 - b;\n sAA = sAA + a * a; sBB = sBB + b * b; sAB = sAB + a * b;\n sAV = sAV + a * v; sBV = sBV + b * v;\n }\n // Rank-1 guard: if every pixel projects to ONE level the system is\n // singular \u2014 det and the numerators are pure float rounding noise and the\n // solve returns garbage endpoints. With \u22652 levels det \u2265 15\xB7(1/3)\xB2 \u2248 1.67.\n if (s_min == s_max) { return out; }\n let det = sAA * sBB - sAB * sAB;\n if (abs(det) < 1e-3) { return out; }\n out.e0 = vec4<i32>(clamp(round((sBB * sAV - sAB * sBV) / det), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.e1 = vec4<i32>(clamp(round((sAA * sBV - sAB * sAV) / det), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.valid = true;\n return out;\n}\n\n// ============================ HIGH PATH ================================ //\n\nstruct Pair { a: vec4<i32>, b: vec4<i32> };\nfn farthest_pair(pixels: ptr<function, array<vec4<i32>, 16>>) -> Pair {\n var best_d: i32 = 0;\n var pa = (*pixels)[0];\n var pb = (*pixels)[1];\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let xi = (*pixels)[i];\n for (var j: u32 = i + 1u; j < 16u; j = j + 1u) {\n let d = dist2(xi, (*pixels)[j]);\n if (d > best_d) { best_d = d; pa = xi; pb = (*pixels)[j]; }\n }\n }\n return Pair(pa, pb);\n}\n\nfn build_palette(e0: vec4<i32>, e1: vec4<i32>, pal: ptr<function, array<vec4<i32>, 4>>) {\n for (var i: u32 = 0u; i < 4u; i = i + 1u) {\n (*pal)[i] = interp4(e0, e1, weight_unq(i));\n }\n}\n\nfn assign_all(\n pixels: ptr<function, array<vec4<i32>, 16>>,\n pal: ptr<function, array<vec4<i32>, 4>>,\n out_idx: ptr<function, array<u32, 16>>,\n) -> i32 {\n var err: i32 = 0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let px = (*pixels)[k];\n var best_i: u32 = 0u;\n var best_d: i32 = 2147483647;\n for (var i: u32 = 0u; i < 4u; i = i + 1u) {\n let d = dist2(px, (*pal)[i]);\n if (d < best_d) { best_d = d; best_i = i; }\n }\n (*out_idx)[k] = best_i;\n err = err + best_d;\n }\n return err;\n}\n\nstruct RefitResult { e0: vec4<i32>, e1: vec4<i32>, valid: bool };\nfn refit_endpoints(\n pixels: ptr<function, array<vec4<i32>, 16>>,\n indices: ptr<function, array<u32, 16>>,\n) -> RefitResult {\n var sAA: f32 = 0.0;\n var sBB: f32 = 0.0;\n var sAB: f32 = 0.0;\n var sAV: vec4<f32> = vec4<f32>(0.0);\n var sBV: vec4<f32> = vec4<f32>(0.0);\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let unq = weight_unq((*indices)[k]);\n let a = f32(64 - unq) / 64.0;\n let b = f32(unq) / 64.0;\n let v = vec4<f32>((*pixels)[k]);\n sAA = sAA + a * a;\n sBB = sBB + b * b;\n sAB = sAB + a * b;\n sAV = sAV + a * v;\n sBV = sBV + b * v;\n }\n let det = sAA * sBB - sAB * sAB;\n var out: RefitResult;\n if (abs(det) < 1e-9) {\n out.valid = false;\n return out;\n }\n let e0f = (sBB * sAV - sAB * sBV) / det;\n let e1f = (sAA * sBV - sAB * sAV) / det;\n out.e0 = vec4<i32>(clamp(round(e0f), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.e1 = vec4<i32>(clamp(round(e1f), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.valid = true;\n return out;\n}\n\n// ------------------------------- Entry ---------------------------------- //\n\n@compute @workgroup_size(8, 8, 1)\nfn encode(@builtin(global_invocation_id) gid: vec3<u32>) {\n if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) {\n return;\n }\n\n let bx = gid.x;\n let by = gid.y;\n let block_index = by * params.blocks_x + bx;\n\n let base = vec2<i32>(i32(bx) * 4, i32(by) * 4);\n let max_xy = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);\n\n var pixels: array<vec4<i32>, 16>;\n var lo = vec4<i32>(255);\n var hi = vec4<i32>(0);\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let p = clamp(base + vec2<i32>(i32(i & 3u), i32(i >> 2u)), vec2<i32>(0, 0), max_xy);\n let px = to8(textureLoad(src_tex, p, 0));\n pixels[i] = px;\n lo = min(lo, px);\n hi = max(hi, px);\n }\n\n // Both branches produce the final endpoints (already ordered so the\n // decoder doesn't apply blue contraction) and the packed weight word\n // (weight k's lsb at bit 31\u22122k, msb at bit 30\u22122k).\n var e0: vec4<i32>;\n var e1: vec4<i32>;\n var w3: u32 = 0u;\n\n if (QUALITY_HIGH != 0u) {\n let fp = farthest_pair(&pixels);\n e0 = fp.a;\n e1 = fp.b;\n var indices: array<u32, 16>;\n var pal: array<vec4<i32>, 4>;\n build_palette(e0, e1, &pal);\n var err = assign_all(&pixels, &pal, &indices);\n let refit = refit_endpoints(&pixels, &indices);\n if (refit.valid) {\n build_palette(refit.e0, refit.e1, &pal);\n var idx2: array<u32, 16>;\n let err2 = assign_all(&pixels, &pal, &idx2);\n if (err2 < err) {\n e0 = refit.e0;\n e1 = refit.e1;\n indices = idx2;\n err = err2;\n }\n }\n // Endpoint ordering, reflecting the assigned weights.\n if (e0.x + e0.y + e0.z > e1.x + e1.y + e1.z) {\n let tmp = e0; e0 = e1; e1 = tmp;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n indices[k] = 3u - indices[k];\n }\n }\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let w = indices[k] & 0x3u;\n w3 = w3 | ((w & 1u) << (31u - 2u * k)) | (((w >> 1u) & 1u) << (30u - 2u * k));\n }\n } else {\n // Fused LSQ fit seeded from the raw bbox, quantised refit endpoints,\n // ordering applied BEFORE the weight pass so no reflection is needed.\n // The refit is clamped to the block bbox: on multi-cluster blocks the\n // unconstrained solve extrapolates far outside the block's colours and the\n // per-channel [0,255] clamp then bends the hue \u2014 fringe pixels decode to\n // colours that exist nowhere in the block. Constraining to the bbox also\n // measures better in plain SSE (+1.8 dB on the colour test card).\n let r = proj_fit(&pixels, lo, hi);\n e0 = lo;\n e1 = hi;\n if (r.valid) { e0 = clamp(r.e0, lo, hi); e1 = clamp(r.e1, lo, hi); }\n if (e0.x + e0.y + e0.z > e1.x + e1.y + e1.z) {\n let tmp = e0; e0 = e1; e1 = tmp;\n }\n let dir = vec4<f32>(e1 - e0);\n let dd = dot(dir, dir);\n if (dd > 0.0) {\n let e0f = vec4<f32>(e0);\n let inv = 3.0 / dd;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let s = u32(clamp(floor(dot(vec4<f32>(pixels[k]) - e0f, dir) * inv + 0.5), 0.0, 3.0));\n w3 = w3 | ((s & 1u) << (31u - 2u * k)) | (((s >> 1u) & 1u) << (30u - 2u * k));\n }\n }\n }\n\n // Straight-line packing: block mode 0x042 @0, partitions\u22121=0 @11, CEM 12\n // @13, endpoints R0 R1 G0 G1 B0 B1 A0 A1 (8 bits each) from bit 17,\n // weights in the last word.\n let E0 = vec4<u32>(e0);\n let E1 = vec4<u32>(e1);\n let w0 = 0x042u | (12u << 13u) | (E0.x << 17u) | (E1.x << 25u);\n let w1 = (E1.x >> 7u) | (E0.y << 1u) | (E1.y << 9u) | (E0.z << 17u) | (E1.z << 25u);\n let w2 = (E1.z >> 7u) | (E0.w << 1u) | (E1.w << 9u);\n\n let out = block_index * 4u;\n dst[out + 0u] = w0;\n dst[out + 1u] = w1;\n dst[out + 2u] = w2;\n dst[out + 3u] = w3;\n}\n";
|
|
1038
|
+
var astc4x4_default = "// ASTC 4\xD74 LDR compute shader encoder.\n//\n// One invocation per 4\xD74 block. Emits 16 bytes = 4 u32s into the storage\n// buffer at `dst[block_index * 4 .. + 3]`. This is the f32 fallback;\n// astc4x4_fast_f16.wgsl is the same algorithm and is preferred when the\n// device reports shader-f16.\n//\n// ALGORITHM: principal-axis seed (covariance power-iteration; bbox on\n// degenerate blocks) at the exact projection extents \u2192 one fused pass that\n// projects each pixel onto the endpoint line (the 4 palette entries are\n// colinear, so the nearest is the rounded projection \u2014 no per-entry search)\n// while accumulating the least-squares refit sums, then a reprojection\n// against the quantised refit endpoints with the weights packed on the fly.\n// The endpoint ordering rule is applied before the weight pass, so no\n// reflection is needed.\n//\n// RESTRICTED SUBSET: single partition, no dual-plane, CEM 12 (LDR RGBA direct),\n// 4\xD74 weight grid with 2-bit weights (QUANT_4), 8-bit endpoints (QUANT_256).\n//\n// BLOCK LAYOUT (128 bits, LSB-first)\n// bits [10:0] block mode = 0x042\n// bits [12:11] partition count \u2212 1 = 0\n// bits [16:13] CEM = 12\n// bits [80:17] endpoints: R0 R1 G0 G1 B0 B1 A0 A1 (8-bit each)\n// bits [127:96] 16 \xD7 2-bit weights; weight k: bit(127\u22122k)=lsb, bit(126\u22122k)=msb\n//\n// ENDPOINT ORDERING: if sum(e0.rgb) > sum(e1.rgb) swap endpoints and reflect\n// indices (w' = 3 \u2212 w) to keep the decoder out of blue contraction.\n\nstruct Params {\n blocks_x: u32,\n blocks_y: u32,\n width: u32,\n height: u32,\n};\n\n@group(0) @binding(0) var src_tex: texture_2d<f32>;\n@group(0) @binding(1) var<storage, read_write> dst: array<u32>;\n@group(0) @binding(2) var<uniform> params: Params;\n\nfn to8(v: vec4<f32>) -> vec4<i32> {\n return vec4<i32>(clamp(floor(v * 255.0 + 0.5), vec4<f32>(0.0), vec4<f32>(255.0)));\n}\n\n// One pass over the block: project every pixel onto the e0\u2192e1 line (4 levels,\n// QUANT_4 \u2248 thirds) and accumulate the least-squares normal-equation sums;\n// solve for the refit endpoints. Weights are not produced here \u2014 the caller\n// reprojects against the quantised refit endpoints anyway.\nstruct Fit { e0: vec4<i32>, e1: vec4<i32>, valid: bool };\nfn proj_fit(pixels: ptr<function, array<vec4<i32>, 16>>, e0: vec4<i32>, e1: vec4<i32>) -> Fit {\n var out: Fit;\n out.valid = false;\n let dir = vec4<f32>(e1 - e0);\n let dd = dot(dir, dir);\n if (dd == 0.0) { return out; }\n let e0f = vec4<f32>(e0);\n let inv = 3.0 / dd;\n var sAA: f32 = 0.0; var sBB: f32 = 0.0; var sAB: f32 = 0.0;\n var sAV: vec4<f32> = vec4<f32>(0.0); var sBV: vec4<f32> = vec4<f32>(0.0);\n var s_min = 3.0; var s_max = 0.0;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let v = vec4<f32>((*pixels)[k]);\n let s = clamp(floor(dot(v - e0f, dir) * inv + 0.5), 0.0, 3.0);\n s_min = min(s_min, s); s_max = max(s_max, s);\n let b = s * (1.0 / 3.0); let a = 1.0 - b;\n sAA = sAA + a * a; sBB = sBB + b * b; sAB = sAB + a * b;\n sAV = sAV + a * v; sBV = sBV + b * v;\n }\n // Rank-1 guard: if every pixel projects to ONE level the system is\n // singular \u2014 det and the numerators are pure float rounding noise and the\n // solve returns garbage endpoints. With \u22652 levels det \u2265 15\xB7(1/3)\xB2 \u2248 1.67.\n if (s_min == s_max) { return out; }\n let det = sAA * sBB - sAB * sAB;\n if (abs(det) < 1e-3) { return out; }\n out.e0 = vec4<i32>(clamp(round((sBB * sAV - sAB * sBV) / det), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.e1 = vec4<i32>(clamp(round((sAA * sBV - sAB * sAV) / det), vec4<f32>(0.0), vec4<f32>(255.0)));\n out.valid = true;\n return out;\n}\n\n// Principal colour axis via covariance power-iteration (RGBA, 8-bit integer\n// pixel domain), seeded with the bbox diagonal. Returns a unit axis, or\n// vec4(0) for a degenerate (constant) block. Used to seed the LSQ fit \u2014 the\n// bbox diagonal is sign-blind and points across anti-correlated data (normal\n// maps, hue edges) instead of along it.\nfn principal_axis4(\n pixels: ptr<function, array<vec4<i32>, 16>>,\n mean: vec4<f32>,\n seed: vec4<f32>,\n) -> vec4<f32> {\n var c0v = vec4<f32>(0.0);\n var c1v = vec4<f32>(0.0);\n var c2v = vec4<f32>(0.0);\n var c3v = vec4<f32>(0.0);\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let d = vec4<f32>((*pixels)[k]) - mean;\n c0v = c0v + d.x * d;\n c1v = c1v + d.y * d;\n c2v = c2v + d.z * d;\n c3v = c3v + d.w * d;\n }\n var v = seed;\n var len = length(v);\n if (len < 1e-9) { return vec4<f32>(0.0); }\n v = v / len;\n for (var iter: u32 = 0u; iter < 8u; iter = iter + 1u) {\n let nv = vec4<f32>(dot(c0v, v), dot(c1v, v), dot(c2v, v), dot(c3v, v));\n len = length(nv);\n if (len < 1e-12) { return vec4<f32>(0.0); }\n v = nv / len;\n }\n return v;\n}\n\n// ------------------------------- Entry ---------------------------------- //\n\n@compute @workgroup_size(8, 8, 1)\nfn encode(@builtin(global_invocation_id) gid: vec3<u32>) {\n if (gid.x >= params.blocks_x || gid.y >= params.blocks_y) {\n return;\n }\n\n let bx = gid.x;\n let by = gid.y;\n let block_index = by * params.blocks_x + bx;\n\n let base = vec2<i32>(i32(bx) * 4, i32(by) * 4);\n let max_xy = vec2<i32>(i32(params.width) - 1, i32(params.height) - 1);\n\n var pixels: array<vec4<i32>, 16>;\n var lo = vec4<i32>(255);\n var hi = vec4<i32>(0);\n var isum = vec4<i32>(0);\n for (var i: u32 = 0u; i < 16u; i = i + 1u) {\n let p = clamp(base + vec2<i32>(i32(i & 3u), i32(i >> 2u)), vec2<i32>(0, 0), max_xy);\n let px = to8(textureLoad(src_tex, p, 0));\n pixels[i] = px;\n lo = min(lo, px);\n hi = max(hi, px);\n isum = isum + px;\n }\n let mean = vec4<f32>(isum) * (1.0 / 16.0);\n\n // Fused LSQ fit seeded from the block's principal colour axis at the exact\n // projection extents, quantised refit endpoints, ordering applied BEFORE\n // the weight pass so no reflection is needed.\n // The refit is clamped to the block bbox: on multi-cluster blocks the\n // unconstrained solve extrapolates far outside the block's colours and the\n // per-channel [0,255] clamp then bends the hue \u2014 fringe pixels decode to\n // colours that exist nowhere in the block. Constraining to the bbox also\n // measures better in plain SSE (+1.8 dB on the colour test card).\n var seed0 = lo;\n var seed1 = hi;\n let axis = principal_axis4(&pixels, mean, vec4<f32>(hi - lo));\n if (dot(axis, axis) > 0.0) {\n var t_min: f32 = 1e30;\n var t_max: f32 = -1e30;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let t = dot(vec4<f32>(pixels[k]) - mean, axis);\n t_min = min(t_min, t);\n t_max = max(t_max, t);\n }\n seed0 = vec4<i32>(clamp(round(mean + t_min * axis), vec4<f32>(0.0), vec4<f32>(255.0)));\n seed1 = vec4<i32>(clamp(round(mean + t_max * axis), vec4<f32>(0.0), vec4<f32>(255.0)));\n }\n let r = proj_fit(&pixels, seed0, seed1);\n var e0 = lo;\n var e1 = hi;\n if (r.valid) { e0 = clamp(r.e0, lo, hi); e1 = clamp(r.e1, lo, hi); }\n if (e0.x + e0.y + e0.z > e1.x + e1.y + e1.z) {\n let tmp = e0; e0 = e1; e1 = tmp;\n }\n\n // Weight pass, packing on the fly (weight k's lsb at bit 31\u22122k, msb at\n // bit 30\u22122k).\n var w3: u32 = 0u;\n let dir = vec4<f32>(e1 - e0);\n let dd = dot(dir, dir);\n if (dd > 0.0) {\n let e0f = vec4<f32>(e0);\n let inv = 3.0 / dd;\n for (var k: u32 = 0u; k < 16u; k = k + 1u) {\n let s = u32(clamp(floor(dot(vec4<f32>(pixels[k]) - e0f, dir) * inv + 0.5), 0.0, 3.0));\n w3 = w3 | ((s & 1u) << (31u - 2u * k)) | (((s >> 1u) & 1u) << (30u - 2u * k));\n }\n }\n\n // Straight-line packing: block mode 0x042 @0, partitions\u22121=0 @11, CEM 12\n // @13, endpoints R0 R1 G0 G1 B0 B1 A0 A1 (8 bits each) from bit 17,\n // weights in the last word.\n let E0 = vec4<u32>(e0);\n let E1 = vec4<u32>(e1);\n let w0 = 0x042u | (12u << 13u) | (E0.x << 17u) | (E1.x << 25u);\n let w1 = (E1.x >> 7u) | (E0.y << 1u) | (E1.y << 9u) | (E0.z << 17u) | (E1.z << 25u);\n let w2 = (E1.z >> 7u) | (E0.w << 1u) | (E1.w << 9u);\n\n let out = block_index * 4u;\n dst[out + 0u] = w0;\n dst[out + 1u] = w1;\n dst[out + 2u] = w2;\n dst[out + 3u] = w3;\n}\n";
|
|
844
1039
|
|
|
845
1040
|
// src/astc4x4_fast_f16.wgsl
|
|
846
1041
|
var astc4x4_fast_f16_default = `// astc4x4 "fast" encoder \u2014 f16 variant (requires the shader-f16 feature).
|
|
847
|
-
// Same algorithm family as the f32 fast path in astc4x4.wgsl (
|
|
848
|
-
// projection weight assignment with a fused least-squares refit \u2192
|
|
1042
|
+
// Same algorithm family as the f32 fast path in astc4x4.wgsl (principal-axis
|
|
1043
|
+
// seed \u2192 projection weight assignment with a fused least-squares refit \u2192
|
|
849
1044
|
// reproject), tuned for throughput:
|
|
850
1045
|
//
|
|
851
1046
|
// \u2022 All projection / refit math in f16 ([0,1] domain). The projection
|
|
@@ -930,10 +1125,57 @@ fn encode(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
|
930
1125
|
var pix: array<h4, 16>;
|
|
931
1126
|
var lo = h4(1.0);
|
|
932
1127
|
var hi = h4(0.0);
|
|
1128
|
+
var mean = h4(0.0);
|
|
933
1129
|
for (var i: u32 = 0u; i < 16u; i = i + 1u) {
|
|
934
1130
|
let p = clamp(base + vec2<i32>(i32(i & 3u), i32(i >> 2u)), vec2<i32>(0), mx);
|
|
935
1131
|
let px = h4(textureLoad(src_tex, p, 0));
|
|
936
1132
|
pix[i] = px; lo = min(lo, px); hi = max(hi, px);
|
|
1133
|
+
mean = mean + px;
|
|
1134
|
+
}
|
|
1135
|
+
mean = mean * h(1.0 / 16.0);
|
|
1136
|
+
|
|
1137
|
+
// Seed endpoints from the block's principal colour axis (covariance
|
|
1138
|
+
// power-iteration, seeded with the bbox diagonal). The bbox diagonal is
|
|
1139
|
+
// sign-blind: on anti-correlated channels (normal maps, hue edges) it
|
|
1140
|
+
// points across the data instead of along it, and the LSQ refit \u2014 which
|
|
1141
|
+
// fits endpoints GIVEN the projection weights \u2014 can't recover from a wrong
|
|
1142
|
+
// axis. Deviations are pre-scaled \xD716 so covariance entries for shallow
|
|
1143
|
+
// blocks stay in f16's normal range (span ~1/255 \u2192 d\xB2 \u2248 1e-3) while
|
|
1144
|
+
// full-range sums stay \u22644096; the iteration renormalises by the max
|
|
1145
|
+
// component (a plain length() of the matvec output could overflow f16), so
|
|
1146
|
+
// only the direction survives.
|
|
1147
|
+
var seed_lo = lo;
|
|
1148
|
+
var seed_hi = hi;
|
|
1149
|
+
var c0v = h4(0.0);
|
|
1150
|
+
var c1v = h4(0.0);
|
|
1151
|
+
var c2v = h4(0.0);
|
|
1152
|
+
var c3v = h4(0.0);
|
|
1153
|
+
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
1154
|
+
let d = (pix[k] - mean) * h(16.0);
|
|
1155
|
+
c0v = c0v + d.x * d;
|
|
1156
|
+
c1v = c1v + d.y * d;
|
|
1157
|
+
c2v = c2v + d.z * d;
|
|
1158
|
+
c3v = c3v + d.w * d;
|
|
1159
|
+
}
|
|
1160
|
+
var axis = hi - lo;
|
|
1161
|
+
var axis_ok = true;
|
|
1162
|
+
for (var it: u32 = 0u; it < 4u; it = it + 1u) {
|
|
1163
|
+
let nv = h4(dot(c0v, axis), dot(c1v, axis), dot(c2v, axis), dot(c3v, axis));
|
|
1164
|
+
let m = max(max(abs(nv.x), abs(nv.y)), max(abs(nv.z), abs(nv.w)));
|
|
1165
|
+
if (m < h(1e-4)) { axis_ok = false; break; }
|
|
1166
|
+
axis = nv / m;
|
|
1167
|
+
}
|
|
1168
|
+
if (axis_ok) {
|
|
1169
|
+
axis = axis / length(axis);
|
|
1170
|
+
var t_min = h(4.0);
|
|
1171
|
+
var t_max = h(-4.0);
|
|
1172
|
+
for (var k: u32 = 0u; k < 16u; k = k + 1u) {
|
|
1173
|
+
let t = dot(pix[k] - mean, axis);
|
|
1174
|
+
t_min = min(t_min, t);
|
|
1175
|
+
t_max = max(t_max, t);
|
|
1176
|
+
}
|
|
1177
|
+
seed_lo = clamp(mean + t_min * axis, h4(0.0), h4(1.0));
|
|
1178
|
+
seed_hi = clamp(mean + t_max * axis, h4(0.0), h4(1.0));
|
|
937
1179
|
}
|
|
938
1180
|
|
|
939
1181
|
// The refit is clamped to the block bbox: on multi-cluster blocks the
|
|
@@ -941,7 +1183,7 @@ fn encode(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
|
941
1183
|
// per-channel [0,1] clamp then bends the hue \u2014 fringe pixels decode to
|
|
942
1184
|
// colours that exist nowhere in the block. Constraining to the bbox also
|
|
943
1185
|
// measures better in plain SSE (+1.8 dB on the colour test card).
|
|
944
|
-
let r = proj_fit(&pix,
|
|
1186
|
+
let r = proj_fit(&pix, seed_lo, seed_hi);
|
|
945
1187
|
var e0 = lo;
|
|
946
1188
|
var e1 = hi;
|
|
947
1189
|
if (r.valid) { e0 = clamp(r.e0, lo, hi); e1 = clamp(r.e1, lo, hi); }
|
|
@@ -995,9 +1237,6 @@ var ASTC4x4Encoder = class extends Encoder {
|
|
|
995
1237
|
get supportsSrgb() {
|
|
996
1238
|
return true;
|
|
997
1239
|
}
|
|
998
|
-
get supportsQuality() {
|
|
999
|
-
return true;
|
|
1000
|
-
}
|
|
1001
1240
|
wgslSource() {
|
|
1002
1241
|
return astc4x4_default;
|
|
1003
1242
|
}
|
|
@@ -1220,7 +1459,7 @@ var WebGLBlockEncoder = class {
|
|
|
1220
1459
|
};
|
|
1221
1460
|
|
|
1222
1461
|
// src/webgl/glsl/bc1.frag.glsl
|
|
1223
|
-
var bc1_frag_default = "#version 300 es\n// BC1 (DXT1) fragment-shader encoder \u2014 WebGL2 port of bc1.wgsl (fast path).\n//\n// One fragment per 4\xD74 block. Output is the 8-byte BC1 block as 2 \xD7 u32 in\n// outColor.rg (outColor.ba unused); the encoder reads back RGBA32UI and keeps\n// the low two words per block.
|
|
1462
|
+
var bc1_frag_default = "#version 300 es\n// BC1 (DXT1) fragment-shader encoder \u2014 WebGL2 port of bc1.wgsl (fast path).\n//\n// One fragment per 4\xD74 block. Output is the 8-byte BC1 block as 2 \xD7 u32 in\n// outColor.rg (outColor.ba unused); the encoder reads back RGBA32UI and keeps\n// the low two words per block. Same algorithm as bc1.wgsl:\n// principal-axis endpoint seed (covariance\n// power-iteration; inset bbox on degenerate blocks), RGB565 quantisation,\n// forced 4-colour mode, full 4-entry L2 index search, then up to TWO\n// least-squares endpoint refit rounds, each accepted only when it lowers the\n// block's error. See bc1.wgsl for the full derivation.\n\nprecision highp float;\nprecision highp int;\n\nuniform sampler2D uSrc;\nuniform ivec2 uSrcSize; // original (unpadded) width, height\nuniform int uFlipY; // 1 = sample bottom-up (matches Three.js flipY)\n\nlayout(location = 0) out uvec4 outColor;\n\n// 4-colour-mode interpolation weights: pal[j] = WA[j]*c0 + WB[j]*c1.\nconst float WA[4] = float[4](1.0, 0.0, 2.0 / 3.0, 1.0 / 3.0);\nconst float WB[4] = float[4](0.0, 1.0, 1.0 / 3.0, 2.0 / 3.0);\n\nuint to565(vec3 c) {\n uint r = uint(clamp(floor(c.r * 31.0 + 0.5), 0.0, 31.0));\n uint g = uint(clamp(floor(c.g * 63.0 + 0.5), 0.0, 63.0));\n uint b = uint(clamp(floor(c.b * 31.0 + 0.5), 0.0, 31.0));\n return (r << 11) | (g << 5) | b;\n}\n\nvec3 from565(uint c) {\n float r = float((c >> 11) & 31u);\n float g = float((c >> 5) & 63u);\n float b = float(c & 31u);\n // 5/6-bit \u2192 8-bit. floor((x*527+23)/64) == (x<<3)|(x>>2): exact hardware\n // bit-replication (white \u2192 255), so index selection matches the GPU decode.\n float r8 = floor((r * 527.0 + 23.0) / 64.0);\n float g8 = floor((g * 259.0 + 33.0) / 64.0);\n float b8 = floor((b * 527.0 + 23.0) / 64.0);\n return vec3(r8, g8, b8) / 255.0;\n}\n\n// Per-invocation scratch (mirrors the WGSL function-scope arrays passed by\n// ptr; GLSL would copy array parameters by value).\nvec3 gPixels[16];\nuint gIdx[16];\n\n// Nearest-palette assignment of gPixels for the decoded palette of (c0,c1),\n// with the block's squared error and the LSQ normal-equation sums of the\n// resulting assignment accumulated in the same pass \u2014 so an accepted refit\n// can seed the next round.\nstruct Assign { float err; float sAA; float sBB; float sAB; vec3 sAV; vec3 sBV; };\nAssign assignStats(uint c0, uint c1, out uint indices[16]) {\n vec3 p0 = from565(c0);\n vec3 p1 = from565(c1);\n vec3 pal[4];\n for (int j = 0; j < 4; j++) pal[j] = WA[j] * p0 + WB[j] * p1;\n Assign r = Assign(0.0, 0.0, 0.0, 0.0, vec3(0.0), vec3(0.0));\n for (int k = 0; k < 16; k++) {\n vec3 c = gPixels[k];\n uint bestJ = 0u;\n float bestD = 1e30;\n for (int j = 0; j < 4; j++) {\n vec3 d = pal[j] - c;\n float d2 = dot(d, d);\n if (d2 < bestD) { bestD = d2; bestJ = uint(j); }\n }\n indices[k] = bestJ;\n r.err += bestD;\n float a = WA[int(bestJ)];\n float b = WB[int(bestJ)];\n r.sAA += a * a; r.sBB += b * b; r.sAB += a * b; r.sAV += a * c; r.sBV += b * c;\n }\n return r;\n}\n\nvoid main() {\n ivec2 base = ivec2(gl_FragCoord.xy) * 4;\n ivec2 maxXY = uSrcSize - ivec2(1);\n\n vec3 bbMin = vec3(1.0);\n vec3 bbMax = vec3(0.0);\n vec3 mean = vec3(0.0);\n for (int i = 0; i < 16; i++) {\n ivec2 p = clamp(base + ivec2(i & 3, i >> 2), ivec2(0), maxXY);\n int sy = (uFlipY != 0) ? (uSrcSize.y - 1 - p.y) : p.y;\n vec3 c = texelFetch(uSrc, ivec2(p.x, sy), 0).rgb;\n gPixels[i] = c;\n bbMin = min(bbMin, c);\n bbMax = max(bbMax, c);\n mean += c;\n }\n mean /= 16.0;\n\n // Seed endpoints from the block's principal colour axis (covariance\n // power-iteration, seeded with the bbox diagonal \u2014 mirrors bc1.wgsl). The\n // bbox diagonal is sign-blind: on anti-correlated channels (normal maps,\n // hue edges) it points across the data instead of along it, and the LSQ\n // refit below \u2014 which fits endpoints GIVEN the indices \u2014 can't recover.\n // Degenerate (near-flat) blocks keep the inset-bbox seed. Both seeds inset\n // by ~half an RGB565 cell (1/16) to tighten the quantised palette.\n vec3 c0v = vec3(0.0);\n vec3 c1v = vec3(0.0);\n vec3 c2v = vec3(0.0);\n for (int k = 0; k < 16; k++) {\n vec3 d = gPixels[k] - mean;\n c0v += d.x * d;\n c1v += d.y * d;\n c2v += d.z * d;\n }\n vec3 hi;\n vec3 lo;\n vec3 axis = bbMax - bbMin;\n float alen = length(axis);\n bool axisOk = alen > 1e-9;\n if (axisOk) {\n axis /= alen;\n for (int it = 0; it < 8; it++) {\n vec3 nv = vec3(dot(c0v, axis), dot(c1v, axis), dot(c2v, axis));\n float nlen = length(nv);\n if (nlen < 1e-12) { axisOk = false; break; }\n axis = nv / nlen;\n }\n }\n if (axisOk) {\n float tMin = 1e30;\n float tMax = -1e30;\n for (int k = 0; k < 16; k++) {\n float t = dot(gPixels[k] - mean, axis);\n tMin = min(tMin, t);\n tMax = max(tMax, t);\n }\n float pad = (tMax - tMin) / 16.0;\n hi = clamp(mean + (tMax - pad) * axis, vec3(0.0), vec3(1.0));\n lo = clamp(mean + (tMin + pad) * axis, vec3(0.0), vec3(1.0));\n } else {\n vec3 inset = (bbMax - bbMin) / 16.0;\n hi = clamp(bbMax - inset, vec3(0.0), vec3(1.0));\n lo = clamp(bbMin + inset, vec3(0.0), vec3(1.0));\n }\n\n uint c0 = to565(hi);\n uint c1 = to565(lo);\n // 4-colour mode requires color0 > color1.\n if (c0 == c1) {\n if (c1 > 0u) { c1 = c1 - 1u; } else { c0 = c0 + 1u; }\n } else if (c0 < c1) {\n uint tmp = c0; c0 = c1; c1 = tmp;\n }\n\n // Seed assignment, then up to TWO least-squares refit rounds (mirroring\n // bc1.wgsl's fast path), each accepted only if the block's squared error\n // drops \u2014 the refit minimises a continuous objective and can lose after\n // 565 quantisation. Every assignment pass re-accumulates the sums, so an\n // accepted round seeds the next.\n Assign cur = assignStats(c0, c1, gIdx);\n for (int it = 0; it < 2; it++) {\n float det = cur.sAA * cur.sBB - cur.sAB * cur.sAB;\n if (abs(det) <= 1e-9) { break; }\n // Clamp the refit to the block bbox (not [0,1]): on multi-cluster blocks\n // the unconstrained LSQ solve extrapolates far outside the block's\n // colours and the per-channel clamp then bends the hue \u2014 fringe pixels\n // decode to colours that exist nowhere in the block. Constraining to the\n // bbox also measures better in plain SSE (+1.6 dB on the colour test\n // card), so the accept-if-better guard below keeps more refits.\n vec3 e0 = clamp((cur.sBB * cur.sAV - cur.sAB * cur.sBV) / det, bbMin, bbMax);\n vec3 e1 = clamp((cur.sAA * cur.sBV - cur.sAB * cur.sAV) / det, bbMin, bbMax);\n uint nc0 = to565(e0);\n uint nc1 = to565(e1);\n if (nc0 < nc1) { uint t = nc0; nc0 = nc1; nc1 = t; }\n if (nc0 == nc1 || (nc0 == c0 && nc1 == c1)) { break; }\n uint idx2[16];\n Assign nxt = assignStats(nc0, nc1, idx2);\n if (nxt.err >= cur.err) { break; }\n c0 = nc0; c1 = nc1;\n cur = nxt;\n for (int k = 0; k < 16; k++) gIdx[k] = idx2[k];\n }\n\n uint indices = 0u;\n for (int k = 0; k < 16; k++) indices |= (gIdx[k] & 3u) << (uint(k) * 2u);\n\n outColor = uvec4(c0 | (c1 << 16), indices, 0u, 0u);\n}\n";
|
|
1224
1463
|
|
|
1225
1464
|
// src/webgl/BC1WebGLEncoder.ts
|
|
1226
1465
|
var BC1WebGLEncoder = class extends WebGLBlockEncoder {
|
|
@@ -1239,7 +1478,7 @@ var BC1WebGLEncoder = class extends WebGLBlockEncoder {
|
|
|
1239
1478
|
};
|
|
1240
1479
|
|
|
1241
1480
|
// src/webgl/glsl/bc5.frag.glsl
|
|
1242
|
-
var bc5_frag_default = "#version 300 es\n// BC5 (RGTC2) fragment-shader encoder \u2014 WebGL2 port of bc5.wgsl (fast path).\n//\n// One fragment per 4\xD74 block \u2192 16-byte BC5 block as 4 \xD7 u32 in outColor.\n// BC5 = two BC4 halves (R then G). This is the *fast* path only: bbox\n// endpoints + a
|
|
1481
|
+
var bc5_frag_default = "#version 300 es\n// BC5 (RGTC2) fragment-shader encoder \u2014 WebGL2 port of bc5.wgsl (fast path).\n//\n// One fragment per 4\xD74 block \u2192 16-byte BC5 block as 4 \xD7 u32 in outColor.\n// BC5 = two BC4 halves (R then G). This is the *fast* path only: bbox\n// endpoints + a full-L2 index assignment per channel with the least-squares\n// refit sums accumulated in the same pass, then one refit accepted only when\n// it lowers the block's error (mirrors bc5.wgsl's fast branch \u2014 worth\n// ~1.3 dB on the normal-map card). Always emits 6-interpolation mode\n// (red0 > red1). See bc5.wgsl for the full derivation.\n\nprecision highp float;\nprecision highp int;\n\nuniform sampler2D uSrc;\nuniform ivec2 uSrcSize;\nuniform int uFlipY;\n\nlayout(location = 0) out uvec4 outColor;\n\n// 6-interpolation-mode palette weights: pal[j] = W0_6[j]*r0 + W1_6[j]*r1.\nconst float W0_6[8] = float[8](1.0, 0.0, 6.0 / 7.0, 5.0 / 7.0, 4.0 / 7.0, 3.0 / 7.0, 2.0 / 7.0, 1.0 / 7.0);\nconst float W1_6[8] = float[8](0.0, 1.0, 1.0 / 7.0, 2.0 / 7.0, 3.0 / 7.0, 4.0 / 7.0, 5.0 / 7.0, 6.0 / 7.0);\n\nuint quantize8(float v) {\n return uint(clamp(floor(v * 255.0 + 0.5), 0.0, 255.0));\n}\n\n// Nearest-palette assignment with the LSQ normal-equation sums and total\n// squared error accumulated in the same pass. Sums are only consumed by the\n// caller's refit; err drives the accept-if-better test.\nstruct Assign { float err; float sAA; float sBB; float sAB; float sAV; float sBV; };\nAssign assignAll(float values[16], float pal[8], out uint indices[16]) {\n Assign r = Assign(0.0, 0.0, 0.0, 0.0, 0.0, 0.0);\n for (int k = 0; k < 16; k++) {\n float v = values[k];\n uint bestJ = 0u;\n float bestD = 1e20;\n for (int j = 0; j < 8; j++) {\n float d = pal[j] - v;\n float d2 = d * d;\n if (d2 < bestD) { bestD = d2; bestJ = uint(j); }\n }\n indices[k] = bestJ;\n r.err += bestD;\n float a = W0_6[int(bestJ)];\n float b = W1_6[int(bestJ)];\n r.sAA += a * a; r.sBB += b * b; r.sAB += a * b; r.sAV += a * v; r.sBV += b * v;\n }\n return r;\n}\n\n// Encode 16 single-channel values into an 8-byte BC4 block (two little-endian\n// u32s). Mirrors encode_bc4() in bc5.wgsl's fast branch: bbox seed, fused\n// assignment + LSQ sums, refit accepted only if the error drops. vmin/vmax\n// are the channel's min/max, computed in the caller's load loop \u2014 fusing\n// that scan there saves a 16-value pass per channel.\nuvec2 encodeBC4(float values[16], float vmin, float vmax) {\n uint r0 = quantize8(vmax);\n uint r1 = quantize8(vmin);\n if (r0 == r1) {\n if (r1 > 0u) { r1 = r1 - 1u; } else { r0 = r0 + 1u; }\n }\n\n float pal[8];\n float r0f = float(r0) / 255.0;\n float r1f = float(r1) / 255.0;\n for (int j = 0; j < 8; j++) {\n pal[j] = W0_6[j] * r0f + W1_6[j] * r1f;\n }\n\n uint indices[16];\n Assign seed = assignAll(values, pal, indices);\n\n // One least-squares refit, accepted only if the requantised endpoints lower\n // the block error. Clamp to [0,1], NOT the block's value range: for a\n // scalar channel, endpoints beyond the data range are often genuinely\n // optimal and there is no colour axis to bend \u2014 the bbox clamp the colour\n // formats need costs ~0.3 dB here. Keep 6-interp mode (r0 > r1 strictly).\n float det = seed.sAA * seed.sBB - seed.sAB * seed.sAB;\n if (abs(det) > 1e-9) {\n float e0 = clamp((seed.sBB * seed.sAV - seed.sAB * seed.sBV) / det, 0.0, 1.0);\n float e1 = clamp((seed.sAA * seed.sBV - seed.sAB * seed.sAV) / det, 0.0, 1.0);\n uint n0 = quantize8(e0);\n uint n1 = quantize8(e1);\n if (n0 > n1 && !(n0 == r0 && n1 == r1)) {\n float pal2[8];\n float n0f = float(n0) / 255.0;\n float n1f = float(n1) / 255.0;\n for (int j = 0; j < 8; j++) {\n pal2[j] = W0_6[j] * n0f + W1_6[j] * n1f;\n }\n uint idx2[16];\n Assign refit = assignAll(values, pal2, idx2);\n if (refit.err < seed.err) {\n r0 = n0; r1 = n1;\n for (int k = 0; k < 16; k++) indices[k] = idx2[k];\n }\n }\n }\n\n // Pack the 48-bit index field (bytes 2..7) split across two u32 halves.\n uint idxLo = 0u;\n uint idxHi = 0u;\n for (int k = 0; k < 16; k++) {\n uint bit = 3u * uint(k);\n uint v = indices[k] & 7u;\n if (bit + 3u <= 32u) {\n idxLo = idxLo | (v << bit);\n } else if (bit >= 32u) {\n idxHi = idxHi | (v << (bit - 32u));\n } else {\n idxLo = idxLo | (v << bit);\n idxHi = idxHi | (v >> (32u - bit));\n }\n }\n\n uint outLo = r0 | (r1 << 8) | ((idxLo & 0xFFFFu) << 16);\n uint outHi = (idxLo >> 16) | (idxHi << 16);\n return uvec2(outLo, outHi);\n}\n\nvoid main() {\n ivec2 base = ivec2(gl_FragCoord.xy) * 4;\n ivec2 maxXY = uSrcSize - ivec2(1);\n\n float rValues[16];\n float gValues[16];\n float rMin = 1.0; float rMax = 0.0;\n float gMin = 1.0; float gMax = 0.0;\n for (int i = 0; i < 16; i++) {\n ivec2 p = clamp(base + ivec2(i & 3, i >> 2), ivec2(0), maxXY);\n int sy = (uFlipY != 0) ? (uSrcSize.y - 1 - p.y) : p.y;\n vec4 c = texelFetch(uSrc, ivec2(p.x, sy), 0);\n rValues[i] = c.r;\n gValues[i] = c.g;\n rMin = min(rMin, c.r); rMax = max(rMax, c.r);\n gMin = min(gMin, c.g); gMax = max(gMax, c.g);\n }\n\n uvec2 rBlock = encodeBC4(rValues, rMin, rMax);\n uvec2 gBlock = encodeBC4(gValues, gMin, gMax);\n outColor = uvec4(rBlock.x, rBlock.y, gBlock.x, gBlock.y);\n}\n";
|
|
1243
1482
|
|
|
1244
1483
|
// src/webgl/BC5WebGLEncoder.ts
|
|
1245
1484
|
var BC5WebGLEncoder = class extends WebGLBlockEncoder {
|
|
@@ -1258,7 +1497,7 @@ var BC5WebGLEncoder = class extends WebGLBlockEncoder {
|
|
|
1258
1497
|
};
|
|
1259
1498
|
|
|
1260
1499
|
// src/webgl/glsl/bc7.frag.glsl
|
|
1261
|
-
var bc7_frag_default = "#version 300 es\n// BC7 (BPTC) mode-6 fragment-shader encoder \u2014 WebGL2 port of bc7.wgsl (fast).\n//\n// One fragment per 4\xD74 block \u2192 16-byte block as 4 \xD7 u32 in outColor. Fast path\n// only:
|
|
1500
|
+
var bc7_frag_default = "#version 300 es\n// BC7 (BPTC) mode-6 fragment-shader encoder \u2014 WebGL2 port of bc7.wgsl (fast).\n//\n// One fragment per 4\xD74 block \u2192 16-byte block as 4 \xD7 u32 in outColor. Fast path\n// only: principal-axis seed (covariance power-iteration; bbox on degenerate\n// blocks) at the exact projection extents, quantised directly (no LSQ refit \u2014\n// mode 6's 16-level palette leaves it under 0.15 dB) \u2192 one projection-based\n// index-assignment pass (palette is colinear, so the nearest entry is found\n// by projecting onto the endpoint line \u2014 O(1) per pixel). Same algorithm as\n// bc7.wgsl; see that file for the mode-6 bit layout and rationale.\n//\n// Determinism note: the WGSL refit uses round() (half-to-even); here we use\n// floor(x + 0.5) for portability. The two differ only at exact .5 ties, a\n// sub-LSB endpoint nudge that is visually identical.\n\nprecision highp float;\nprecision highp int;\n\nuniform sampler2D uSrc;\nuniform ivec2 uSrcSize;\nuniform int uFlipY;\n\nlayout(location = 0) out uvec4 outColor;\n\n// Per-invocation scratch (mirrors the WGSL function-scope arrays passed by ptr).\nivec4 gPixels[16];\nuint gIdx[16];\n\nstruct QuantPair { ivec4 seven; ivec4 eight; };\nstruct Ep { ivec4 seven; ivec4 eight; uint p; };\nstruct Fit { ivec4 e0; ivec4 e1; bool valid; };\n\nivec4 to8(vec4 v) {\n return ivec4(clamp(floor(v * 255.0 + 0.5), vec4(0.0), vec4(255.0)));\n}\n\nint dist2(ivec4 a, ivec4 b) {\n ivec4 d = a - b;\n ivec4 e = d * d;\n return e.x + e.y + e.z + e.w;\n}\n\n// Quantize an 8-bit ideal endpoint to (7-bit value, reconstructed 8-bit) under\n// a fixed p-bit, all four channels at once.\nQuantPair quantizeEndpoint(ivec4 ideal8, uint p) {\n ivec4 q = ivec4(clamp(floor((vec4(ideal8) - float(p)) / 2.0 + 0.5), vec4(0.0), vec4(127.0)));\n // eff = (q << 1) | p. q*2 is even and p \u2208 {0,1}, so q*2 + p is identical and\n // avoids any vector-shift-by-scalar portability question.\n ivec4 eff = q * 2 + ivec4(int(p));\n return QuantPair(q, eff);\n}\n\n// Endpoint with its chosen p-bit, picked by minimum quantisation error.\nEp pickEp(ivec4 ideal) {\n QuantPair a = quantizeEndpoint(ideal, 0u);\n QuantPair b = quantizeEndpoint(ideal, 1u);\n if (dist2(b.eight, ideal) < dist2(a.eight, ideal)) {\n return Ep(b.seven, b.eight, 1u);\n }\n return Ep(a.seven, a.eight, 0u);\n}\n\n// Principal colour axis via power-iteration over precomputed, mean-corrected\n// covariance rows (the moments are accumulated for free in the pixel-load\n// loop), seeded with the bbox diagonal. Returns a unit axis, or vec4(0.0)\n// for a degenerate (constant) block. The bbox diagonal alone is sign-blind\n// and points across anti-correlated data (normal maps, hue edges) instead of\n// along it.\nvec4 principalAxis(vec4 c0v, vec4 c1v, vec4 c2v, vec4 c3v, vec4 seed) {\n vec4 v = seed;\n float len = length(v);\n if (len < 1e-9) { return vec4(0.0); }\n v /= len;\n for (int it = 0; it < 8; it++) {\n vec4 nv = vec4(dot(c0v, v), dot(c1v, v), dot(c2v, v), dot(c3v, v));\n len = length(nv);\n if (len < 1e-12) { return vec4(0.0); }\n v = nv / len;\n }\n return v;\n}\n\n// Projection index assignment over gPixels \u2192 gIdx. When `fit`, accumulate the\n// LSQ normal-equation sums in the same pass and return refitted endpoints.\nFit projAssign(ivec4 pe0, ivec4 pe1, bool fit) {\n Fit res;\n res.e0 = ivec4(0);\n res.e1 = ivec4(0);\n res.valid = false;\n ivec4 dir = pe1 - pe0;\n int dd = dir.x * dir.x + dir.y * dir.y + dir.z * dir.z + dir.w * dir.w;\n if (dd == 0) {\n for (int k = 0; k < 16; k++) { gIdx[k] = 0u; }\n return res;\n }\n float inv = 15.0 / float(dd);\n float sAA = 0.0, sBB = 0.0, sAB = 0.0;\n vec4 sAV = vec4(0.0), sBV = vec4(0.0);\n for (int k = 0; k < 16; k++) {\n ivec4 q = gPixels[k] - pe0;\n float proj = float(q.x * dir.x + q.y * dir.y + q.z * dir.z + q.w * dir.w) * inv;\n float s = clamp(floor(proj + 0.5), 0.0, 15.0);\n gIdx[k] = uint(s);\n if (fit) {\n vec4 v = vec4(gPixels[k]);\n float b = s / 15.0;\n float a = 1.0 - b;\n sAA += a * a; sBB += b * b; sAB += a * b; sAV += a * v; sBV += b * v;\n }\n }\n if (!fit) { return res; }\n float det = sAA * sBB - sAB * sAB;\n if (abs(det) < 1e-9) { return res; }\n res.e0 = ivec4(clamp(floor((sBB * sAV - sAB * sBV) / det + 0.5), vec4(0.0), vec4(255.0)));\n res.e1 = ivec4(clamp(floor((sAA * sBV - sAB * sAV) / det + 0.5), vec4(0.0), vec4(255.0)));\n res.valid = true;\n return res;\n}\n\nvoid writeBits(inout uint block[4], uint pos, uint nbits, uint value) {\n uint v = value & ((1u << nbits) - 1u);\n uint wordLo = pos / 32u;\n uint bitLo = pos % 32u;\n uint bitsInLo = min(nbits, 32u - bitLo);\n uint maskLo = ((1u << bitsInLo) - 1u) << bitLo;\n block[wordLo] = (block[wordLo] & ~maskLo) | ((v << bitLo) & maskLo);\n if (bitsInLo < nbits) {\n uint bitsInHi = nbits - bitsInLo;\n uint maskHi = (1u << bitsInHi) - 1u;\n uint valHi = v >> bitsInLo;\n block[wordLo + 1u] = (block[wordLo + 1u] & ~maskHi) | (valHi & maskHi);\n }\n}\n\nvoid main() {\n ivec2 base = ivec2(gl_FragCoord.xy) * 4;\n ivec2 maxXY = uSrcSize - ivec2(1);\n\n // Load pass with the covariance moments FUSED in: d = px \u2212 pixel0\n // (first-pixel-relative, so the sums scale with the block's span).\n ivec4 lo = ivec4(255);\n ivec4 hi = ivec4(0);\n vec4 p0f = vec4(0.0);\n vec4 sd = vec4(0.0);\n vec4 c0v = vec4(0.0);\n vec4 c1v = vec4(0.0);\n vec4 c2v = vec4(0.0);\n vec4 c3v = vec4(0.0);\n for (int i = 0; i < 16; i++) {\n ivec2 p = clamp(base + ivec2(i & 3, i >> 2), ivec2(0), maxXY);\n int sy = (uFlipY != 0) ? (uSrcSize.y - 1 - p.y) : p.y;\n ivec4 px = to8(texelFetch(uSrc, ivec2(p.x, sy), 0));\n gPixels[i] = px;\n lo = min(lo, px);\n hi = max(hi, px);\n if (i == 0) { p0f = vec4(px); }\n vec4 d = vec4(px) - p0f;\n sd += d;\n c0v += d.x * d;\n c1v += d.y * d;\n c2v += d.z * d;\n c3v += d.w * d;\n }\n vec4 mean = p0f + sd / 16.0;\n // Mean-correct the fused moments: C = \u03A3dd\u1D40 \u2212 (\u03A3d)(\u03A3d)\u1D40/16.\n vec4 sd16 = sd / 16.0;\n c0v -= sd.x * sd16;\n c1v -= sd.y * sd16;\n c2v -= sd.z * sd16;\n c3v -= sd.w * sd16;\n\n ivec4 seed0 = lo;\n ivec4 seed1 = hi;\n vec4 axis = principalAxis(c0v, c1v, c2v, c3v, vec4(hi - lo));\n if (dot(axis, axis) > 0.0) {\n // Exact projection extents along the axis. (A Rayleigh-quotient span\n // estimate was tried in place of this pass \u2014 it saves 16 dots but costs\n // 0.1\u20130.8 dB and 4\u201310\xD7 on the worst-easy-block gate: \u03C3 misjudges\n // two-cluster and outlier blocks. The pass stays.)\n float tMin = 1e30;\n float tMax = -1e30;\n for (int k = 0; k < 16; k++) {\n float t = dot(vec4(gPixels[k]) - mean, axis);\n tMin = min(tMin, t);\n tMax = max(tMax, t);\n }\n seed0 = ivec4(clamp(floor(mean + tMin * axis + 0.5), vec4(0.0), vec4(255.0)));\n seed1 = ivec4(clamp(floor(mean + tMax * axis + 0.5), vec4(0.0), vec4(255.0)));\n }\n\n // Quantise the PCA-extents seed directly and assign indices in one\n // projection pass \u2014 no LSQ refit: with the seed already on the principal\n // axis, mode 6's 16-level palette leaves the refit under 0.15 dB (the\n // coarse 4-level BC1/ASTC fast paths DO keep theirs).\n Ep ep0 = pickEp(seed0);\n Ep ep1 = pickEp(seed1);\n projAssign(ep0.eight, ep1.eight, false);\n ivec4 e0_7 = ep0.seven;\n ivec4 e1_7 = ep1.seven;\n uint p0 = ep0.p;\n uint p1 = ep1.p;\n\n // Anchor rule \u2014 pixel 0's index MSB must be 0; otherwise swap endpoints and\n // reflect every index (decoded image unchanged).\n if ((gIdx[0] & 0x8u) != 0u) {\n ivec4 t = e0_7; e0_7 = e1_7; e1_7 = t;\n uint tp = p0; p0 = p1; p1 = tp;\n for (int k = 0; k < 16; k++) { gIdx[k] = 15u - gIdx[k]; }\n }\n\n uint block[4];\n block[0] = 0u; block[1] = 0u; block[2] = 0u; block[3] = 0u;\n uint pos = 0u;\n writeBits(block, pos, 7u, 0x40u); pos += 7u;\n writeBits(block, pos, 7u, uint(e0_7.x)); pos += 7u;\n writeBits(block, pos, 7u, uint(e1_7.x)); pos += 7u;\n writeBits(block, pos, 7u, uint(e0_7.y)); pos += 7u;\n writeBits(block, pos, 7u, uint(e1_7.y)); pos += 7u;\n writeBits(block, pos, 7u, uint(e0_7.z)); pos += 7u;\n writeBits(block, pos, 7u, uint(e1_7.z)); pos += 7u;\n writeBits(block, pos, 7u, uint(e0_7.w)); pos += 7u;\n writeBits(block, pos, 7u, uint(e1_7.w)); pos += 7u;\n writeBits(block, pos, 1u, p0); pos += 1u;\n writeBits(block, pos, 1u, p1); pos += 1u;\n writeBits(block, pos, 3u, gIdx[0] & 0x7u); pos += 3u;\n for (int k = 1; k < 16; k++) {\n writeBits(block, pos, 4u, gIdx[k] & 0xFu);\n pos += 4u;\n }\n\n outColor = uvec4(block[0], block[1], block[2], block[3]);\n}\n";
|
|
1262
1501
|
|
|
1263
1502
|
// src/webgl/BC7WebGLEncoder.ts
|
|
1264
1503
|
var BC7WebGLEncoder = class extends WebGLBlockEncoder {
|
|
@@ -1277,7 +1516,7 @@ var BC7WebGLEncoder = class extends WebGLBlockEncoder {
|
|
|
1277
1516
|
};
|
|
1278
1517
|
|
|
1279
1518
|
// src/webgl/glsl/astc4x4.frag.glsl
|
|
1280
|
-
var astc4x4_frag_default = "#version 300 es\n// ASTC 4\xD74 LDR fragment-shader encoder \u2014 WebGL2 port of astc4x4.wgsl (fast).\n//\n// One fragment per 4\xD74 block \u2192 16-byte block as 4 \xD7 u32 in outColor. Restricted\n// subset: single partition, no dual-plane, CEM 12 (LDR RGBA direct), 4\xD74 weight\n// grid with 2-bit weights (QUANT_4), 8-bit endpoints (QUANT_256). Fast path:\n//
|
|
1519
|
+
var astc4x4_frag_default = "#version 300 es\n// ASTC 4\xD74 LDR fragment-shader encoder \u2014 WebGL2 port of astc4x4.wgsl (fast).\n//\n// One fragment per 4\xD74 block \u2192 16-byte block as 4 \xD7 u32 in outColor. Restricted\n// subset: single partition, no dual-plane, CEM 12 (LDR RGBA direct), 4\xD74 weight\n// grid with 2-bit weights (QUANT_4), 8-bit endpoints (QUANT_256). Fast path:\n// principal-axis seed (covariance power-iteration; bbox on degenerate blocks)\n// \u2192 one LSQ refit fused into a projection weight assignment (4 colinear\n// levels). Mirrors astc4x4.wgsl; see that file for the 128-bit block layout.\n//\n// Determinism note: floor(x + 0.5) replaces WGSL round() for the refit endpoints\n// (sub-LSB difference at exact .5 ties only).\n\nprecision highp float;\nprecision highp int;\n\nuniform sampler2D uSrc;\nuniform ivec2 uSrcSize;\nuniform int uFlipY;\n\nlayout(location = 0) out uvec4 outColor;\n\nivec4 gPixels[16];\nuint gIdx[16];\n\nstruct Fit { ivec4 e0; ivec4 e1; bool valid; };\n\nivec4 to8(vec4 v) {\n return ivec4(clamp(floor(v * 255.0 + 0.5), vec4(0.0), vec4(255.0)));\n}\n\n// Principal colour axis of gPixels via covariance power-iteration, seeded\n// with the bbox diagonal. Returns a unit axis, or vec4(0.0) for a degenerate\n// (constant) block. The bbox diagonal alone is sign-blind and points across\n// anti-correlated data (normal maps, hue edges) instead of along it.\nvec4 principalAxis(vec4 mean, vec4 seed) {\n vec4 c0v = vec4(0.0);\n vec4 c1v = vec4(0.0);\n vec4 c2v = vec4(0.0);\n vec4 c3v = vec4(0.0);\n for (int k = 0; k < 16; k++) {\n vec4 d = vec4(gPixels[k]) - mean;\n c0v += d.x * d;\n c1v += d.y * d;\n c2v += d.z * d;\n c3v += d.w * d;\n }\n vec4 v = seed;\n float len = length(v);\n if (len < 1e-9) { return vec4(0.0); }\n v /= len;\n for (int it = 0; it < 8; it++) {\n vec4 nv = vec4(dot(c0v, v), dot(c1v, v), dot(c2v, v), dot(c3v, v));\n len = length(nv);\n if (len < 1e-12) { return vec4(0.0); }\n v = nv / len;\n }\n return v;\n}\n\n// Projection weight assignment over 4 levels (QUANT_4 \u2248 thirds), with the LSQ\n// normal-equation sums accumulated in the same pass for a fused refit.\nFit projAssign(ivec4 pe0, ivec4 pe1, bool fit) {\n Fit res;\n res.e0 = ivec4(0);\n res.e1 = ivec4(0);\n res.valid = false;\n ivec4 dir = pe1 - pe0;\n int dd = dir.x * dir.x + dir.y * dir.y + dir.z * dir.z + dir.w * dir.w;\n if (dd == 0) {\n for (int k = 0; k < 16; k++) { gIdx[k] = 0u; }\n return res;\n }\n float inv = 3.0 / float(dd);\n float sAA = 0.0, sBB = 0.0, sAB = 0.0;\n vec4 sAV = vec4(0.0), sBV = vec4(0.0);\n for (int k = 0; k < 16; k++) {\n ivec4 q = gPixels[k] - pe0;\n float proj = float(q.x * dir.x + q.y * dir.y + q.z * dir.z + q.w * dir.w) * inv;\n float s = clamp(floor(proj + 0.5), 0.0, 3.0);\n gIdx[k] = uint(s);\n if (fit) {\n vec4 v = vec4(gPixels[k]);\n float b = s / 3.0;\n float a = 1.0 - b;\n sAA += a * a; sBB += b * b; sAB += a * b; sAV += a * v; sBV += b * v;\n }\n }\n if (!fit) { return res; }\n float det = sAA * sBB - sAB * sAB;\n if (abs(det) < 1e-9) { return res; }\n res.e0 = ivec4(clamp(floor((sBB * sAV - sAB * sBV) / det + 0.5), vec4(0.0), vec4(255.0)));\n res.e1 = ivec4(clamp(floor((sAA * sBV - sAB * sAV) / det + 0.5), vec4(0.0), vec4(255.0)));\n res.valid = true;\n return res;\n}\n\nvoid writeBits(inout uint block[4], uint pos, uint nbits, uint value) {\n uint v = value & ((1u << nbits) - 1u);\n uint wordLo = pos / 32u;\n uint bitLo = pos % 32u;\n uint bitsInLo = min(nbits, 32u - bitLo);\n uint maskLo = ((1u << bitsInLo) - 1u) << bitLo;\n block[wordLo] = (block[wordLo] & ~maskLo) | ((v << bitLo) & maskLo);\n if (bitsInLo < nbits) {\n uint bitsInHi = nbits - bitsInLo;\n uint maskHi = (1u << bitsInHi) - 1u;\n uint valHi = v >> bitsInLo;\n block[wordLo + 1u] = (block[wordLo + 1u] & ~maskHi) | (valHi & maskHi);\n }\n}\n\nvoid main() {\n ivec2 base = ivec2(gl_FragCoord.xy) * 4;\n ivec2 maxXY = uSrcSize - ivec2(1);\n\n ivec4 lo = ivec4(255);\n ivec4 hi = ivec4(0);\n ivec4 isum = ivec4(0);\n for (int i = 0; i < 16; i++) {\n ivec2 p = clamp(base + ivec2(i & 3, i >> 2), ivec2(0), maxXY);\n int sy = (uFlipY != 0) ? (uSrcSize.y - 1 - p.y) : p.y;\n ivec4 px = to8(texelFetch(uSrc, ivec2(p.x, sy), 0));\n gPixels[i] = px;\n lo = min(lo, px);\n hi = max(hi, px);\n isum += px;\n }\n vec4 mean = vec4(isum) / 16.0;\n\n ivec4 e0 = lo;\n ivec4 e1 = hi;\n vec4 axis = principalAxis(mean, vec4(hi - lo));\n if (dot(axis, axis) > 0.0) {\n float tMin = 1e30;\n float tMax = -1e30;\n for (int k = 0; k < 16; k++) {\n float t = dot(vec4(gPixels[k]) - mean, axis);\n tMin = min(tMin, t);\n tMax = max(tMax, t);\n }\n e0 = ivec4(clamp(floor(mean + tMin * axis + 0.5), vec4(0.0), vec4(255.0)));\n e1 = ivec4(clamp(floor(mean + tMax * axis + 0.5), vec4(0.0), vec4(255.0)));\n }\n Fit r = projAssign(e0, e1, true);\n if (r.valid) {\n // Clamp the refit to the block bbox: on multi-cluster blocks the\n // unconstrained LSQ solve extrapolates far outside the block's colours and\n // the per-channel [0,255] clamp then bends the hue \u2014 fringe pixels decode\n // to colours that exist nowhere in the block. Constraining to the bbox\n // also measures better in plain SSE (+1.8 dB on the colour test card).\n e0 = clamp(r.e0, lo, hi);\n e1 = clamp(r.e1, lo, hi);\n projAssign(e0, e1, false);\n }\n\n // Endpoint ordering so the decoder doesn't apply blue contraction.\n if (e0.x + e0.y + e0.z > e1.x + e1.y + e1.z) {\n ivec4 t = e0; e0 = e1; e1 = t;\n for (int k = 0; k < 16; k++) { gIdx[k] = 3u - gIdx[k]; }\n }\n\n uint block[4];\n block[0] = 0u; block[1] = 0u; block[2] = 0u; block[3] = 0u;\n writeBits(block, 0u, 11u, 0x042u);\n writeBits(block, 11u, 2u, 0u);\n writeBits(block, 13u, 4u, 12u);\n writeBits(block, 17u + 0u * 8u, 8u, uint(e0.x));\n writeBits(block, 17u + 1u * 8u, 8u, uint(e1.x));\n writeBits(block, 17u + 2u * 8u, 8u, uint(e0.y));\n writeBits(block, 17u + 3u * 8u, 8u, uint(e1.y));\n writeBits(block, 17u + 4u * 8u, 8u, uint(e0.z));\n writeBits(block, 17u + 5u * 8u, 8u, uint(e1.z));\n writeBits(block, 17u + 6u * 8u, 8u, uint(e0.w));\n writeBits(block, 17u + 7u * 8u, 8u, uint(e1.w));\n\n uint w3 = 0u;\n for (int k = 0; k < 16; k++) {\n uint w = gIdx[k] & 0x3u;\n w3 = w3 | ((w & 1u) << (31u - 2u * uint(k))) | (((w >> 1u) & 1u) << (30u - 2u * uint(k)));\n }\n block[3] = w3;\n\n outColor = uvec4(block[0], block[1], block[2], block[3]);\n}\n";
|
|
1281
1520
|
|
|
1282
1521
|
// src/webgl/ASTC4x4WebGLEncoder.ts
|
|
1283
1522
|
var ASTC4x4WebGLEncoder = class extends WebGLBlockEncoder {
|
|
@@ -1658,11 +1897,11 @@ function threeFormatFor(format) {
|
|
|
1658
1897
|
function buildCompressedTexture(levels, format) {
|
|
1659
1898
|
return assembleCompressedTexture(levels, threeFormatFor(format), isSrgbFormat(format));
|
|
1660
1899
|
}
|
|
1661
|
-
async function encodeToTexture(encoder, source, { colorSpace = "srgb",
|
|
1900
|
+
async function encodeToTexture(encoder, source, { colorSpace = "srgb", flipY = false } = {}) {
|
|
1662
1901
|
const formats = encoder.constructor.textureFormats;
|
|
1663
1902
|
const wantSrgb = colorSpace === "srgb" && encoder.supportsSrgb;
|
|
1664
1903
|
const format = formats.find((f) => isSrgbFormat(f) === wantSrgb) ?? formats[0];
|
|
1665
|
-
const bytes = await encoder.encodeToBytes(source, { flipY
|
|
1904
|
+
const bytes = await encoder.encodeToBytes(source, { flipY });
|
|
1666
1905
|
const texture = buildCompressedTexture([bytes], format);
|
|
1667
1906
|
return { ...bytes, texture };
|
|
1668
1907
|
}
|
|
@@ -1748,7 +1987,6 @@ async function compressTexture(source, options = {}) {
|
|
|
1748
1987
|
svgSize,
|
|
1749
1988
|
flipY = true,
|
|
1750
1989
|
mipmaps = false,
|
|
1751
|
-
quality = "fast",
|
|
1752
1990
|
device: providedDevice,
|
|
1753
1991
|
adapter: providedAdapter
|
|
1754
1992
|
} = options;
|
|
@@ -1796,9 +2034,9 @@ async function compressTexture(source, options = {}) {
|
|
|
1796
2034
|
if (needsWriteTexture) {
|
|
1797
2035
|
const level02 = bitmapToMipLevel(bitmap, flipY);
|
|
1798
2036
|
const imageData = mipLevelToImageData(level02);
|
|
1799
|
-
bytes = await encoder.encodeToBytes(imageData
|
|
2037
|
+
bytes = await encoder.encodeToBytes(imageData);
|
|
1800
2038
|
} else {
|
|
1801
|
-
bytes = await encoder.encodeToBytes(bitmap, { flipY
|
|
2039
|
+
bytes = await encoder.encodeToBytes(bitmap, { flipY });
|
|
1802
2040
|
}
|
|
1803
2041
|
const tex3 = buildCompressedTexture([bytes], selection.format);
|
|
1804
2042
|
return {
|
|
@@ -1824,7 +2062,7 @@ async function compressTexture(source, options = {}) {
|
|
|
1824
2062
|
for (const level of chain) {
|
|
1825
2063
|
const padded = padToBlockMultiple(level);
|
|
1826
2064
|
const imageData = mipLevelToImageData(padded);
|
|
1827
|
-
const bytes = await encoder.encodeToBytes(imageData
|
|
2065
|
+
const bytes = await encoder.encodeToBytes(imageData);
|
|
1828
2066
|
encodedLevels.push(bytes);
|
|
1829
2067
|
totalEncodeMs += bytes.encodeMs;
|
|
1830
2068
|
}
|
|
@@ -1933,8 +2171,6 @@ var GputexLoader = class extends Loader {
|
|
|
1933
2171
|
flipY = true;
|
|
1934
2172
|
/** Generate + encode a full mip chain. Default false. */
|
|
1935
2173
|
mipmaps = false;
|
|
1936
|
-
/** Encode quality / speed trade-off. Default 'fast' (~2–4× faster, ≤0.36 dB). */
|
|
1937
|
-
quality = "fast";
|
|
1938
2174
|
/**
|
|
1939
2175
|
* Optional pre-existing WebGPU device. Reusing the renderer's device
|
|
1940
2176
|
* avoids spinning up a second WebGPU context for encoding.
|
|
@@ -1963,7 +2199,6 @@ var GputexLoader = class extends Loader {
|
|
|
1963
2199
|
svgSize: this.svgSize,
|
|
1964
2200
|
flipY: this.flipY,
|
|
1965
2201
|
mipmaps: this.mipmaps,
|
|
1966
|
-
quality: this.quality,
|
|
1967
2202
|
device: this.device,
|
|
1968
2203
|
adapter: this.adapter
|
|
1969
2204
|
}).then(
|