ARTICLE DETAIL

资讯详情

深耕网站建设与运营推广的一线实战洞察。

WebGPU 后期处理管线深度实现:手写 WGSL 屏幕空间全屏泛光(Bloom)着色器

WebGPU 后期处理管线深度实现:手写 WGSL 屏幕空间全屏泛光(Bloom)着色器 在次世代 Web 3D 场景、数字孪生以及赛博朋克风格的交互界面中全屏泛光Bloom Effect是赋予发光材质、霓虹光带与超强光源“真实物理光芒”的灵魂级后期处理技术。在物理光学世界中当进入人眼或摄影机镜头的光子通量远超感光芯片的动态范围时镜头内部的镜片微小瑕疵与眼球玻璃体会导致强烈光线向周围扩散漫溢Light Bleeding。在 WebGL 时代实现高质量 Bloom 往往受制于繁琐的多 Pass 切换开销与移动端显存带宽瓶颈。很多开发者为了省事只做一次简单的全屏高斯模糊结果导致低亮度区域发灰、高光区域边缘生硬失真。进入 WebGPU 时代得益于现代显式管线Explicit Pipeline的高性能调度以及 WGSLWebGPU Shading Language对硬件级双线性过滤的高效支持我们可以构建媲美《使命召唤》等 3A 端游质量的物理级双向金字塔 Bloom 管线Dual Kawase / Mip-chain Downsample Upsample。本文将从物理光学逻辑、降采样金字塔拓扑到手写 WGSL 着色器深入拆解这一高性能泛光管线的实现。现代 Bloom 管线架构降采样与升采样金字塔直接在全分辨率屏幕如 4K 视网膜屏上执行大半径模糊其像素采样开销将彻底摧毁 GPU。现代工业级 Bloom 算法由 Jorge Jimenez 在 SIGGRAPH 2014 提出并广泛应用于现代游戏引擎采用了类似拉普拉斯金字塔的多级迭代流程[原始 HDR 渲染画面] │ ▼ (Pass 1: 亮度阈值提取 Threshold Pass) [提取的高亮纹理 Level 0 (全分辨率)] │ ▼ (Pass 2: 13-tap 降采样) [高光纹理 Level 1 (1/2 分辨率)] │ ▼ (Pass 3: 13-tap 降采样) [高光纹理 Level 2 (1/4 分辨率)] │ ▼ (Pass 4: 13-tap 降采样) [高光纹理 Level 3 (1/8 分辨率)] │ │ (渐进升采样与帐篷滤波 Tent Filter 逐级混合) ▼ (Pass 5 ~ 7: 9-tap 升采样混合) [最终大光圈泛光贴图] │ ▼ (Final Pass: 与原图合成 ACES 色调映射) [屏幕最终输出]为什么必须分级降采样在 $1/8$ 分辨率的贴图上采样 1 个像素相当于在原始全屏画面上覆盖了巨大的物理物理光圈半径。通过从高分辨率到低分辨率逐级降采样Downsample再从低分辨率逐级升采样Upsample并与上一级叠加混合我们能以极小的采样次数合成出兼具“中心高锐度辉光”与“外围宽广柔和光晕”的物理真实光芒。WGSL 着色器实现一高光阈值提取Threshold Extract在第一阶段我们需要将画面中普通亮度的物体过滤掉只保留超过 HDR 阈值如亮度 $ 1.0$的发光像素并提供平滑软阈值Soft Knee避免明暗交界处产生阶跃闪烁Fireflies。// bloom_threshold.wgsl struct BloomThresholdUniforms { threshold: f32, // 基础阈值 (如 1.0) knee: f32, // 软拐点宽度 (如 0.5) }; group(0) binding(0) varuniform uniforms: BloomThresholdUniforms; group(0) binding(1) var inputTexture: texture_2df32; group(0) binding(2) var textureSampler: sampler; struct VertexOutput { builtin(position) position: vec4f32, location(0) uv: vec2f32, }; // 全屏三角顶点着色器 (无需顶点缓冲区) vertex fn vs_main(builtin(vertex_index) vertexIndex: u32) - VertexOutput { var out: VertexOutput; let uv vec2f32( f32((vertexIndex 1u) 2u), f32(vertexIndex 2u) ); out.position vec4f32(uv * 2.0 - 1.0, 0.0, 1.0); out.uv vec2f32(uv.x, 1.0 - uv.y); return out; } fragment fn fs_threshold(in: VertexOutput) - location(0) vec4f32 { let color textureSample(inputTexture, textureSampler, in.uv).rgb; // 计算感知亮度 (Rec. 709 亮度系数) let brightness max(color.r, max(color.g, color.b)); // 软阈值过渡计算 (Quadratic Knee) let soft clamp(brightness - uniforms.threshold uniforms.knee, 0.0, 2.0 * uniforms.knee); let softFactor (soft * soft) / (4.0 * uniforms.knee 1e-4); let contrib max(softFactor, brightness - uniforms.threshold) / max(brightness, 1e-4); let filteredColor color * max(0.0, contrib); return vec4f32(filteredColor, 1.0); }WGSL 着色器实现二13-Tap 抗走样降采样Downsample传统的单像素双线性采样在大幅度缩小图像时会丢失关键的高光孤岛导致画面在镜头移动时高光疯狂闪烁。这里我们采用 Jimenez 提出的13 点加权盒状采样法通过交错采样覆盖 36 个屏幕像素的几何重心// bloom_downsample.wgsl struct DownsampleUniforms { texelSize: vec2f32, // 1.0 / 上一级纹理分辨率 }; group(0) binding(0) varuniform uniforms: DownsampleUniforms; group(0) binding(1) var inputTexture: texture_2df32; group(0) binding(2) var linearSampler: sampler; fragment fn fs_downsample(in: VertexOutput) - location(0) vec4f32 { let uv in.uv; let d uniforms.texelSize; // 13 个关键采样点采样 (4 个内圈角点 4 个外圈角点 4 个边缘点 1 个中心点) let a textureSample(inputTexture, linearSampler, uv vec2f32(-2.0 * d.x, 2.0 * d.y)).rgb; let b textureSample(inputTexture, linearSampler, uv vec2f32(0.0, 2.0 * d.y)).rgb; let c textureSample(inputTexture, linearSampler, uv vec2f32(2.0 * d.x, 2.0 * d.y)).rgb; let dCorner textureSample(inputTexture, linearSampler, uv vec2f32(-2.0 * d.x, 0.0)).rgb; let e textureSample(inputTexture, linearSampler, uv).rgb; let f textureSample(inputTexture, linearSampler, uv vec2f32(2.0 * d.x, 0.0)).rgb; let g textureSample(inputTexture, linearSampler, uv vec2f32(-2.0 * d.x, -2.0 * d.y)).rgb; let h textureSample(inputTexture, linearSampler, uv vec2f32(0.0, -2.0 * d.y)).rgb; let i textureSample(inputTexture, linearSampler, uv vec2f32(2.0 * d.x, -2.0 * d.y)).rgb; let j textureSample(inputTexture, linearSampler, uv vec2f32(-d.x, d.y)).rgb; let k textureSample(inputTexture, linearSampler, uv vec2f32(d.x, d.y)).rgb; let l textureSample(inputTexture, linearSampler, uv vec2f32(-d.x, -d.y)).rgb; let m textureSample(inputTexture, linearSampler, uv vec2f32(d.x, -d.y)).rgb; // 加权平均 (Jimenez 13-tap 权重分布) var downsampleColor e * 0.125; downsampleColor (a c g i) * 0.03125; downsampleColor (b dCorner f h) * 0.0625; downsampleColor (j k l m) * 0.125; return vec4f32(downsampleColor, 1.0); }WGSL 着色器实现三9-Tap 升采样与帐篷滤波混合Upsample在升采样放大的过程中如果仅依靠硬件默认的双线性插值放大的色块会呈现菱形走样。我们使用 9-Tap 帐篷滤波核Tent Filter并将其以加法混合Additive Blending的方式逐级累加回高一级别的纹理缓冲区中// bloom_upsample.wgsl struct UpsampleUniforms { filterRadius: vec2f32, // 滤波扩散半径 (通常为 0.005) }; group(0) binding(0) varuniform uniforms: UpsampleUniforms; group(0) binding(1) var currentLevelTexture: texture_2df32; group(0) binding(2) var linearSampler: sampler; fragment fn fs_upsample(in: VertexOutput) - location(0) vec4f32 { let uv in.uv; let r uniforms.filterRadius; // 9-Tap 帐篷滤波采样核 (3x3 高斯逼近) let a textureSample(currentLevelTexture, linearSampler, uv vec2f32(-r.x, r.y)).rgb; let b textureSample(currentLevelTexture, linearSampler, uv vec2f32(0.0, r.y)).rgb; let c textureSample(currentLevelTexture, linearSampler, uv vec2f32(r.x, r.y)).rgb; let d textureSample(currentLevelTexture, linearSampler, uv vec2f32(-r.x, 0.0)).rgb; let e textureSample(currentLevelTexture, linearSampler, uv).rgb; let f textureSample(currentLevelTexture, linearSampler, uv vec2f32(r.x, 0.0)).rgb; let g textureSample(currentLevelTexture, linearSampler, uv vec2f32(-r.x, -r.y)).rgb; let h textureSample(currentLevelTexture, linearSampler, uv vec2f32(0.0, -r.y)).rgb; let i textureSample(currentLevelTexture, linearSampler, uv vec2f32(r.x, -r.y)).rgb; // 帐篷加权分布 var upsampleColor e * 4.0; upsampleColor (b d f h) * 2.0; upsampleColor (a c g i) * 1.0; upsampleColor * (1.0 / 16.0); return vec4f32(upsampleColor, 1.0); }最终合成与 ACES 色调映射Tone Mapping升采样迭代完成后我们得到了一张纯粹记录漫溢光晕的高精度浮点纹理。在最后一个 Composite Pass 中将原始渲染画面与 Bloom 贴图按强度系数相加并执行电影工业级ACES Filmic Tone Mapping将可能超过 1.0 的 HDR 能量平滑映射回 LDR 显示器所能呈现的[0, 1]范围// 电影级 ACES 胶片色调映射曲线 fn acesToneMapping(color: vec3f32) - vec3f32 { let a 2.51; let b 0.03; let c 2.43; let d 0.59; let e 0.14; return clamp((color * (a * color b)) / (color * (c * color d) e), vec3f32(0.0), vec3f32(1.0)); }显存复用与移动端极致能效在 WebGPU 中调度多级金字塔时最核心的性能铁律是严禁在每一帧重复创建纹理或 RenderPassDescriptor 对象。纹理复用池在引擎初始化阶段预先分配一套固定层级通常 5 层从 1/2 分辨率一直到 1/32 分辨率的双向纹理链Mip-Chain整个渲染生命周期保持常驻利用 RenderPass 的清晰语义在升采样的每个阶段将混合模式配置为blend: { color: { operation: add, srcFactor: one, dstFactor: one } }直接让 GPU 硬件固定功能管线执行纹理累加节省昂贵的片段着色器手动双纹理读取带宽。通过手写完整的 WGSL 双向金字塔 Bloom 管线我们在现代浏览器中复现了端游级绚烂、深邃、呼吸感十足的光影漫溢效果。每一束穿透黑夜的高光都在精确的算力调度下绽放出震撼人心的光学之美。
返回列表