载入中...
搜索中...
未找到
GraphicsGpuDriven.cpp
浏览该文件的文档.
1// Vulkan backend implementation — GPU-driven rendering (stages 0-3).
2//
3// Bindless resources, GPU tables, per-frame arena, CPU/GPU indirect draws,
4// HZB + frustum cull chain, visibility-buffer pass, fullscreen resolve and
5// VirtualGeometry hardware-raster integration. Kept in its own TU (matching
6// dev's split-graphics-backend layout) so the feature stays reviewable.
7
11#include "common/Assert.h"
13#include "graphics/Light.h"
14#include "graphics/Material.h"
16#include "graphics/Shadow.h"
17
18#include <SDL2/SDL.h>
19
20#include <algorithm>
21#include <array>
22#include <cmath>
23#include <cstdio>
24#include <cstdlib>
25#include <cstring>
26#include <functional>
27#include <stdexcept>
28#include <string>
29#include <vector>
30
31#include "common/Exception.h"
33#include "zeroerr/assert.h"
34
35#include <glm/gtc/matrix_transform.hpp>
36
37#include "graphics/shaders/gpu_emit_comp_spv.inc"
38#include "graphics/shaders/vg_main_cull_comp_spv.inc"
39#include "graphics/shaders/hzb_build_comp_spv.inc"
40#include "graphics/shaders/gpu_cull_comp_spv.inc"
41#include "graphics/shaders/mesh3d_gpudriven_vert_spv.inc"
42#include "graphics/shaders/mesh3d_gpudriven_frag_spv.inc"
43#include "graphics/shaders/mesh3d_gbuffer_vis_vert_spv.inc"
44#include "graphics/shaders/mesh3d_gbuffer_vis_frag_spv.inc"
45#include "graphics/shaders/mesh3d_gbuffer_vgvis_vert_spv.inc"
46#include "graphics/shaders/mesh3d_gbuffer_vgvis_frag_spv.inc"
47#include "graphics/shaders/resolve_vis_vert_spv.inc"
48#include "graphics/shaders/resolve_vis_frag_spv.inc"
49
50namespace eve::graphics::vulkan {
51
53 return gpuDrivenEnabled_ && gpuDrivenCaps_.gpuDrivenCullAvailable() &&
54 renderControl_ && renderControl_->isEnabled("visResolve") &&
55 sceneColorSamples == vk::SampleCountFlagBits::e1;
56}
57
58vk::DescriptorSet Graphics::bindlessSetForFrame() const {
59 if (bindlessSets_.empty()) return nullptr;
60 return bindlessSets_[currentFrameSlot() % bindlessSets_.size()];
61}
62
63void Graphics::initGpuDrivenResources() {
64 if (!gpuDrivenCaps_.gpuDrivenAvailable()) return;
65 frameArenas_.resize(frameSlotCount());
66 for (auto &arena : frameArenas_) {
67 arena.ensure(device, 8u << 20,
68 vk::BufferUsageFlagBits::eStorageBuffer |
69 vk::BufferUsageFlagBits::eIndirectBuffer |
70 vk::BufferUsageFlagBits::eTransferDst);
71 }
72 createBindlessSet();
73 createMesh3DGpuDrivenPipeline();
74}
75
76void Graphics::createGpuDrivenVisResources(int width, int height) {
77 if (!gpuDrivenCaps_.gpuDrivenAvailable()) return;
78 if (gbufferVisRenderPass || gbufferSlots.empty()) return;
79 const uint32_t w = uint32_t(width);
80 const uint32_t h = uint32_t(height);
81 const vk::Format colorFmt = pickGBufferColorFormat(device);
82 const vk::Format depthFmt = vk::Format::eD32Sfloat;
83 const vk::Format visIDFmt = vk::Format::eR32G32Uint;
84 const vk::Format visBaryFmt = vk::Format::eR16G16Sfloat;
85
86 gbufferVisRenderPass =
87 device.createRenderPass()
88 .addSampledColorAttachment(colorFmt)
89 .addSampledColorAttachment(colorFmt)
90 .addSampledColorAttachment(colorFmt)
91 .addSampledColorAttachment(visIDFmt)
92 .addSampledColorAttachment(visBaryFmt)
93 .addSampledDepthAttachment(depthFmt)
94 .addSubpass(vkb::SubpassBuilder()
95 .addAttachmentRef(0, vk::ImageLayout::eColorAttachmentOptimal)
96 .addAttachmentRef(1, vk::ImageLayout::eColorAttachmentOptimal)
97 .addAttachmentRef(2, vk::ImageLayout::eColorAttachmentOptimal)
98 .addAttachmentRef(3, vk::ImageLayout::eColorAttachmentOptimal)
99 .addAttachmentRef(4, vk::ImageLayout::eColorAttachmentOptimal)
100 .setDepthStencilAttachment(
101 5, vk::ImageLayout::eDepthStencilAttachmentOptimal))
102 .addExternalShaderReadDependencies()
103 .build();
104 for (auto &slot : gbufferSlots) {
105 slot.visFramebuffer = gbufferVisRenderPass.createFramebuffer(
106 device, w, h,
107 {slot.normal.asAttachment(), slot.depthColor.asAttachment(), slot.albedo.asAttachment(),
108 slot.visID.asAttachment(), slot.visBary.asAttachment(), slot.depth.asAttachment()});
109 }
110
111 if (!mesh3dGpuDrivenPipelineLayout) return;
112 // Instance vis pass: no vertex input (fetches from the pooled buffers via
113 // gl_VertexIndex); draws non-indexed with the cull chain's NI commands.
114 {
115 std::vector<uint32_t> visVert(mesh3d_gbuffer_vis_vert_spv,
116 mesh3d_gbuffer_vis_vert_spv +
117 mesh3d_gbuffer_vis_vert_spv_count);
118 std::vector<uint32_t> visFrag(mesh3d_gbuffer_vis_frag_spv,
119 mesh3d_gbuffer_vis_frag_spv +
120 mesh3d_gbuffer_vis_frag_spv_count);
121 vk::ShaderModule visVertModule =
122 vkb::PipelineBuilder::createShaderModule(device.instance, visVert);
123 vk::ShaderModule visFragModule =
124 vkb::PipelineBuilder::createShaderModule(device.instance, visFrag);
125 gbufferVisPipeline =
126 device.createPipeline()
127 .useClassicPipeline(visVertModule, visFragModule)
128 .setPipelineLayout(mesh3dGpuDrivenPipelineLayout)
129 .setDynamicStatesViewportScissor()
130 .setRasterizer(vk::PolygonMode::eFill, false, false, 1.0f,
131 vk::CullModeFlagBits::eNone, vk::FrontFace::eClockwise)
132 .setDepthStencil(true, true, vk::CompareOp::eLess)
133 .setColorAttachmentCount(5)
134 .build(gbufferVisRenderPass);
135 device->destroyShaderModule(visVertModule);
136 device->destroyShaderModule(visFragModule);
137 }
138 // VG variant: same vis attachments, cluster-stream fetch (no vertex input).
139 {
140 std::vector<uint32_t> vgVisVert(mesh3d_gbuffer_vgvis_vert_spv,
141 mesh3d_gbuffer_vgvis_vert_spv +
142 mesh3d_gbuffer_vgvis_vert_spv_count);
143 std::vector<uint32_t> vgVisFrag(mesh3d_gbuffer_vgvis_frag_spv,
144 mesh3d_gbuffer_vgvis_frag_spv +
145 mesh3d_gbuffer_vgvis_frag_spv_count);
146 vk::ShaderModule vgVisVertModule =
147 vkb::PipelineBuilder::createShaderModule(device.instance, vgVisVert);
148 vk::ShaderModule vgVisFragModule =
149 vkb::PipelineBuilder::createShaderModule(device.instance, vgVisFrag);
150 gbufferVgVisPipeline =
151 device.createPipeline()
152 .useClassicPipeline(vgVisVertModule, vgVisFragModule)
153 .setPipelineLayout(mesh3dGpuDrivenPipelineLayout)
154 .setDynamicStatesViewportScissor()
155 .setRasterizer(vk::PolygonMode::eFill, false, false, 1.0f,
156 vk::CullModeFlagBits::eNone, vk::FrontFace::eClockwise)
157 .setDepthStencil(true, true, vk::CompareOp::eLess)
158 .setColorAttachmentCount(5)
159 .build(gbufferVisRenderPass);
160 device->destroyShaderModule(vgVisVertModule);
161 device->destroyShaderModule(vgVisFragModule);
162 }
163}
164
165// ---- GPU-driven (stage 0): bindless set + per-frame arena + tables ----
166
168 if (frameArenas_.empty()) {
169 static FrameArena fallback; // never used for recording; guards odd call order
170 return fallback;
171 }
172 return frameArenas_[currentFrameSlot() % frameArenas_.size()];
173}
174
175void Graphics::createBindlessSet() {
176 if (!gpuDrivenCaps_.gpuDrivenAvailable()) return;
177 if (!whiteTexture || !whiteTexture->gpuHandle) return;
178 if (!defaultBindlessCube || !defaultBindlessCube->gpuHandle) return;
179 auto *white = static_cast<GpuTexture *>(whiteTexture->gpuHandle);
180 auto *whiteCube = static_cast<GpuTexture *>(defaultBindlessCube->gpuHandle);
181
182 auto samplerBinding = [](uint32_t binding, uint32_t count, vk::ShaderStageFlags stages) {
183 vk::DescriptorSetLayoutBinding b{};
184 b.binding = binding;
185 b.descriptorType = vk::DescriptorType::eCombinedImageSampler;
186 b.descriptorCount = count;
187 b.stageFlags = stages;
188 return b;
189 };
190 auto storageBinding = [](uint32_t binding, vk::ShaderStageFlags stages) {
191 vk::DescriptorSetLayoutBinding b{};
192 b.binding = binding;
193 b.descriptorType = vk::DescriptorType::eStorageBuffer;
194 b.descriptorCount = 1;
195 b.stageFlags = stages;
196 return b;
197 };
198 auto uniformBinding = [](uint32_t binding, vk::ShaderStageFlags stages) {
199 vk::DescriptorSetLayoutBinding b{};
200 b.binding = binding;
201 b.descriptorType = vk::DescriptorType::eUniformBuffer;
202 b.descriptorCount = 1;
203 b.stageFlags = stages;
204 return b;
205 };
206 const vk::ShaderStageFlags allStages = vk::ShaderStageFlagBits::eVertex |
207 vk::ShaderStageFlagBits::eFragment |
208 vk::ShaderStageFlagBits::eCompute;
209 const vk::ShaderStageFlags computeOnly = vk::ShaderStageFlagBits::eCompute;
210 const vk::ShaderStageFlags vertFrag =
211 vk::ShaderStageFlagBits::eVertex | vk::ShaderStageFlagBits::eFragment;
212 std::vector<vk::DescriptorSetLayoutBinding> bindings{
213 samplerBinding(0, kMaxBindlessTextures, allStages),
214 samplerBinding(1, kMaxBindlessCubemaps, allStages),
215 storageBinding(2, allStages),
216 storageBinding(3, allStages),
217 storageBinding(4, allStages),
218 storageBinding(5, allStages), // unused by shaders; kept valid
219 uniformBinding(6, computeOnly),
220 storageBinding(7, computeOnly),
221 storageBinding(8, computeOnly),
222 storageBinding(9, computeOnly),
223 storageBinding(10, computeOnly),
224 samplerBinding(11, 1, computeOnly),
225 storageBinding(12, computeOnly),
226 storageBinding(13, computeOnly),
227 storageBinding(14, computeOnly),
228 samplerBinding(15, 1, vk::ShaderStageFlagBits::eFragment), // visID (resolve)
229 samplerBinding(16, 1, vk::ShaderStageFlagBits::eFragment), // visBary (resolve)
230 storageBinding(17, computeOnly),
231 storageBinding(18, vertFrag), // pooled vertex positions
232 storageBinding(19, vertFrag), // pooled vertex normals
233 storageBinding(20, vertFrag), // pooled vertex uvs
234 storageBinding(21, vertFrag), // pooled indices
235 storageBinding(22, allStages), // VG position stream (float xyz)
236 storageBinding(23, allStages), // VG triangle stream (u32)
237 storageBinding(24, allStages), // VG cluster table (uvec4 x 4)
238 storageBinding(25, allStages), // VG cluster -> asset id
239 storageBinding(26, allStages), // VG visible list (per slot)
240 storageBinding(27, allStages), // VG indirect commands (per slot)
241 storageBinding(28, vk::ShaderStageFlagBits::eFragment), // VG asset -> material
242 storageBinding(29, vk::ShaderStageFlagBits::eVertex |
243 vk::ShaderStageFlagBits::eFragment |
244 vk::ShaderStageFlagBits::eCompute), // VG asset models
245 storageBinding(30, computeOnly), // non-indexed indirect commands (vis pass)
246 };
247
248 vk::DescriptorSetLayoutCreateInfo layoutInfo{};
249 layoutInfo.bindingCount = uint32_t(bindings.size());
250 layoutInfo.pBindings = bindings.data();
251 bindlessSetLayoutUnique_ = device->createDescriptorSetLayoutUnique(layoutInfo);
252 bindlessSetLayout_ = *bindlessSetLayoutUnique_;
253
254 // kAsyncResourceCopies sets: one per frame-in-flight slot, so frame N+1 can
255 // rewrite its own set while frame N's pending command buffers still hold
256 // the previous contents of the other set (no UPDATE_AFTER_BIND required).
257 const uint32_t setCount = kAsyncResourceCopies;
258 std::array<vk::DescriptorPoolSize, 3> poolSizes{
259 vk::DescriptorPoolSize{vk::DescriptorType::eCombinedImageSampler,
260 (kMaxBindlessTextures + kMaxBindlessCubemaps + 3) * setCount},
261 vk::DescriptorPoolSize{vk::DescriptorType::eStorageBuffer, 25 * setCount},
262 vk::DescriptorPoolSize{vk::DescriptorType::eUniformBuffer, 1 * setCount},
263 };
264 vk::DescriptorPoolCreateInfo poolInfo{};
265 poolInfo.maxSets = setCount;
266 poolInfo.poolSizeCount = uint32_t(poolSizes.size());
267 poolInfo.pPoolSizes = poolSizes.data();
268 bindlessPool_ = device->createDescriptorPool(poolInfo);
269
270 {
271 std::vector<vk::DescriptorSetLayout> layouts(setCount, bindlessSetLayout_);
272 vk::DescriptorSetAllocateInfo alloc{};
273 alloc.descriptorPool = bindlessPool_;
274 alloc.descriptorSetCount = setCount;
275 alloc.pSetLayouts = layouts.data();
276 bindlessSets_ = device->allocateDescriptorSets(alloc);
277 }
278
279 bindlessTextures2D_.assign(kMaxBindlessTextures, white);
280 bindlessCubemaps_.assign(kMaxBindlessCubemaps, whiteCube);
281 bindlessFree2D_.clear();
282 bindlessFreeCube_.clear();
283 bindlessFree2D_.reserve(kMaxBindlessTextures);
284 bindlessFreeCube_.reserve(kMaxBindlessCubemaps);
285 for (uint32_t i = 0; i < kMaxBindlessTextures; ++i) bindlessFree2D_.push_back(i);
286 for (uint32_t i = 0; i < kMaxBindlessCubemaps; ++i) bindlessFreeCube_.push_back(i);
287
288 vk::DescriptorImageInfo white2D{white->sampler, white->imageView(),
289 vk::ImageLayout::eShaderReadOnlyOptimal};
290 vk::DescriptorImageInfo whiteCubeInfo{whiteCube->sampler, whiteCube->cubeImage.imageView(),
291 vk::ImageLayout::eShaderReadOnlyOptimal};
292 // Placeholder fill: single-element writes keep the per-call update tiny;
293 // ~1.1k single-element calls per set at init cost ~6 ms, which is fine.
294 constexpr uint32_t kDescriptorChunk = 1;
295 auto chunkFill = [&](vk::DescriptorSet set, uint32_t binding, uint32_t total,
296 const vk::DescriptorImageInfo &info) {
297 for (uint32_t base = 0; base < total; base += kDescriptorChunk) {
298 vk::WriteDescriptorSet w{};
299 w.dstSet = set;
300 w.dstBinding = binding;
301 w.dstArrayElement = base;
302 w.descriptorCount = std::min(kDescriptorChunk, total - base);
303 w.descriptorType = vk::DescriptorType::eCombinedImageSampler;
304 w.pImageInfo = &info;
305 device->updateDescriptorSets(1, &w, 0, nullptr);
306 }
307 };
308 for (vk::DescriptorSet set : bindlessSets_) {
309 chunkFill(set, 0, kMaxBindlessTextures, white2D);
310 chunkFill(set, 1, kMaxBindlessCubemaps, whiteCubeInfo);
311 }
312
313 // GPU resource tables (fixed capacity at startup; entries registered lazily).
314 constexpr uint32_t kMaxMeshRecords = 4096;
315 constexpr uint32_t kMaxMaterialRecords = 1024;
316 meshTableBuffer_ = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer,
317 kMaxMeshRecords * sizeof(GpuMeshRecord),
318 kHostVisibleCoherent);
319 meshTableCapacity_ = kMaxMeshRecords;
320 meshTableRecords_.reserve(kMaxMeshRecords);
321 materialTableBuffer_ = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer,
322 kMaxMaterialRecords * sizeof(GpuMaterialRecord),
323 kHostVisibleCoherent);
324 materialTableCapacity_ = kMaxMaterialRecords;
325 materialTableRecords_.reserve(kMaxMaterialRecords);
326
327 // Bind the resource tables into the bindless set (bindings 2-5). Binding 4
328 // (per-frame instances) is rewritten before each frame; these writes make
329 // every binding valid so the set is safe to bind at any time.
330 auto tableWrite = [&](vk::DescriptorSet set, uint32_t binding, vk::Buffer buffer) {
331 vk::DescriptorBufferInfo info{buffer, 0, VK_WHOLE_SIZE};
332 vk::WriteDescriptorSet w{};
333 w.dstSet = set;
334 w.dstBinding = binding;
335 w.descriptorCount = 1;
336 w.descriptorType = vk::DescriptorType::eStorageBuffer;
337 w.pBufferInfo = &info;
338 device->updateDescriptorSets(1, &w, 0, nullptr);
339 };
340 for (vk::DescriptorSet set : bindlessSets_) {
341 tableWrite(set, 2, meshTableBuffer_.buffer);
342 tableWrite(set, 3, materialTableBuffer_.buffer);
343 tableWrite(set, 4, meshTableBuffer_.buffer); // placeholder; rewritten per frame
344 tableWrite(set, 5, meshTableBuffer_.buffer); // unused by shaders
345 }
346 // Stage 2 placeholders: every binding must be valid before the set binds.
347 gpuDrivenCullParamsPlaceholder_ =
348 vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eUniformBuffer, 256,
349 kHostVisibleCoherent);
350 {
351 vk::DescriptorBufferInfo ubo{gpuDrivenCullParamsPlaceholder_.buffer, 0, 256};
352 for (vk::DescriptorSet set : bindlessSets_) {
353 vk::WriteDescriptorSet w{};
354 w.dstSet = set;
355 w.dstBinding = 6;
356 w.descriptorCount = 1;
357 w.descriptorType = vk::DescriptorType::eUniformBuffer;
358 w.pBufferInfo = &ubo;
359 device->updateDescriptorSets(1, &w, 0, nullptr);
360 }
361 }
362 for (uint32_t b : {7u, 8u, 9u, 10u, 12u, 13u, 14u, 17u})
363 for (vk::DescriptorSet set : bindlessSets_) tableWrite(set, b, meshTableBuffer_.buffer);
364 {
365 vk::DescriptorImageInfo depthInfo{white->sampler, white->imageView(),
366 vk::ImageLayout::eShaderReadOnlyOptimal};
367 for (vk::DescriptorSet set : bindlessSets_) {
368 vk::WriteDescriptorSet w{};
369 w.dstSet = set;
370 w.dstBinding = 11;
371 w.descriptorCount = 1;
372 w.descriptorType = vk::DescriptorType::eCombinedImageSampler;
373 w.pImageInfo = &depthInfo;
374 device->updateDescriptorSets(1, &w, 0, nullptr);
375 }
376 }
377 // Stage 3 placeholders: visID/visBary are rewritten per frame by the
378 // resolve; keep them valid so the set is safe to bind anywhere.
379 for (vk::DescriptorSet set : bindlessSets_) {
380 vk::DescriptorImageInfo visInfo{white->sampler, white->imageView(),
381 vk::ImageLayout::eShaderReadOnlyOptimal};
382 vk::WriteDescriptorSet w{};
383 w.dstSet = set;
384 w.dstBinding = 15;
385 w.descriptorCount = 1;
386 w.descriptorType = vk::DescriptorType::eCombinedImageSampler;
387 w.pImageInfo = &visInfo;
388 device->updateDescriptorSets(1, &w, 0, nullptr);
389 w.dstBinding = 16;
390 device->updateDescriptorSets(1, &w, 0, nullptr);
391 }
392 // Pooled vertex/index buffers for the vis resolve (grows lazily).
393 ensureGpuVertexPool();
394 bindGpuVertexPoolBindless();
395 // Shared virtual-geometry cluster pool (grows lazily on upload).
396 ensureVgBuffers();
397 bindVgPoolBindless();
398}
399
400uint32_t Graphics::registerBindlessTexture2D(GpuTexture *tex) {
401 if (!tex || bindlessSets_.empty()) return kInvalidBindlessSlot;
402 if (tex->bindlessIndex2D != kInvalidBindlessSlot) return tex->bindlessIndex2D;
403 if (bindlessFree2D_.empty()) return kInvalidBindlessSlot;
404 const uint32_t slot = bindlessFree2D_.front();
405 bindlessFree2D_.erase(bindlessFree2D_.begin());
406 bindlessTextures2D_[slot] = tex;
407 tex->bindlessIndex2D = slot;
408 vk::DescriptorImageInfo img{tex->sampler, tex->imageView(),
409 vk::ImageLayout::eShaderReadOnlyOptimal};
410 for (vk::DescriptorSet set : bindlessSets_) {
411 vk::WriteDescriptorSet write{};
412 write.dstSet = set;
413 write.dstBinding = 0;
414 write.dstArrayElement = slot;
415 write.descriptorCount = 1;
416 write.descriptorType = vk::DescriptorType::eCombinedImageSampler;
417 write.pImageInfo = &img;
418 device->updateDescriptorSets(1, &write, 0, nullptr);
419 }
420 return slot;
421}
422
423uint32_t Graphics::registerBindlessTextureCube(GpuTexture *tex) {
424 if (!tex || bindlessSets_.empty()) return kInvalidBindlessSlot;
425 if (tex->bindlessIndexCube != kInvalidBindlessSlot) return tex->bindlessIndexCube;
426 if (bindlessFreeCube_.empty()) return kInvalidBindlessSlot;
427 const uint32_t slot = bindlessFreeCube_.front();
428 bindlessFreeCube_.erase(bindlessFreeCube_.begin());
429 bindlessCubemaps_[slot] = tex;
430 tex->bindlessIndexCube = slot;
431 vk::DescriptorImageInfo img{tex->sampler, tex->imageView(),
432 vk::ImageLayout::eShaderReadOnlyOptimal};
433 for (vk::DescriptorSet set : bindlessSets_) {
434 vk::WriteDescriptorSet write{};
435 write.dstSet = set;
436 write.dstBinding = 1;
437 write.dstArrayElement = slot;
438 write.descriptorCount = 1;
439 write.descriptorType = vk::DescriptorType::eCombinedImageSampler;
440 write.pImageInfo = &img;
441 device->updateDescriptorSets(1, &write, 0, nullptr);
442 }
443 return slot;
444}
445
447 if (!cubemap || !cubemap->gpuHandle) return kInvalidGpuDrivenSlot;
448 auto *gpu = static_cast<GpuTexture *>(cubemap->gpuHandle);
449 if (!gpu->isCube) return kInvalidGpuDrivenSlot;
450 return registerBindlessTextureCube(gpu);
451}
452
453void Graphics::unregisterBindlessTexture(GpuTexture *tex) {
454 if (!tex || bindlessSets_.empty()) return;
455 auto *white = static_cast<GpuTexture *>(whiteTexture->gpuHandle);
456 GpuTexture *whiteCube = white;
457 if (defaultBindlessCube && defaultBindlessCube->gpuHandle)
458 whiteCube = static_cast<GpuTexture *>(defaultBindlessCube->gpuHandle);
459
460 auto restore = [&](uint32_t binding, uint32_t slot, GpuTexture *placeholder,
461 bool cube) {
462 vk::DescriptorImageInfo img{
463 placeholder->sampler,
464 cube ? placeholder->cubeImage.imageView() : placeholder->imageView(),
465 vk::ImageLayout::eShaderReadOnlyOptimal};
466 for (vk::DescriptorSet set : bindlessSets_) {
467 vk::WriteDescriptorSet write{};
468 write.dstSet = set;
469 write.dstBinding = binding;
470 write.dstArrayElement = slot;
471 write.descriptorCount = 1;
472 write.descriptorType = vk::DescriptorType::eCombinedImageSampler;
473 write.pImageInfo = &img;
474 device->updateDescriptorSets(1, &write, 0, nullptr);
475 }
476 };
477
479 const uint32_t slot = tex->bindlessIndex2D;
480 bindlessTextures2D_[slot] = white;
481 bindlessFree2D_.push_back(slot);
482 restore(0, slot, white, false);
484 }
486 const uint32_t slot = tex->bindlessIndexCube;
487 bindlessCubemaps_[slot] = whiteCube;
488 bindlessFreeCube_.push_back(slot);
489 restore(1, slot, whiteCube, true);
491 }
492}
493
494void Graphics::ensureVgBuffers() {
495 if (vgGpu_.positions.buffer) return;
496 const auto hostMem = kHostVisibleCoherent;
497 constexpr uint32_t kInitClusters = 16384;
498 constexpr uint32_t kInitVertFloats = 1u << 20; // ~1M floats (~4MB)
499 constexpr uint32_t kInitTriangles = 2u << 20; // ~2M u32 indices
500 vgGpu_.positions = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer,
501 kInitVertFloats * sizeof(float), hostMem);
502 vgGpu_.triangles = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer,
503 kInitTriangles * sizeof(uint32_t), hostMem);
504 vgGpu_.clusters = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer,
505 kInitClusters * sizeof(GpuVgCluster), hostMem);
506 vgGpu_.clusterAssets =
507 vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer,
508 kInitClusters * sizeof(uint32_t), hostMem);
509 constexpr uint32_t kVisBytes = (kMaxVgClusters + 1) * sizeof(uint32_t);
510 constexpr uint32_t kIndBytes = kMaxVgClusters * sizeof(glm::uvec4);
511 vgGpu_.visible = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer |
512 vk::BufferUsageFlagBits::eIndirectBuffer,
513 kVisBytes * kAsyncResourceCopies, hostMem);
514 vgGpu_.indirect = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer |
515 vk::BufferUsageFlagBits::eIndirectBuffer,
516 kIndBytes * kAsyncResourceCopies, hostMem);
517 vgGpu_.assetMaterials =
518 vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer,
519 kMaxVgAssets * sizeof(uint32_t) * kAsyncResourceCopies, hostMem);
520 vgGpu_.assetModels =
521 vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer,
522 kMaxVgAssets * sizeof(glm::mat4) * kAsyncResourceCopies, hostMem);
523 {
524 void *m = vgGpu_.visible.map();
525 std::memset(m, 0, kVisBytes * kAsyncResourceCopies);
526 vgGpu_.visible.unmap();
527 m = vgGpu_.assetMaterials.map();
528 std::memset(m, 0, kMaxVgAssets * sizeof(uint32_t) * kAsyncResourceCopies);
529 vgGpu_.assetMaterials.unmap();
530 m = vgGpu_.assetModels.map();
531 for (size_t s = 0; s < kAsyncResourceCopies; ++s)
532 for (uint32_t a = 0; a < kMaxVgAssets; ++a)
533 static_cast<glm::mat4 *>(m)[s * kMaxVgAssets + a] = glm::mat4(1.f);
534 vgGpu_.assetModels.unmap();
535 }
536 bindVgPoolBindless();
537}
538
539void Graphics::growVgBuffers(uint32_t needClusters, uint32_t needVertices,
540 uint32_t needTriangles) {
541 device->waitIdle();
542 const auto hostMem = kHostVisibleCoherent;
543 auto growBuffer = [&](vkb::GenericBuffer &dst, uint32_t elementSize, uint32_t oldElems,
544 uint32_t newElems) {
545 if (newElems <= oldElems) return;
546 vkb::GenericBuffer grown(device, vk::BufferUsageFlagBits::eStorageBuffer,
547 vk::DeviceSize(newElems) * elementSize, hostMem);
548 if (oldElems > 0) {
549 void *srcMap = dst.map();
550 void *dstMap = grown.map();
551 std::memcpy(dstMap, srcMap, vk::DeviceSize(oldElems) * elementSize);
552 grown.unmap();
553 dst.unmap();
554 }
555 dst.release();
556 dst = std::move(grown);
557 };
558 const uint32_t curClusters = uint32_t(vgGpu_.clusters.size / sizeof(GpuVgCluster));
559 const uint32_t curVertFloats = uint32_t(vgGpu_.positions.size / sizeof(float));
560 const uint32_t curTriangles = uint32_t(vgGpu_.triangles.size / sizeof(uint32_t));
561 growBuffer(vgGpu_.clusters, uint32_t(sizeof(GpuVgCluster)), curClusters,
562 std::max(needClusters, curClusters * 2u));
563 growBuffer(vgGpu_.clusterAssets, uint32_t(sizeof(uint32_t)), curClusters,
564 std::max(needClusters, curClusters * 2u));
565 growBuffer(vgGpu_.positions, uint32_t(sizeof(float)), curVertFloats,
566 std::max(needVertices, curVertFloats * 2u));
567 growBuffer(vgGpu_.triangles, uint32_t(sizeof(uint32_t)), curTriangles,
568 std::max(needTriangles, curTriangles * 2u));
569 bindVgPoolBindless();
570}
571
572void Graphics::bindVgPoolBindless() {
573 if (bindlessSets_.empty() || !vgGpu_.positions.buffer) return;
574 auto bufWrite = [&](vk::DescriptorSet set, uint32_t binding, vk::Buffer buffer,
575 vk::DeviceSize offset, vk::DeviceSize size) {
576 vk::DescriptorBufferInfo info{buffer, offset, size};
577 vk::WriteDescriptorSet w{};
578 w.dstSet = set;
579 w.dstBinding = binding;
580 w.descriptorCount = 1;
581 w.descriptorType = vk::DescriptorType::eStorageBuffer;
582 w.pBufferInfo = &info;
583 device->updateDescriptorSets(1, &w, 0, nullptr);
584 };
585 const vk::DeviceSize posBytes = vgGpu_.positions.size;
586 const vk::DeviceSize triBytes = vgGpu_.triangles.size;
587 const vk::DeviceSize clBytes = vgGpu_.clusters.size;
588 const vk::DeviceSize claBytes = vgGpu_.clusterAssets.size;
589 for (vk::DescriptorSet set : bindlessSets_) {
590 bufWrite(set, 22, vgGpu_.positions.buffer, 0, posBytes);
591 bufWrite(set, 23, vgGpu_.triangles.buffer, 0, triBytes);
592 bufWrite(set, 24, vgGpu_.clusters.buffer, 0, clBytes);
593 bufWrite(set, 25, vgGpu_.clusterAssets.buffer, 0, claBytes);
594 bufWrite(set, 28, vgGpu_.assetMaterials.buffer, 0, vgGpu_.assetMaterials.size);
595 bufWrite(set, 29, vgGpu_.assetModels.buffer, 0, vgGpu_.assetModels.size);
596 bufWrite(set, 26, vgGpu_.visible.buffer, 0, vgGpu_.visible.size);
597 bufWrite(set, 27, vgGpu_.indirect.buffer, 0, vgGpu_.indirect.size);
598 }
599}
600
601void Graphics::bindVgFrameBindless(vk::DescriptorSet bindless, size_t slot) {
602 if (!bindless || !vgGpu_.visible.buffer) return;
603 const vk::DeviceSize slotVis = (vk::DeviceSize(kMaxVgClusters) + 1) * sizeof(uint32_t);
604 const vk::DeviceSize slotInd = vk::DeviceSize(kMaxVgClusters) * sizeof(glm::uvec4);
605 const vk::DeviceSize slotMat = vk::DeviceSize(kMaxVgAssets) * sizeof(uint32_t);
606 const vk::DeviceSize slotModel = vk::DeviceSize(kMaxVgAssets) * sizeof(glm::mat4);
607 const vk::DeviceSize visOff = slotVis * slot;
608 const vk::DeviceSize indOff = slotInd * slot;
609 const vk::DeviceSize matOff = slotMat * slot;
610 const vk::DeviceSize modelOff = slotModel * slot;
611 vk::DescriptorBufferInfo infos[4]{
612 {vgGpu_.visible.buffer, visOff, slotVis},
613 {vgGpu_.indirect.buffer, indOff, slotInd},
614 {vgGpu_.assetMaterials.buffer, matOff, slotMat},
615 {vgGpu_.assetModels.buffer, modelOff, slotModel},
616 };
617 vk::WriteDescriptorSet w[4]{};
618 const uint32_t bindings[4] = {26, 27, 28, 29};
619 for (int i = 0; i < 4; ++i) {
620 w[i].dstSet = bindless;
621 w[i].dstBinding = bindings[i];
622 w[i].descriptorCount = 1;
623 w[i].descriptorType = vk::DescriptorType::eStorageBuffer;
624 w[i].pBufferInfo = &infos[i];
625 }
626 device->updateDescriptorSets(4, w, 0, nullptr);
627}
628
630 // The compacted cluster command stream needs an indirect-count draw. Keep
631 // ordinary GPU-driven meshes available on devices without that optional
632 // Vulkan 1.2 feature, but reject VG explicitly instead of accepting an
633 // asset that can never be rasterized.
634 if (!gpuDrivenCaps_.drawIndirectCount) return kInvalidBindlessSlot;
635 if (!asset.positions || asset.vertexCount <= 0 || !asset.triangles ||
636 asset.triangleCount <= 0 || !asset.clusters || asset.clusterCount <= 0)
638 if (vgAssetCount_ >= kMaxVgAssets) return kInvalidBindlessSlot;
639 ensureVgBuffers();
640
641 const uint32_t assetId = vgAssetCount_;
642 const uint32_t vertBase = vgVertexCount_; // global vertex index base
643 const uint32_t triBase = vgTriangleCount_; // global index-stream base
644 const uint32_t clusterBase = vgClusterCount_;
645 const uint32_t needClusters = clusterBase + uint32_t(asset.clusterCount);
646 const uint32_t needVerts = uint32_t(vertBase + uint32_t(asset.vertexCount)) * 3;
647 const uint32_t needTris = triBase + uint32_t(asset.triangleCount);
648 const uint32_t curClusters = uint32_t(vgGpu_.clusters.size / sizeof(GpuVgCluster));
649 const uint32_t curVerts = uint32_t(vgGpu_.positions.size / sizeof(float));
650 const uint32_t curTris = uint32_t(vgGpu_.triangles.size / sizeof(uint32_t));
651 if (needClusters > curClusters || needVerts > curVerts || needTris > curTris)
652 growVgBuffers(needClusters, needVerts, needTris);
653
654 void *posMap = vgGpu_.positions.map();
655 void *triMap = vgGpu_.triangles.map();
656 void *clMap = vgGpu_.clusters.map();
657 void *claMap = vgGpu_.clusterAssets.map();
658 if (!posMap || !triMap || !clMap || !claMap) {
659 if (posMap) vgGpu_.positions.unmap();
660 if (triMap) vgGpu_.triangles.unmap();
661 if (clMap) vgGpu_.clusters.unmap();
662 if (claMap) vgGpu_.clusterAssets.unmap();
664 }
665 std::memcpy(static_cast<char *>(posMap) + size_t(vertBase) * 3 * sizeof(float), asset.positions,
666 size_t(asset.vertexCount) * 3 * sizeof(float));
667 {
668 auto *dst = static_cast<uint32_t *>(triMap) + triBase;
669 for (int i = 0; i < asset.triangleCount; ++i) dst[i] = asset.triangles[i] + vertBase;
670 }
671 std::memcpy(static_cast<char *>(clMap) + vgClusterCount_ * sizeof(GpuVgCluster),
672 asset.clusters, size_t(asset.clusterCount) * sizeof(GpuVgCluster));
673 {
674 auto *dst = static_cast<GpuVgCluster *>(clMap) + vgClusterCount_;
675 for (int i = 0; i < asset.clusterCount; ++i) {
676 // Cluster triStart is a triangle ordinal; vgTriangleCount_ tracks
677 // uint32 indices. The vertex/resolve shaders multiply triStart by
678 // three, so convert the index offset before rebasing the cluster.
679 dst[i].u1[0] += triBase / 3u;
680 }
681 }
682 auto *cla = static_cast<uint32_t *>(claMap);
683 for (int i = 0; i < asset.clusterCount; ++i) cla[vgClusterCount_ + uint32_t(i)] = assetId;
684 vgGpu_.positions.unmap();
685 vgGpu_.triangles.unmap();
686 vgGpu_.clusters.unmap();
687 vgGpu_.clusterAssets.unmap();
688
689 vgClusterCount_ = needClusters;
690 vgVertexCount_ = vertBase + uint32_t(asset.vertexCount);
691 vgTriangleCount_ = needTris;
692 vgAssetCount_ = assetId + 1;
693 // Default identity model per slot; vgSetInstance overwrites the current slot.
694 {
695 void *m = vgGpu_.assetModels.map();
696 auto *models = static_cast<glm::mat4 *>(m);
697 for (size_t s = 0; s < kAsyncResourceCopies; ++s)
698 models[s * kMaxVgAssets + assetId] = glm::mat4(1.f);
699 vgGpu_.assetModels.unmap();
700 }
701 return assetId;
702}
703
705 if (!mesh || !mesh->gpuHandle) return kInvalidBindlessSlot;
706 const auto *gpu = static_cast<const GpuMesh *>(mesh->gpuHandle);
707 return gpu->record.vgAssetId;
708}
709
710bool Graphics::gpuDrivenVgAttachToMesh(Mesh *mesh, uint32_t vgAssetId) {
711 if (!mesh || !mesh->gpuHandle) return false;
712 if (vgAssetId >= vgAssetCount_ || vgAssetId >= kMaxVgAssets) return false;
713 auto *gpu = static_cast<GpuMesh *>(mesh->gpuHandle);
714 gpu->record.vgAssetId = vgAssetId;
715 if (gpu->gpuRecordIndex != kInvalidBindlessSlot &&
716 gpu->gpuRecordIndex < meshTableRecords_.size()) {
717 meshTableRecords_[gpu->gpuRecordIndex].vgAssetId = vgAssetId;
718 syncMeshTable();
719 }
720 return true;
721}
722
723bool Graphics::gpuDrivenVgSetInstance(uint32_t vgAssetId, const glm::mat4 &model,
724 uint32_t materialId) {
725 if (vgAssetId >= vgAssetCount_ || vgAssetId >= kMaxVgAssets) return false;
726 if (vgGpu_.assetModels.buffer) {
727 const size_t slot = currentFrameSlot() % kAsyncResourceCopies;
728 void *m = vgGpu_.assetModels.map();
729 static_cast<glm::mat4 *>(m)[slot * kMaxVgAssets + vgAssetId] = model;
730 vgGpu_.assetModels.unmap();
731 void *am = vgGpu_.assetMaterials.map();
732 static_cast<uint32_t *>(am)[slot * kMaxVgAssets + vgAssetId] = materialId;
733 vgGpu_.assetMaterials.unmap();
734 }
735 vgAnyThisFrame_ = true;
736 return true;
737}
738
740 if (!vgGpu_.visible.buffer || vgAssetCount_ == 0) return 0;
741 void *map = vgGpu_.visible.map();
742 if (!map) return 0;
743 const size_t slot = vgLastVisible_ % kAsyncResourceCopies;
744 const uint32_t count = static_cast<const uint32_t *>(map)[slot * (kMaxVgClusters + 1)];
745 vgGpu_.visible.unmap();
746 return count;
747}
748
749void Graphics::syncMeshTable() {
750 if (!meshTableBuffer_.buffer || meshTableRecords_.empty()) return;
751 meshTableBuffer_.updateLocal(vkb::FrameSlot::gpuIdle(), meshTableRecords_.data(),
752 meshTableRecords_.size() * sizeof(GpuMeshRecord));
753}
754
755GpuMaterialRecord Graphics::buildMaterialRecord(Material *material) {
756 GpuMaterialRecord rec{};
757 if (!material) return rec;
758 rec.tint = glm::vec4(material->getTintR(), material->getTintG(), material->getTintB(),
759 material->getTintA());
760 rec.pbr = glm::vec4(material->getMetallic(), material->getRoughness(),
761 material->getReceiveShadow() ? 1.f : 0.f,
762 material->getReceiveLight() ? 1.f : 0.f);
763 rec.texBomb = glm::vec4(material->getTexCellBombScale(), material->getTexCellBombStrength(),
764 material->getTexCellBombRotation(), 0.f);
765 rec.parallax = glm::vec4(material->getParallaxScale(), material->getParallaxMinLayers(),
766 material->getParallaxMaxLayers(), 0.f);
767 auto slotOf = [](Texture *t) {
768 if (!t || !t->gpuHandle) return kInvalidBindlessSlot;
769 return static_cast<GpuTexture *>(t->gpuHandle)->bindlessIndex2D;
770 };
771 rec.textureSlots[0] = slotOf(material->getAlbedoTexture());
772 rec.textureSlots[1] = slotOf(material->getNormalTexture());
773 rec.textureSlots[2] = slotOf(material->getHeightTexture());
774 rec.textureSlots[3] = kInvalidBindlessSlot; // env is camera state, not material state
775 const std::string model = material->getShadingModel();
776 rec.shadingModel = (model == "unlit") ? 1u : (model == "hair") ? 2u
777 : (model == "custom") ? 3u
778 : 0u;
779 if (material->getCastShadow()) rec.flags |= 1u;
780 if (material->getCastOcclusion()) rec.flags |= 2u;
781 return rec;
782}
783
785 if (!material) return kInvalidBindlessSlot;
786 auto it = materialTableIndex_.find(material);
787 if (it != materialTableIndex_.end()) {
788 materialTableRecords_[it->second] = buildMaterialRecord(material);
790 return it->second;
791 }
792 if (!materialTableFree_.empty()) {
793 const uint32_t idx = materialTableFree_.back();
794 materialTableFree_.pop_back();
795 materialTableIndex_.emplace(material, idx);
796 materialTableRecords_[idx] = buildMaterialRecord(material);
798 return idx;
799 }
800 if (materialTableRecords_.size() >= materialTableCapacity_) return kInvalidBindlessSlot;
801 const uint32_t idx = uint32_t(materialTableRecords_.size());
802 materialTableIndex_.emplace(material, idx);
803 materialTableRecords_.push_back(buildMaterialRecord(material));
805 return idx;
806}
807
809 if (!materialTableBuffer_.buffer || materialTableRecords_.empty()) return;
810 materialTableBuffer_.updateLocal(vkb::FrameSlot::gpuIdle(), materialTableRecords_.data(),
811 materialTableRecords_.size() *
812 sizeof(GpuMaterialRecord));
813}
814
816 if (!mesh || !mesh->gpuHandle) return kInvalidBindlessSlot;
817 auto *gpu = static_cast<GpuMesh *>(mesh->gpuHandle);
818 if (gpu->gpuRecordIndex == kInvalidBindlessSlot) registerMeshRecord(gpu);
819 return gpu->gpuRecordIndex;
820}
821
823 if (!material || material->virtualTextureMode() == MaterialVirtualTextureMode::AtlasPageTable ||
824 material->surfaceMode() == SurfaceMode::Transparent || material->hasPbrSurface())
825 return false;
826 return materialTableIndex_.contains(material) || !materialTableFree_.empty() ||
827 materialTableRecords_.size() < materialTableCapacity_;
828}
829
831 if (!material)
833 "cannot release a null GPU-driven material"));
834 const auto found = materialTableIndex_.find(material);
835 if (found == materialTableIndex_.end()) return Result<void>::success();
836 const uint32_t slot = found->second;
837 materialTableIndex_.erase(found);
838 materialTableRecords_[slot] = {};
839 materialTableFree_.push_back(slot);
841 return Result<void>::success();
842}
843
845 if (!gpuDrivenEnabled() || !gpuDrivenCaps_.gpuDrivenAvailable()) return false;
846 initGpuDrivenResources();
847 if (!mesh3dGpuDrivenPipeline || bindlessSets_.empty() || !meshTableBuffer_.buffer) return false;
848 if (!instances || instanceCount == 0) return false;
849
850 // Sort by (material, mesh) and merge buckets using the GPU mesh table.
852 for (uint32_t i = 0; i < instanceCount; ++i) {
853 builder.add(i, instances[i].meshId, instances[i].materialId, 0);
854 }
855 const uint32_t drawCount = builder.build(meshTableRecords_);
856 if (drawCount == 0) return false;
857 lastGpuDrivenDrawCount_ = drawCount;
858 const auto &cmds = builder.commands();
859 const auto &order = builder.sortedInstanceOrder();
860
861 // Upload instances in the sorted order so each bucket is a contiguous range.
862 std::vector<GpuInstance> sorted(instanceCount);
863 for (uint32_t i = 0; i < instanceCount; ++i) sorted[i] = instances[order[i]];
864
865 auto &arena = currentFrameArena();
866 FrameArena::Alloc instAlloc = arena.alloc(instanceCount * sizeof(GpuInstance), 16);
867 FrameArena::Alloc cmdAlloc = arena.alloc(drawCount * sizeof(GpuIndirectCommand), 16);
868 if (!instAlloc.mapped || !cmdAlloc.mapped) return false; // arena overflow: caller falls back
869 std::memcpy(instAlloc.mapped, sorted.data(), instanceCount * sizeof(GpuInstance));
870 std::memcpy(cmdAlloc.mapped, cmds.data(), drawCount * sizeof(GpuIndirectCommand));
871
872 // Bind the arena instance buffer as bindless binding 4 (update before bind,
873 // on the current frame slot's set only; the other slot's pending command
874 // buffers keep pointing at their own frame's arena).
875 const vk::DescriptorSet bindless = bindlessSetForFrame();
876 if (!bindless) return false;
877 vk::DescriptorBufferInfo instInfo{arena.buffer(), instAlloc.offset, instAlloc.size};
878 vk::WriteDescriptorSet instWrite{};
879 instWrite.dstSet = bindless;
880 instWrite.dstBinding = 4;
881 instWrite.descriptorCount = 1;
882 instWrite.descriptorType = vk::DescriptorType::eStorageBuffer;
883 instWrite.pBufferInfo = &instInfo;
884 device->updateDescriptorSets(1, &instWrite, 0, nullptr);
885
886 // Per-frame set0 through the shared ring (dynamic UBO offsets).
887 const Graphics::GpuDrivenFrameSet0 s0 = gpuDrivenFrameSet0();
888 if (!s0.set) return false;
889 const uint32_t dynOffsets[2] = {s0.uboOffset, s0.shadowOffset};
890
891 auto &cb = currentPresentCb();
892 cb.bindPipeline(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipeline);
893 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 0, 1,
894 &s0.set, 2, dynOffsets);
895 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 1, 1,
896 &bindless, 0, nullptr);
897 const vk::DeviceSize stride = sizeof(GpuIndirectCommand);
898 // Stage 1 keeps one host buffer pair per mesh (no pool yet), so each draw
899 // group must bind the owning mesh's vertex/index buffers. The builder sorts
900 // by (pipeline, material, mesh), so commands sharing a mesh are contiguous;
901 // bind only when the mesh changes.
902 GpuMesh *boundMesh = nullptr;
903 for (uint32_t i = 0; i < drawCount; ++i) {
904 const GpuInstance &first = sorted[cmds[i].firstInstance];
905 if (first.meshId >= meshRecordOwners_.size()) {
906 EV_ASSERT(false, "indirect draw references an unregistered mesh record");
907 continue;
908 }
909 GpuMesh *mesh = meshRecordOwners_[first.meshId];
910 if (mesh != boundMesh) {
911 const vk::DeviceSize vbOffset = 0;
912 cb.bindVertexBuffers(0, 1, mesh->vertices, &vbOffset);
913 cb.bindIndexBuffer(mesh->indices.buffer, 0, mesh->indexType);
914 boundMesh = mesh;
915 }
916 cb.drawIndexedIndirect(arena.buffer(), cmdAlloc.offset + vk::DeviceSize(i) * stride, 1,
917 stride);
918 }
919 return true;
920}
921
923 if (!gpuDrivenEnabled() || !gpuDrivenCaps_.gpuDrivenAvailable())
925 initGpuDrivenResources();
926 if (!mesh3dGpuDrivenPipeline || bindlessSets_.empty() || !meshTableBuffer_.buffer)
930 batch.buffer.strideBytes != sizeof(GpuInstance) || !batch.buckets || batch.bucketCount == 0 ||
931 batch.instanceCount == 0)
935 const uint64_t required = batch.buffer.offsetBytes + uint64_t(batch.instanceCount) * sizeof(GpuInstance);
936 if (required < batch.buffer.offsetBytes || required > batch.buffer.sizeBytes)
938
939 std::vector<GpuIndirectCommand> commands(batch.bucketCount);
940 uint64_t coveredInstances = 0;
941 for (uint32_t i = 0; i < batch.bucketCount; ++i) {
942 const GpuResidentInstanceBucket &bucket = batch.buckets[i];
943 const uint64_t end = uint64_t(bucket.firstInstance) + bucket.instanceCount;
944 if (bucket.instanceCount == 0 || bucket.firstInstance != coveredInstances || end > batch.instanceCount ||
945 bucket.meshId >= meshTableRecords_.size() || bucket.meshId >= meshRecordOwners_.size() ||
946 bucket.materialId >= materialTableRecords_.size() || !meshRecordOwners_[bucket.meshId])
948 const GpuMeshRecord &mesh = meshTableRecords_[bucket.meshId];
949 commands[i] = {mesh.indexCount, bucket.instanceCount, mesh.firstIndex, mesh.vertexBase, bucket.firstInstance};
950 coveredInstances = end;
951 }
952 if (coveredInstances != batch.instanceCount) return GpuResidentSubmitStatus::InvalidArgument;
953
954 auto &arena = currentFrameArena();
955 FrameArena::Alloc commandAlloc = arena.alloc(batch.bucketCount * sizeof(GpuIndirectCommand), 16);
956 if (!commandAlloc.mapped) return GpuResidentSubmitStatus::CapacityExceeded;
957 std::memcpy(commandAlloc.mapped, commands.data(), batch.bucketCount * sizeof(GpuIndirectCommand));
958
959 VkBuffer rawBuffer{};
960 static_assert(sizeof(rawBuffer) <= sizeof(batch.buffer.nativeHandle));
961 std::memcpy(&rawBuffer, &batch.buffer.nativeHandle, sizeof(rawBuffer));
962 vk::Buffer residentBuffer(rawBuffer);
963 const vk::DescriptorSet bindless = bindlessSetForFrame();
964 if (!bindless || !residentBuffer) return GpuResidentSubmitStatus::ResourceUnavailable;
965 vk::DescriptorBufferInfo instanceInfo{residentBuffer, batch.buffer.offsetBytes,
966 uint64_t(batch.instanceCount) * sizeof(GpuInstance)};
967 vk::WriteDescriptorSet instanceWrite{};
968 instanceWrite.dstSet = bindless;
969 instanceWrite.dstBinding = 4;
970 instanceWrite.descriptorCount = 1;
971 instanceWrite.descriptorType = vk::DescriptorType::eStorageBuffer;
972 instanceWrite.pBufferInfo = &instanceInfo;
973 device->updateDescriptorSets(1, &instanceWrite, 0, nullptr);
974
975 const Graphics::GpuDrivenFrameSet0 frame = gpuDrivenFrameSet0();
978 const uint32_t dynamicOffsets[2] = {frame.uboOffset, frame.shadowOffset};
979 auto &cb = currentPresentCb();
980 cb.bindPipeline(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipeline);
981 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 0, 1, &frame.set, 2,
982 dynamicOffsets);
983 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 1, 1, &bindless, 0, nullptr);
984
985 GpuMesh *boundMesh = nullptr;
986 for (uint32_t i = 0; i < batch.bucketCount; ++i) {
987 GpuMesh *mesh = meshRecordOwners_[batch.buckets[i].meshId];
988 if (mesh != boundMesh) {
989 const vk::DeviceSize vertexOffset = 0;
990 cb.bindVertexBuffers(0, 1, mesh->vertices, &vertexOffset);
991 cb.bindIndexBuffer(mesh->indices.buffer, 0, mesh->indexType);
992 boundMesh = mesh;
993 }
994 cb.drawIndexedIndirect(arena.buffer(), commandAlloc.offset + uint64_t(i) * sizeof(GpuIndirectCommand), 1,
995 sizeof(GpuIndirectCommand));
996 }
997 lastGpuDrivenDrawCount_ = batch.bucketCount;
999}
1000
1001// --- Stage 2: HZB + GPU cull ------------------------------------------------
1002
1003namespace {
1004
1006void gpuDrivenFrustumPlanes(const glm::mat4 &m, glm::vec4 planes[6]) {
1007 // glm is column-major: row i is (m[0][i], m[1][i], m[2][i], m[3][i]).
1008 const glm::vec4 row0(m[0][0], m[1][0], m[2][0], m[3][0]);
1009 const glm::vec4 row1(m[0][1], m[1][1], m[2][1], m[3][1]);
1010 const glm::vec4 row2(m[0][2], m[1][2], m[2][2], m[3][2]);
1011 const glm::vec4 row3(m[0][3], m[1][3], m[2][3], m[3][3]);
1012 planes[0] = row3 + row0; // left
1013 planes[1] = row3 - row0; // right
1014 planes[2] = row3 + row1; // bottom
1015 planes[3] = row3 - row1; // top
1016 planes[4] = row3 + row2; // near
1017 planes[5] = row3 - row2; // far
1018 for (int i = 0; i < 6; ++i) {
1019 const float len = glm::length(glm::vec3(planes[i]));
1020 if (len > 1e-8f) planes[i] /= len;
1021 }
1022}
1023
1024} // namespace
1025
1026Graphics::GpuDrivenCullSlot &Graphics::gpuDrivenCullSlot(uint32_t frameSlot) {
1027 // Callers guard with gpuDrivenCullReady_; this is a debug-only invariant.
1028 EV_ASSERT(!gpuDrivenCullSlots_.empty(), "gpuDriven cull slot accessed before resources");
1029 return gpuDrivenCullSlots_[frameSlot % gpuDrivenCullSlots_.size()];
1030}
1031
1032Graphics::GpuDrivenCullSlot &Graphics::currentGpuDrivenCullSlot() {
1033 return gpuDrivenCullSlot(static_cast<uint32_t>(currentFrameSlot()));
1034}
1035
1036void Graphics::ensureGpuDrivenCullResources(int width, int height) {
1037 if (!gpuDrivenCaps_.gpuDrivenCullAvailable() || width <= 0 || height <= 0) return;
1038 if (!bindlessSetLayout_) return;
1039 if (gpuDrivenCullReady_ && gpuDrivenCullWidth == width && gpuDrivenCullHeight == height)
1040 return;
1041 destroyGpuDrivenCullResources();
1042
1043 gpuDrivenCullWidth = width;
1044 gpuDrivenCullHeight = height;
1045
1046 // HZB word layout per slot: [16 header uints][mip0][mip1]...
1047 uint32_t mipOffsets[kMaxHzbMips] = {};
1048 uint32_t totalWords = kHZBHeaderWords;
1049 uint32_t maxMip = 0;
1050 {
1051 uint32_t w = uint32_t(width);
1052 uint32_t h = uint32_t(height);
1053 for (uint32_t m = 0; m < kMaxHzbMips; ++m) {
1054 mipOffsets[m] = totalWords;
1055 const uint32_t mw = std::max(w >> m, 1u);
1056 const uint32_t mh = std::max(h >> m, 1u);
1057 totalWords += mw * mh;
1058 maxMip = m;
1059 if (mw == 1 && mh == 1) break;
1060 }
1061 }
1062 gpuDrivenCullMaxMip = maxMip;
1063
1064 const vk::DeviceSize flagBytes = kMaxGpuDrivenInstances * sizeof(uint32_t);
1065 const vk::DeviceSize compactBytes = kMaxGpuDrivenInstances * sizeof(GpuInstance);
1066 const vk::DeviceSize indirectBytes = kMaxGpuDrivenBuckets * sizeof(GpuIndirectCommand);
1067 const vk::DeviceSize indirectNIBytes = kMaxGpuDrivenBuckets * sizeof(glm::uvec4);
1068 const vk::DeviceSize counterBytes = kMaxGpuDrivenBuckets * sizeof(uint32_t);
1069 const vk::DeviceSize hzbBytes = totalWords * sizeof(uint32_t);
1070
1071 gpuDrivenCullSlots_.resize(frameSlotCount());
1072 for (auto &slot : gpuDrivenCullSlots_) {
1073 slot.visibleFlags =
1074 vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer, flagBytes,
1075 kHostVisibleCoherent);
1076 slot.compacted = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer,
1077 compactBytes, kHostVisibleCoherent);
1078 slot.indirect = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer |
1079 vk::BufferUsageFlagBits::eIndirectBuffer,
1080 indirectBytes, kHostVisibleCoherent);
1081 slot.indirectNI = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer |
1082 vk::BufferUsageFlagBits::eIndirectBuffer,
1083 indirectNIBytes, kHostVisibleCoherent);
1084 slot.bucketCounters =
1085 vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer, counterBytes,
1086 kHostVisibleCoherent);
1087 slot.hzb = vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eStorageBuffer, hzbBytes,
1088 kHostVisibleCoherent);
1089 slot.cullParams =
1090 vkb::GenericBuffer(device, vk::BufferUsageFlagBits::eUniformBuffer,
1091 sizeof(GpuCullParams), kHostVisibleCoherent);
1092 // Header offsets + zero depth (0 = near plane -> everything visible).
1093 void *hzbMap = slot.hzb.map();
1094 auto *u32 = static_cast<uint32_t *>(hzbMap);
1095 for (uint32_t m = 0; m <= maxMip; ++m) u32[m] = mipOffsets[m];
1096 for (uint32_t m = maxMip + 1; m < kHZBHeaderWords; ++m) u32[m] = 0;
1097 std::memset(static_cast<char *>(hzbMap) + kHZBHeaderWords * sizeof(uint32_t), 0,
1098 (totalWords - kHZBHeaderWords) * sizeof(uint32_t));
1099 slot.hzb.unmap();
1100 }
1101
1102 // Compute pipeline layout: set 1 = bindless (set 0 = empty placeholder so
1103 // the shaders' `layout(set = 1, ...)` bindings resolve to index 1).
1104 vk::DescriptorSetLayoutCreateInfo emptyInfo{};
1105 gpuDrivenComputeEmptyLayout_ = device->createDescriptorSetLayout(emptyInfo);
1106 vk::PushConstantRange pcr{vk::ShaderStageFlagBits::eCompute, 0, 20};
1107 vk::PipelineLayoutCreateInfo pli{};
1108 std::array<vk::DescriptorSetLayout, 2> computeLayouts{gpuDrivenComputeEmptyLayout_,
1109 bindlessSetLayout_};
1110 pli.setLayoutCount = uint32_t(computeLayouts.size());
1111 pli.pSetLayouts = computeLayouts.data();
1112 pli.pushConstantRangeCount = 1;
1113 pli.pPushConstantRanges = &pcr;
1114 gpuDrivenComputeLayout = device->createPipelineLayout(pli);
1115
1116 const auto spvOf = [](const uint32_t *p, size_t n) {
1117 return std::vector<uint32_t>(p, p + n);
1118 };
1119 hzbBuildPass_.create(device, gpuDrivenComputeLayout,
1120 spvOf(hzb_build_comp_spv, hzb_build_comp_spv_count));
1121 cullPass_.create(device, gpuDrivenComputeLayout,
1122 spvOf(gpu_cull_comp_spv, gpu_cull_comp_spv_count));
1123 emitPass_.create(device, gpuDrivenComputeLayout,
1124 spvOf(gpu_emit_comp_spv, gpu_emit_comp_spv_count));
1125 vgCullPass_.create(device, gpuDrivenComputeLayout,
1126 spvOf(vg_main_cull_comp_spv, vg_main_cull_comp_spv_count));
1127 gpuDrivenCullReady_ =
1128 hzbBuildPass_.pipeline() && cullPass_.pipeline() && emitPass_.pipeline();
1129}
1130
1131void Graphics::destroyGpuDrivenCullResources() {
1132 hzbBuildPass_ = ComputePass{};
1133 cullPass_ = ComputePass{};
1134 emitPass_ = ComputePass{};
1135 vgCullPass_ = ComputePass{};
1136 if (gpuDrivenComputeLayout) {
1137 device->destroyPipelineLayout(gpuDrivenComputeLayout);
1138 gpuDrivenComputeLayout = nullptr;
1139 }
1140 if (gpuDrivenComputeEmptyLayout_) {
1141 device->destroyDescriptorSetLayout(gpuDrivenComputeEmptyLayout_);
1142 gpuDrivenComputeEmptyLayout_ = nullptr;
1143 }
1144 for (auto &slot : gpuDrivenCullSlots_) {
1145 slot.visibleFlags.release();
1146 slot.compacted.release();
1147 slot.indirect.release();
1148 slot.indirectNI.release();
1149 slot.bucketCounters.release();
1150 slot.hzb.release();
1151 slot.cullParams.release();
1152 }
1153 gpuDrivenCullSlots_.clear();
1154 gpuDrivenCullReady_ = false;
1155 gpuDrivenCullWidth = 0;
1156 gpuDrivenCullHeight = 0;
1157}
1158
1159void Graphics::recordGpuDrivenHzbBuild() {
1160 if (!gpuDrivenCullReady_) return;
1161 auto *slot = currentGBufferSlot();
1162 if (!slot || !slot->depthGpu.sampler || !slot->depthGpu.imageView()) return;
1163 auto &cull = currentGpuDrivenCullSlot();
1164 auto &cb = currentPresentCb();
1165
1166 // Depth attachment write -> compute shader read.
1167 vk::ImageMemoryBarrier imb{};
1168 imb.image = slot->depth.image();
1169 imb.oldLayout = vk::ImageLayout::eShaderReadOnlyOptimal;
1170 imb.newLayout = vk::ImageLayout::eShaderReadOnlyOptimal;
1171 imb.srcAccessMask = vk::AccessFlagBits::eDepthStencilAttachmentWrite;
1172 imb.dstAccessMask = vk::AccessFlagBits::eShaderRead;
1173 imb.subresourceRange = {vk::ImageAspectFlagBits::eDepth, 0, 1, 0, 1};
1174 cb.pipelineBarrier(vk::PipelineStageFlagBits::eEarlyFragmentTests |
1175 vk::PipelineStageFlagBits::eLateFragmentTests,
1176 vk::PipelineStageFlagBits::eComputeShader, {}, 0, nullptr, 0, nullptr, 1,
1177 &imb);
1178
1179 uint32_t mipOffsets[kMaxHzbMips] = {};
1180 uint32_t totalWords = kHZBHeaderWords;
1181 {
1182 uint32_t w = uint32_t(gpuDrivenCullWidth);
1183 uint32_t h = uint32_t(gpuDrivenCullHeight);
1184 for (uint32_t m = 0; m <= gpuDrivenCullMaxMip; ++m) {
1185 mipOffsets[m] = totalWords;
1186 const uint32_t mw = std::max(w >> m, 1u);
1187 const uint32_t mh = std::max(h >> m, 1u);
1188 totalWords += mw * mh;
1189 }
1190 }
1191 for (uint32_t m = 0; m <= gpuDrivenCullMaxMip; ++m) {
1192 if (m > 0) {
1193 // Previous mip write -> this mip read.
1194 vk::BufferMemoryBarrier bmb{};
1195 bmb.buffer = cull.hzb.buffer;
1196 bmb.size = VK_WHOLE_SIZE;
1197 bmb.srcAccessMask = vk::AccessFlagBits::eShaderWrite;
1198 bmb.dstAccessMask = vk::AccessFlagBits::eShaderRead;
1199 cb.pipelineBarrier(vk::PipelineStageFlagBits::eComputeShader,
1200 vk::PipelineStageFlagBits::eComputeShader, {}, 0, nullptr, 1, &bmb,
1201 0, nullptr);
1202 }
1203 const uint32_t mw = std::max(uint32_t(gpuDrivenCullWidth) >> m, 1u);
1204 const uint32_t mh = std::max(uint32_t(gpuDrivenCullHeight) >> m, 1u);
1205 struct HzbPush {
1206 uint32_t mip;
1207 uint32_t width;
1208 uint32_t height;
1209 uint32_t slotBase;
1210 uint32_t prevOffset;
1211 } push{m, mw, mh, 0u, mipOffsets[m > 0 ? m - 1 : 0]};
1212 cb.pushConstants(gpuDrivenComputeLayout, vk::ShaderStageFlagBits::eCompute, 0,
1213 sizeof(push), &push);
1214 hzbBuildPass_.record(cb, (mw + 7) / 8, (mh + 7) / 8, 1);
1215 }
1216}
1217
1219 if (!gpuDrivenCullEnabled() || !gpuDrivenCullReady_) return false;
1220 if (!instances || instanceCount == 0) return false;
1222
1224 for (uint32_t i = 0; i < instanceCount; ++i) {
1225 builder.add(i, instances[i].meshId, instances[i].materialId, 0);
1226 }
1227 const uint32_t bucketCount = builder.build(meshTableRecords_);
1228 if (bucketCount == 0 || bucketCount > kMaxGpuDrivenBuckets) return false;
1229 const auto &cmds = builder.commands();
1230 const auto &order = builder.sortedInstanceOrder();
1231
1232 std::vector<GpuInstance> sorted(instanceCount);
1233 for (uint32_t i = 0; i < instanceCount; ++i) sorted[i] = instances[order[i]];
1234
1235 gpuDrivenBucketIds_.assign(instanceCount, 0);
1236 gpuDrivenBucketOffsets_.resize(bucketCount);
1237 gpuDrivenBucketMeshIds_.resize(bucketCount);
1238 for (uint32_t b = 0; b < bucketCount; ++b) {
1239 gpuDrivenBucketOffsets_[b] = cmds[b].firstInstance;
1240 gpuDrivenBucketMeshIds_[b] = sorted[cmds[b].firstInstance].meshId;
1241 }
1242 {
1243 uint32_t b = 0;
1244 for (uint32_t j = 0; j < instanceCount; ++j) {
1245 while (b + 1 < bucketCount && cmds[b + 1].firstInstance <= j) ++b;
1246 gpuDrivenBucketIds_[j] = b;
1247 }
1248 }
1249
1250 auto &arena = currentFrameArena();
1251 // Storage-buffer descriptor offsets must honor minStorageBufferOffsetAlignment
1252 // (64 on this Intel driver); align conservatively to 64.
1253 gpuDrivenInstAlloc_ = arena.alloc(instanceCount * sizeof(GpuInstance), 64);
1254 gpuDrivenBucketIdAlloc_ = arena.alloc(instanceCount * sizeof(uint32_t), 64);
1255 gpuDrivenBucketOffAlloc_ = arena.alloc(bucketCount * sizeof(uint32_t), 64);
1256 if (!gpuDrivenInstAlloc_.mapped || !gpuDrivenBucketIdAlloc_.mapped ||
1257 !gpuDrivenBucketOffAlloc_.mapped) {
1258 gpuDrivenCullInstanceCount_ = 0;
1259 return false;
1260 }
1261 std::memcpy(gpuDrivenInstAlloc_.mapped, sorted.data(),
1262 instanceCount * sizeof(GpuInstance));
1263 std::memcpy(gpuDrivenBucketIdAlloc_.mapped, gpuDrivenBucketIds_.data(),
1264 instanceCount * sizeof(uint32_t));
1265 std::memcpy(gpuDrivenBucketOffAlloc_.mapped, gpuDrivenBucketOffsets_.data(),
1266 bucketCount * sizeof(uint32_t));
1267
1268 gpuDrivenBucketCount_ = bucketCount;
1269 gpuDrivenCullInstanceCount_ = instanceCount;
1270 lastGpuDrivenDrawCount_ = bucketCount; // debug counter: bucket draws the cull path emits
1271
1272 // Cull source buffers: sorted instances (17), bucket ids (12), offsets (13).
1273 const vk::DescriptorSet bindless = bindlessSetForFrame();
1274 if (!bindless) {
1275 gpuDrivenCullInstanceCount_ = 0;
1276 return false;
1277 }
1278 auto bufWrite = [&](uint32_t binding, vk::Buffer buffer, vk::DeviceSize offset,
1279 vk::DeviceSize size) {
1280 vk::DescriptorBufferInfo info{buffer, offset, size};
1281 vk::WriteDescriptorSet w{};
1282 w.dstSet = bindless;
1283 w.dstBinding = binding;
1284 w.descriptorCount = 1;
1285 w.descriptorType = vk::DescriptorType::eStorageBuffer;
1286 w.pBufferInfo = &info;
1287 device->updateDescriptorSets(1, &w, 0, nullptr);
1288 };
1289 bufWrite(17, arena.buffer(), gpuDrivenInstAlloc_.offset, gpuDrivenInstAlloc_.size);
1290 bufWrite(12, arena.buffer(), gpuDrivenBucketIdAlloc_.offset, gpuDrivenBucketIdAlloc_.size);
1291 bufWrite(13, arena.buffer(), gpuDrivenBucketOffAlloc_.offset, gpuDrivenBucketOffAlloc_.size);
1292 return true;
1293}
1294
1295void Graphics::gpuDrivenRecordComputeSection(const glm::mat4 &viewProj, const glm::vec3 &eye,
1296 float fovYDeg, float nearZ, float farZ) {
1297 if (!gpuDrivenCullReady_) return;
1298 auto &slot = currentGpuDrivenCullSlot();
1299 auto &cb = currentPresentCb();
1300 const vk::DescriptorSet bindless = bindlessSetForFrame();
1301 if (!bindless) return;
1302 // Lazily-created defaults register into the bindless set (bindings 0/1);
1303 // do it BEFORE the set is bound to the recording command buffer.
1304 ensureFlatNormalTexture3D();
1305 ensureFlatHeightTexture3D();
1306 ensureDefaultEnvCubemap();
1307 // Per-frame bindless updates must happen BEFORE the set is bound to the
1308 // recording command buffer (no UPDATE_AFTER_BIND): updating it afterwards
1309 // invalidates the command buffer (VUID-vkCmdBindPipeline-commandBuffer-recording).
1310 createGBufferResources(gbufferWidth > 0 ? gbufferWidth : int(swapchain.extent.width),
1311 gbufferHeight > 0 ? gbufferHeight : int(swapchain.extent.height));
1312 {
1313 auto *gbSlot = currentGBufferSlot();
1314 if (gbSlot && gbSlot->visIDGpu.sampler && gbSlot->visBaryGpu.sampler) {
1315 vk::DescriptorImageInfo visIDInfo{gbSlot->visIDGpu.sampler, gbSlot->visIDGpu.imageView(),
1316 vk::ImageLayout::eShaderReadOnlyOptimal};
1317 vk::DescriptorImageInfo visBaryInfo{gbSlot->visBaryGpu.sampler,
1318 gbSlot->visBaryGpu.imageView(),
1319 vk::ImageLayout::eShaderReadOnlyOptimal};
1320 vk::WriteDescriptorSet w[2]{};
1321 w[0].dstSet = bindless;
1322 w[0].dstBinding = 15;
1323 w[0].descriptorCount = 1;
1324 w[0].descriptorType = vk::DescriptorType::eCombinedImageSampler;
1325 w[0].pImageInfo = &visIDInfo;
1326 w[1].dstSet = bindless;
1327 w[1].dstBinding = 16;
1328 w[1].descriptorCount = 1;
1329 w[1].descriptorType = vk::DescriptorType::eCombinedImageSampler;
1330 w[1].pImageInfo = &visBaryInfo;
1331 device->updateDescriptorSets(2, w, 0, nullptr);
1332 }
1333 }
1334 bindVgFrameBindless(bindless, currentFrameSlot() % kAsyncResourceCopies);
1335
1336 GpuCullParams params{};
1337 params.viewProj = viewProj;
1338 gpuDrivenFrustumPlanes(viewProj, params.frustumPlanes);
1339 params.cameraPos = glm::vec4(eye, 0.f);
1340 const int w = gpuDrivenCullWidth;
1341 const int h = gpuDrivenCullHeight;
1342 params.screen = glm::vec4(float(w), float(h), 1.f / float(w), 1.f / float(h));
1343 const float fovRad = fovYDeg * 0.017453292519943295f;
1344 params.clipNearFar =
1345 glm::vec4(nearZ, farZ, float(h) * 0.5f / std::tan(fovRad * 0.5f), 1.f);
1346 params.hzbInfo =
1347 glm::vec4(float(gpuDrivenCullMaxMip), 0.f, float(w), float(h));
1348 params.counts = glm::uvec4(gpuDrivenCullInstanceCount_, gpuDrivenBucketCount_,
1350 {
1351 void *pMap = slot.cullParams.map();
1352 std::memcpy(pMap, &params, sizeof(params));
1353 slot.cullParams.unmap();
1354 }
1355
1356 auto bufWrite = [&](uint32_t binding, vk::Buffer buffer, vk::DeviceSize offset,
1357 vk::DeviceSize size, vk::DescriptorType type) {
1358 vk::DescriptorBufferInfo info{buffer, offset, size};
1359 vk::WriteDescriptorSet w{};
1360 w.dstSet = bindless;
1361 w.dstBinding = binding;
1362 w.descriptorCount = 1;
1363 w.descriptorType = type;
1364 w.pBufferInfo = &info;
1365 device->updateDescriptorSets(1, &w, 0, nullptr);
1366 };
1367 bufWrite(6, slot.cullParams.buffer, 0, sizeof(GpuCullParams),
1368 vk::DescriptorType::eUniformBuffer);
1369 bufWrite(7, slot.visibleFlags.buffer, 0, VK_WHOLE_SIZE,
1370 vk::DescriptorType::eStorageBuffer);
1371 bufWrite(8, slot.compacted.buffer, 0, VK_WHOLE_SIZE, vk::DescriptorType::eStorageBuffer);
1372 bufWrite(9, slot.indirect.buffer, 0, VK_WHOLE_SIZE, vk::DescriptorType::eStorageBuffer);
1373 bufWrite(30, slot.indirectNI.buffer, 0, VK_WHOLE_SIZE, vk::DescriptorType::eStorageBuffer);
1374 bufWrite(14, slot.bucketCounters.buffer, 0, VK_WHOLE_SIZE,
1375 vk::DescriptorType::eStorageBuffer);
1376 // The draw consumes the compacted buffer as its instance source (binding 4).
1377 bufWrite(4, slot.compacted.buffer, 0, VK_WHOLE_SIZE, vk::DescriptorType::eStorageBuffer);
1378 // HZB buffer + GBuffer depth sampler for the build pass.
1379 bufWrite(10, slot.hzb.buffer, 0, VK_WHOLE_SIZE, vk::DescriptorType::eStorageBuffer);
1380 {
1381 auto *gbSlot = currentGBufferSlot();
1382 if (gbSlot && gbSlot->depthGpu.sampler && gbSlot->depthGpu.imageView()) {
1383 vk::DescriptorImageInfo depthInfo{gbSlot->depthGpu.sampler, gbSlot->depthGpu.imageView(),
1384 vk::ImageLayout::eShaderReadOnlyOptimal};
1385 vk::WriteDescriptorSet wd{};
1386 wd.dstSet = bindless;
1387 wd.dstBinding = 11;
1388 wd.descriptorCount = 1;
1389 wd.descriptorType = vk::DescriptorType::eCombinedImageSampler;
1390 wd.pImageInfo = &depthInfo;
1391 device->updateDescriptorSets(1, &wd, 0, nullptr);
1392 }
1393 }
1394
1395 // All descriptor updates are done; bind the bindless set ONCE for the
1396 // whole compute section (the set is not UPDATE_AFTER_BIND, so updating it
1397 // while bound to a recording command buffer would invalidate the buffer).
1398 cb.bindDescriptorSets(vk::PipelineBindPoint::eCompute, gpuDrivenComputeLayout, 1, 1,
1399 &bindless, 0, nullptr);
1400
1401 recordGpuDrivenHzbBuild();
1402
1403 // HZB build writes (storage buffer) -> cull reads.
1404 {
1405 vk::BufferMemoryBarrier bmb{};
1406 bmb.buffer = slot.hzb.buffer;
1407 bmb.size = VK_WHOLE_SIZE;
1408 bmb.srcAccessMask = vk::AccessFlagBits::eShaderWrite;
1409 bmb.dstAccessMask = vk::AccessFlagBits::eShaderRead;
1410 cb.pipelineBarrier(vk::PipelineStageFlagBits::eComputeShader,
1411 vk::PipelineStageFlagBits::eComputeShader, {}, 0, nullptr, 1, &bmb, 0,
1412 nullptr);
1413 }
1414}
1415
1416void Graphics::gpuDrivenVgComputeSection(const glm::mat4 &viewProj, const glm::vec3 &eye,
1417 float fovYDeg, float nearZ, float farZ) {
1418 gpuDrivenRecordComputeSection(viewProj, eye, fovYDeg, nearZ, farZ);
1419}
1420
1421void Graphics::gpuDrivenCullEmit(const glm::mat4 &viewProj, const glm::vec3 &eye, float fovYDeg,
1422 float nearZ, float farZ) {
1423 if (!gpuDrivenCullReady_ || gpuDrivenCullInstanceCount_ == 0) return;
1424 gpuDrivenLastCullSlot_ = uint32_t(currentFrameSlot());
1425 auto &slot = currentGpuDrivenCullSlot();
1426 auto &cb = currentPresentCb();
1427 // This frame's own bindless set: the previous frame that owned this slot
1428 // completed at acquireForFrame()'s fence wait, so rewriting it is safe.
1429 const vk::DescriptorSet bindless = bindlessSetForFrame();
1430 if (!bindless) return;
1431
1432 // CPU reset of GPU-owned per-slot state (slot's previous frame is complete).
1433 {
1434 void *flagsMap = slot.visibleFlags.map();
1435 std::memset(flagsMap, 0, kMaxGpuDrivenInstances * sizeof(uint32_t));
1436 slot.visibleFlags.unmap();
1437 void *counterMap = slot.bucketCounters.map();
1438 std::memset(counterMap, 0, kMaxGpuDrivenBuckets * sizeof(uint32_t));
1439 slot.bucketCounters.unmap();
1440 // The emit pass only writes buckets that survived culling; clear the
1441 // commands so a stale instanceCount from an earlier frame cannot make
1442 // a culled bucket draw again (readbacks / validation read this too).
1443 void *cmdMap = slot.indirect.map();
1444 std::memset(cmdMap, 0, kMaxGpuDrivenBuckets * sizeof(GpuIndirectCommand));
1445 slot.indirect.unmap();
1446 }
1447
1448 gpuDrivenRecordComputeSection(viewProj, eye, fovYDeg, nearZ, farZ);
1449
1450 const uint32_t groups = (gpuDrivenCullInstanceCount_ + 63u) / 64u;
1451 // Arena host writes + previous-frame HZB shader writes visible to cull.
1452 {
1453 vk::BufferMemoryBarrier bmb{};
1454 bmb.buffer = currentFrameArena().buffer();
1455 bmb.size = VK_WHOLE_SIZE;
1456 bmb.srcAccessMask = vk::AccessFlagBits::eHostWrite | vk::AccessFlagBits::eShaderWrite;
1457 bmb.dstAccessMask = vk::AccessFlagBits::eShaderRead;
1458 cb.pipelineBarrier(vk::PipelineStageFlagBits::eHost |
1459 vk::PipelineStageFlagBits::eComputeShader,
1460 vk::PipelineStageFlagBits::eComputeShader, {}, 0, nullptr, 1, &bmb, 0,
1461 nullptr);
1462 }
1463 cullPass_.record(cb, groups, 1, 1);
1464
1465 // Cull flags write -> emit read.
1466 {
1467 vk::BufferMemoryBarrier bmb{};
1468 bmb.buffer = slot.visibleFlags.buffer;
1469 bmb.size = VK_WHOLE_SIZE;
1470 bmb.srcAccessMask = vk::AccessFlagBits::eShaderWrite;
1471 bmb.dstAccessMask = vk::AccessFlagBits::eShaderRead;
1472 cb.pipelineBarrier(vk::PipelineStageFlagBits::eComputeShader,
1473 vk::PipelineStageFlagBits::eComputeShader, {}, 0, nullptr, 1, &bmb, 0,
1474 nullptr);
1475 }
1476 emitPass_.record(cb, groups, 1, 1);
1477
1478 // Emit writes (compacted + commands) -> vertex read + indirect read.
1479 {
1480 vk::BufferMemoryBarrier bmb[2]{};
1481 bmb[0].buffer = slot.compacted.buffer;
1482 bmb[0].size = VK_WHOLE_SIZE;
1483 bmb[0].srcAccessMask = vk::AccessFlagBits::eShaderWrite;
1484 bmb[0].dstAccessMask = vk::AccessFlagBits::eShaderRead;
1485 bmb[1].buffer = slot.indirect.buffer;
1486 bmb[1].size = VK_WHOLE_SIZE;
1487 bmb[1].srcAccessMask = vk::AccessFlagBits::eShaderWrite;
1488 bmb[1].dstAccessMask = vk::AccessFlagBits::eIndirectCommandRead;
1489 cb.pipelineBarrier(vk::PipelineStageFlagBits::eComputeShader,
1490 vk::PipelineStageFlagBits::eVertexShader |
1491 vk::PipelineStageFlagBits::eDrawIndirect,
1492 {}, 0, nullptr, 2, bmb, 0, nullptr);
1493 }
1494}
1495
1497 if (!gpuDrivenScenePassPending_) return;
1498 gpuDrivenScenePassPending_ = false;
1499 if (beginSceneColorRenderPass()) {
1500 ensureScenePassPipelines(activeScenePass(), activeSceneSamples());
1501 } else {
1502 ensureScenePassPipelines(renderpass, vk::SampleCountFlagBits::e1);
1503 beginSwapchainColorPass();
1504 }
1505 swapchainPassOpen = true;
1506}
1507
1508Graphics::GpuDrivenFrameSet0 Graphics::gpuDrivenFrameSet0() {
1509 // Per-frame set0 + UBO (same shading state as the stage-1 path).
1510 ensureFlatNormalTexture3D();
1511 ensureFlatHeightTexture3D();
1512 ensureDefaultEnvCubemap();
1513 Texture *tex = whiteTexture;
1514 auto *gpuTex = static_cast<GpuTexture *>(tex->gpuHandle);
1515 auto *gpuNormal = static_cast<GpuTexture *>(mesh3dNormalTexture ? mesh3dNormalTexture->gpuHandle
1516 : flatNormalTexture3D->gpuHandle);
1517 auto *gpuHeight = static_cast<GpuTexture *>(mesh3dHeightTexture ? mesh3dHeightTexture->gpuHandle
1518 : flatHeightTexture3D->gpuHandle);
1519 Texture *envTex = mesh3dEnvTexture ? mesh3dEnvTexture : defaultEnvCubemap;
1520 auto *gpuEnv = static_cast<GpuTexture *>(envTex->gpuHandle);
1521 auto *gpuDepth = static_cast<GpuTexture *>(whiteTexture->gpuHandle);
1522 ensureDecalPlaceholders();
1523 auto *gpuDecalAlb = static_cast<GpuTexture *>(decalFlatAlbedo->gpuHandle);
1524 auto *gpuDecalNrm = static_cast<GpuTexture *>(decalFlatNormal->gpuHandle);
1525 auto *gpuDecalPrm = static_cast<GpuTexture *>(decalFlatParams->gpuHandle);
1526 if (decalLayerFresh) {
1527 if (auto *dslot = currentDecalSlot()) {
1528 gpuDecalAlb = &dslot->albedoGpu;
1529 gpuDecalNrm = &dslot->normalGpu;
1530 gpuDecalPrm = &dslot->paramsGpu;
1531 }
1532 }
1533
1534 Mesh3DUBO ubo = mesh3dFrameUbo;
1535 ubo.model = glm::mat4(1.f);
1536 ubo.tint = glm::vec4(1.f);
1537 const int lightCount = std::max(0, std::min(mesh3dLighting.count, Lighting3DPack::kMaxLights));
1538 ubo.lightDir.w = float(lightCount);
1539 ubo.cameraPos.w = mesh3dRoughness;
1540 ubo.lightColor.w = mesh3dEnvIntensity;
1541 const uint32_t envSlot = gpuEnv->bindlessIndexCube;
1544 for (int i = 0; i < ReflectionProbeUpload::kMaxProbes; ++i) {
1545 if (i >= mesh3dReflectionProbes.count) continue;
1546 const auto &probe = mesh3dReflectionProbes.probes[i];
1547 if (!probe.cubemap || !probe.cubemap->gpuHandle) continue;
1548 auto *gpuProbe = static_cast<GpuTexture *>(probe.cubemap->gpuHandle);
1549 if (!gpuProbe->isCube || gpuProbe->bindlessIndexCube == kInvalidBindlessSlot) continue;
1550 probeSlots[i] = gpuProbe->bindlessIndexCube;
1551 ubo.reflectionProbeCenter[i] = glm::vec4(probe.center, probe.intensity);
1552 ubo.reflectionProbeExtent[i] = glm::vec4(probe.extent, probe.blendDistance);
1553 }
1554 ubo.bindlessEnv = glm::vec4(float(envSlot), mesh3dEnvIntensity,
1555 float(probeSlots[0]), float(probeSlots[1]));
1556 ubo.ambient = glm::vec4(glm::vec3(mesh3dLighting.ambient), mesh3dMetallic);
1557 for (int i = 0; i < lightCount; ++i) ubo.lights[i] = mesh3dLighting.lights[i];
1558 int dirI = -1;
1559 for (int i = 0; i < lightCount; ++i) {
1560 if (mesh3dLighting.lights[i].posRadius.w <= 0.f) {
1561 dirI = i;
1562 break;
1563 }
1564 }
1565 if (dirI >= 0) {
1566 glm::vec3 d(mesh3dLighting.lights[dirI].posRadius);
1567 if (glm::length(d) < 1e-6f) d = glm::vec3(0.f, 1.f, 0.f);
1568 else d = glm::normalize(d);
1569 ubo.lightDir = glm::vec4(d, float(lightCount));
1570 ubo.lightColor =
1571 glm::vec4(glm::vec3(mesh3dLighting.lights[dirI].color), mesh3dEnvIntensity);
1572 } else {
1573 ubo.lightDir = glm::vec4(0.f, 1.f, 0.f, float(lightCount));
1574 ubo.lightColor = glm::vec4(0.f, 0.f, 0.f, mesh3dEnvIntensity);
1575 }
1576 auto &fslots = currentMesh3dFrameSlots();
1577 if (fslots.drawIndex >= fslots.capacity) {
1578 std::fprintf(stderr, "[vulkan] mesh3d UBO ring exhausted; gpu-driven draw skipped\n");
1579 return {};
1580 }
1581 const size_t slot = fslots.drawIndex++;
1582 ensureMesh3dStrides();
1583 const uint32_t uboOffset = uint32_t(slot) * mesh3dUboStride;
1584 const uint32_t shadowOffset = uint32_t(slot) * shadowUboStride;
1585 updateRingLocal(fslots.uboRing, uboOffset, &ubo, sizeof(ubo));
1586 ShadowUBO shadowUbo = mesh3dShadows.ubo;
1587 if (!mesh3dShadows.active) {
1588 shadowUbo.bias.y = 0.f;
1589 shadowUbo.splits.w = 0.f;
1590 }
1591 shadowUbo.bias.z = mesh3dShadowReceive ? 1.f : 0.f;
1592 updateRingLocal(fslots.shadowRing, shadowOffset, &shadowUbo, sizeof(shadowUbo));
1593 auto *gpuSceneColor = mesh3dSceneColorTexture && mesh3dSceneColorTexture->gpuHandle
1594 ? static_cast<GpuTexture *>(mesh3dSceneColorTexture->gpuHandle)
1595 : sceneColorHistoryValid && completedSceneColorSlot < sceneColorSlots.size()
1596 ? &sceneColorSlots[completedSceneColorSlot].colorGpu
1597 : static_cast<GpuTexture *>(whiteTexture->gpuHandle);
1598 uploadSkinPalette(nullptr, fslots);
1599 vk::DescriptorSet set = mesh3dSetFor(gpuTex, gpuNormal, gpuEnv, gpuHeight, gpuDepth, gpuSceneColor,
1600 gpuDecalAlb, gpuDecalNrm, gpuDecalPrm, fslots);
1601 return {set, uboOffset, shadowOffset};
1602}
1603
1605 if (!gpuDrivenCullReady_ || gpuDrivenBucketCount_ == 0) return;
1606 if (!swapchainPassOpen && !sceneColorPassOpen) return;
1607 auto &slot = currentGpuDrivenCullSlot();
1608
1609 const Graphics::GpuDrivenFrameSet0 s0 = gpuDrivenFrameSet0();
1610 if (!s0.set) return;
1611 const uint32_t dynOffsets[2] = {s0.uboOffset, s0.shadowOffset};
1612
1613 auto &cb = currentPresentCb();
1614 cb.bindPipeline(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipeline);
1615 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 0, 1,
1616 &s0.set, 2, dynOffsets);
1617 const vk::DescriptorSet bindless = bindlessSetForFrame();
1618 if (!bindless) return;
1619 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 1, 1,
1620 &bindless, 0, nullptr);
1621
1622 const vk::DeviceSize stride = sizeof(GpuIndirectCommand);
1623 GpuMesh *boundMesh = nullptr;
1624 for (uint32_t b = 0; b < gpuDrivenBucketCount_; ++b) {
1625 const uint32_t meshId = gpuDrivenBucketMeshIds_[b];
1626 if (meshId >= meshRecordOwners_.size()) continue;
1627 GpuMesh *mesh = meshRecordOwners_[meshId];
1628 if (mesh != boundMesh) {
1629 const vk::DeviceSize vbOffset = 0;
1630 cb.bindVertexBuffers(0, 1, mesh->vertices, &vbOffset);
1631 cb.bindIndexBuffer(mesh->indices.buffer, 0, mesh->indexType);
1632 boundMesh = mesh;
1633 }
1634 cb.drawIndexedIndirect(slot.indirect.buffer, vk::DeviceSize(b) * stride, 1, stride);
1635 }
1636}
1637
1639 if (!gpuDrivenCullReady_) return;
1640 if (gpuDrivenBucketCount_ == 0 && !vgAnyThisFrame_) return;
1641 // The vis attachments live in the GBuffer slot set; make sure it exists
1642 // even when the AO/gbuffer feature is off (resolve needs it anyway).
1643 createGBufferResources(gbufferWidth > 0 ? gbufferWidth : int(swapchain.extent.width),
1644 gbufferHeight > 0 ? gbufferHeight : int(swapchain.extent.height));
1645 auto *slot = currentGBufferSlot();
1646 if (!slot || !gbufferVisPipeline || !gbufferVisRenderPass || !slot->visFramebuffer) return;
1647 auto &cull = currentGpuDrivenCullSlot();
1648 auto &cb = currentPresentCb();
1649 // Stage 3 VG: cluster cull (frustum + HZB) runs right before the vis pass.
1650 recordVgCull();
1651
1652 const uint32_t w = uint32_t(gbufferWidth);
1653 const uint32_t h = uint32_t(gbufferHeight);
1654 std::array<vk::ClearValue, 6> clears{};
1655 clears[0].color = vk::ClearColorValue(std::array<float, 4>{0, 0, 0, 0});
1656 clears[1].color = vk::ClearColorValue(std::array<float, 4>{1, 1, 1, 1});
1657 clears[2].color = vk::ClearColorValue(std::array<float, 4>{0, 0, 0, 0});
1658 clears[3].color = vk::ClearColorValue(std::array<uint32_t, 4>{0xFFFFFFFFu, 0u, 0u, 0u});
1659 clears[4].color = vk::ClearColorValue(std::array<float, 4>{0, 0, 0, 0});
1660 clears[5].depthStencil = vk::ClearDepthStencilValue{1.0f, 0};
1661 vk::RenderPassBeginInfo rpBegin{};
1662 rpBegin.renderPass = gbufferVisRenderPass;
1663 rpBegin.framebuffer = slot->visFramebuffer;
1664 rpBegin.renderArea = vk::Rect2D{{0, 0}, {w, h}};
1665 rpBegin.clearValueCount = uint32_t(clears.size());
1666 rpBegin.pClearValues = clears.data();
1667 slot->normal.beginColorAttachment();
1668 slot->depthColor.beginColorAttachment();
1669 slot->albedo.beginColorAttachment();
1670 slot->visID.beginColorAttachment();
1671 slot->visBary.beginColorAttachment();
1672 slot->depth.beginDepthAttachment();
1673 cb.beginRenderPass(rpBegin, vk::SubpassContents::eInline);
1674 setViewportAndScissor(cb, w, h);
1675
1676 const Graphics::GpuDrivenFrameSet0 s0 = gpuDrivenFrameSet0();
1677 const vk::DescriptorSet bindless = bindlessSetForFrame();
1678 if (s0.set && bindless) {
1679 const uint32_t dynOffsets[2] = {s0.uboOffset, s0.shadowOffset};
1680 cb.bindPipeline(vk::PipelineBindPoint::eGraphics, gbufferVisPipeline);
1681 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 0,
1682 1, &s0.set, 2, dynOffsets);
1683 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 1,
1684 1, &bindless, 0, nullptr);
1685 const vk::DeviceSize stride = sizeof(glm::uvec4);
1686 for (uint32_t b = 0; b < gpuDrivenBucketCount_; ++b) {
1687 cb.drawIndirect(cull.indirectNI.buffer, vk::DeviceSize(b) * stride, 1, stride);
1688 }
1689 if (vgAnyThisFrame_) drawVgClusters(cb);
1690 }
1691 cb.endRenderPass();
1692 slot->normal.endSampledLayout();
1693 slot->depthColor.endSampledLayout();
1694 slot->albedo.endSampledLayout();
1695 slot->visID.endSampledLayout();
1696 slot->visBary.endSampledLayout();
1697 slot->depth.endSampledLayout();
1698}
1699
1701 if (!gpuDrivenCullReady_) return;
1702 if (gpuDrivenBucketCount_ == 0 && !vgAnyThisFrame_) return;
1703 if (!swapchainPassOpen && !sceneColorPassOpen) return;
1704 if (!resolveVisPipeline || !mesh3dGpuDrivenPipelineLayout) return;
1705 auto *slot = currentGBufferSlot();
1706 if (!slot || !slot->visIDGpu.sampler || !slot->visBaryGpu.sampler) return;
1707 auto &cb = currentPresentCb();
1708 const vk::DescriptorSet bindless = bindlessSetForFrame();
1709 if (!bindless) return;
1710
1711 const Graphics::GpuDrivenFrameSet0 s0 = gpuDrivenFrameSet0();
1712 if (!s0.set) return;
1713 const uint32_t dynOffsets[2] = {s0.uboOffset, s0.shadowOffset};
1714 cb.bindPipeline(vk::PipelineBindPoint::eGraphics, resolveVisPipeline);
1715 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 0, 1,
1716 &s0.set, 2, dynOffsets);
1717 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 1, 1,
1718 &bindless, 0, nullptr);
1719 const uint32_t w =
1720 uint32_t(sceneColorPassOpen ? sceneColorWidth : int(swapchain.extent.width));
1721 const uint32_t h =
1722 uint32_t(sceneColorPassOpen ? sceneColorHeight : int(swapchain.extent.height));
1723 setViewportAndScissor(cb, w, h);
1724 cb.draw(3, 1, 0, 0);
1725}
1726
1727void Graphics::createResolveVisPipeline(const vkb::BuiltRenderPass &rp,
1728 vk::SampleCountFlagBits samples) {
1729 destroyPipeline(device, resolveVisPipeline);
1730 if (!mesh3dGpuDrivenPipelineLayout) return;
1731 std::vector<uint32_t> vert(resolve_vis_vert_spv, resolve_vis_vert_spv +
1732 resolve_vis_vert_spv_count);
1733 std::vector<uint32_t> frag(resolve_vis_frag_spv, resolve_vis_frag_spv +
1734 resolve_vis_frag_spv_count);
1735 vk::ShaderModule vertModule = vkb::PipelineBuilder::createShaderModule(device.instance, vert);
1736 vk::ShaderModule fragModule = vkb::PipelineBuilder::createShaderModule(device.instance, frag);
1737 resolveVisPipeline =
1738 device.createPipeline()
1739 .useClassicPipeline(vertModule, fragModule)
1740 .setPipelineLayout(mesh3dGpuDrivenPipelineLayout)
1741 .setDynamicStatesViewportScissor()
1742 .setRasterizer(vk::PolygonMode::eFill, false, false, 1.0f,
1743 vk::CullModeFlagBits::eNone, vk::FrontFace::eClockwise)
1744 .setMultisampler(false, samples)
1745 .setDepthStencil(true, true, vk::CompareOp::eLessOrEqual)
1746 .setColorAttachmentCount(1)
1747 .build(rp);
1748 device->destroyShaderModule(vertModule);
1749 device->destroyShaderModule(fragModule);
1750}
1751
1752void Graphics::recordVgCull() {
1753 if (!gpuDrivenCullReady_ || vgAssetCount_ == 0 || vgClusterCount_ == 0) return;
1754 if (!vgCullPass_.pipeline() || !vgGpu_.visible.buffer) return;
1755 auto &cb = currentPresentCb();
1756 const vk::DescriptorSet bindless = bindlessSetForFrame();
1757 if (!bindless) return;
1758
1759 const size_t slot = currentFrameSlot() % kAsyncResourceCopies;
1760 vgLastVisible_ = uint32_t(slot);
1761
1762 // Uploads and per-frame instance/material projections are written through
1763 // host-visible mappings. Make those writes available before the compute
1764 // cull and the following vertex/fragment fetches; validation layers can
1765 // otherwise hide this race by perturbing submission timing.
1766 {
1767 std::array<vk::BufferMemoryBarrier, 6> barriers{};
1768 const vk::Buffer buffers[] = {
1769 vgGpu_.positions.buffer, vgGpu_.triangles.buffer, vgGpu_.clusters.buffer,
1770 vgGpu_.clusterAssets.buffer, vgGpu_.assetMaterials.buffer, vgGpu_.assetModels.buffer,
1771 };
1772 for (size_t i = 0; i < barriers.size(); ++i) {
1773 barriers[i].buffer = buffers[i];
1774 barriers[i].size = VK_WHOLE_SIZE;
1775 barriers[i].srcAccessMask = vk::AccessFlagBits::eHostWrite;
1776 barriers[i].dstAccessMask = vk::AccessFlagBits::eShaderRead;
1777 }
1778 cb.pipelineBarrier(vk::PipelineStageFlagBits::eHost,
1779 vk::PipelineStageFlagBits::eComputeShader |
1780 vk::PipelineStageFlagBits::eVertexShader |
1781 vk::PipelineStageFlagBits::eFragmentShader,
1782 {}, 0, nullptr, uint32_t(barriers.size()), barriers.data(), 0, nullptr);
1783 }
1784
1785 // Reset this slot's visible counter (the slot's fence was waited at frame
1786 // begin, so the previous use of this slot's buffers has completed).
1787 {
1788 const vk::DeviceSize slotVis = (vk::DeviceSize(kMaxVgClusters) + 1) * sizeof(uint32_t);
1789 void *map = vgGpu_.visible.map();
1790 if (!map) return;
1791 std::memset(static_cast<char *>(map) + slotVis * slot, 0, sizeof(uint32_t));
1792 vgGpu_.visible.unmap();
1793 }
1794
1795 cb.bindDescriptorSets(vk::PipelineBindPoint::eCompute, gpuDrivenComputeLayout, 1, 1,
1796 &bindless, 0, nullptr);
1797 const uint32_t push[4]{vgClusterCount_, 0u, 0u, 0u};
1798 cb.pushConstants(gpuDrivenComputeLayout, vk::ShaderStageFlagBits::eCompute, 0, sizeof(push),
1799 push);
1800 vgCullPass_.record(cb, (vgClusterCount_ + 63u) / 64u);
1801
1802 // VG cull writes (visible + indirect) -> vertex read + indirect read.
1803 {
1804 vk::BufferMemoryBarrier bmb[2]{};
1805 bmb[0].buffer = vgGpu_.visible.buffer;
1806 bmb[0].size = VK_WHOLE_SIZE;
1807 bmb[0].srcAccessMask = vk::AccessFlagBits::eShaderWrite;
1808 bmb[0].dstAccessMask = vk::AccessFlagBits::eShaderRead;
1809 bmb[1].buffer = vgGpu_.indirect.buffer;
1810 bmb[1].size = VK_WHOLE_SIZE;
1811 bmb[1].srcAccessMask = vk::AccessFlagBits::eShaderWrite;
1812 bmb[1].dstAccessMask = vk::AccessFlagBits::eIndirectCommandRead;
1813 cb.pipelineBarrier(vk::PipelineStageFlagBits::eComputeShader,
1814 vk::PipelineStageFlagBits::eVertexShader |
1815 vk::PipelineStageFlagBits::eDrawIndirect,
1816 {}, 0, nullptr, 2, bmb, 0, nullptr);
1817 }
1818}
1819
1820void Graphics::drawVgClusters(vk::CommandBuffer cb) {
1821 if (!gbufferVgVisPipeline || vgAssetCount_ == 0 || vgClusterCount_ == 0) return;
1822 const vk::DescriptorSet bindless = bindlessSetForFrame();
1823 if (!bindless || !mesh3dGpuDrivenPipelineLayout) return;
1824 const size_t slot = currentFrameSlot() % kAsyncResourceCopies;
1825 const vk::DeviceSize slotInd = vk::DeviceSize(kMaxVgClusters) * sizeof(glm::uvec4);
1826 const vk::DeviceSize slotVis = (vk::DeviceSize(kMaxVgClusters) + 1) * sizeof(uint32_t);
1827 cb.bindPipeline(vk::PipelineBindPoint::eGraphics, gbufferVgVisPipeline);
1828 cb.bindDescriptorSets(vk::PipelineBindPoint::eGraphics, mesh3dGpuDrivenPipelineLayout, 1, 1,
1829 &bindless, 0, nullptr);
1830 cb.drawIndirectCount(vgGpu_.indirect.buffer, slotInd * slot, vgGpu_.visible.buffer,
1831 slotVis * slot, kMaxVgClusters, sizeof(glm::uvec4));
1832}
1833
1835 if (!gpuDrivenCullReady_ || gpuDrivenCullSlots_.empty()) return 0;
1836 auto &slot = gpuDrivenCullSlots_[gpuDrivenLastCullSlot_ % gpuDrivenCullSlots_.size()];
1837 uint32_t count = 0;
1838 void *map = slot.visibleFlags.map();
1839 if (!map) return 0;
1840 const uint32_t n = std::min(gpuDrivenCullInstanceCount_, kMaxGpuDrivenInstances);
1841 const auto *flags = static_cast<const uint32_t *>(map);
1842 for (uint32_t i = 0; i < n; ++i) count += flags[i] != 0 ? 1u : 0u;
1843 slot.visibleFlags.unmap();
1844 return count;
1845}
1846
1848 if (!gpuDrivenCullReady_ || gpuDrivenCullSlots_.empty()) return 0;
1849 auto &slot = gpuDrivenCullSlots_[gpuDrivenLastCullSlot_ % gpuDrivenCullSlots_.size()];
1850 uint32_t count = 0;
1851 void *map = slot.indirect.map();
1852 if (!map) return 0;
1853 const auto *cmds = static_cast<const GpuIndirectCommand *>(map);
1854 const uint32_t n = std::min(gpuDrivenBucketCount_, kMaxGpuDrivenBuckets);
1855 for (uint32_t b = 0; b < n; ++b) count += cmds[b].instanceCount > 0 ? 1u : 0u;
1856 slot.indirect.unmap();
1857 return count;
1858}
1859
1861 if (!tex || !tex->gpuHandle) return kInvalidBindlessSlot;
1862 return static_cast<GpuTexture *>(tex->gpuHandle)->bindlessIndex2D;
1863}
1864
1866 if (!mesh || !mesh->gpuHandle) return kInvalidBindlessSlot;
1867 return static_cast<GpuMesh *>(mesh->gpuHandle)->gpuRecordIndex;
1868}
1869
1870void Graphics::createMesh3DGpuDrivenPipeline() {
1871 if (mesh3dGpuDrivenPipeline) return;
1872 if (!mesh3dSetLayout || !bindlessSetLayout_ || !gpuDrivenCaps_.gpuDrivenAvailable()) return;
1873
1874 // Pipeline layout: set0 = per-frame (legacy mesh3d layout), set1 = bindless.
1875 // No push constants: instance indexing relies on gl_InstanceIndex, which
1876 // already includes the command's firstInstance per the Vulkan spec.
1877 std::array<vk::DescriptorSetLayout, 2> setLayouts{mesh3dSetLayout, bindlessSetLayout_};
1878 vk::PipelineLayoutCreateInfo pli{};
1879 pli.setLayoutCount = uint32_t(setLayouts.size());
1880 pli.pSetLayouts = setLayouts.data();
1881 mesh3dGpuDrivenPipelineLayout = device->createPipelineLayout(pli);
1882 mesh3dGpuDrivenPipeline =
1883 createMesh3DStylePipeline(embeddedSpirv(mesh3d_gpudriven_vert_spv),
1884 embeddedSpirv(mesh3d_gpudriven_frag_spv),
1885 mesh3dGpuDrivenPipelineLayout, renderpass,
1886 vk::SampleCountFlagBits::e1);
1887}
1888} // namespace eve::graphics::vulkan
float w
Definition AnimClip.cpp:738
EVEngine assertion entry point, backed by zeroerr.
#define EV_ASSERT(cond,...)
Assert an internal engine invariant (state that must always hold).
Definition Assert.h:37
const std::string & s
std::vector< BuildingInstanceSnapshot > instances
float planes[6][4]
glm::vec4 p[6]
vk::ShaderModule vert
vk::ShaderModule frag
std::uint32_t firstInstance
std::uint32_t instanceCount
float u
Definition Grass.cpp:233
glm::vec3 n
Definition Grass.cpp:63
std::int32_t first
int h
std::uint32_t height
std::uint32_t width
size_t offset
bool required
MeleePoint3 b
Definition MeleeHit.cpp:41
MeleePoint3 a
Definition MeleeHit.cpp:40
std::vector< std::int32_t > order
uint32_t groups
Definition OnnxGpgpu.cpp:39
std::vector< std::shared_ptr< DeviceBytes > > bindings
Definition OnnxGpgpu.cpp:38
std::unique_ptr< gpgpu::GpuBuffer > buffer
Definition OnnxGpgpu.cpp:26
eve::action::ActionVfxBinding binding
int idx
float d
float t
glm::vec3 eye
float fovRad
Mesh * mesh
glm::mat4 viewProj
glm::mat4 model
Material * material
bool found
std::uint32_t count
float size
Definition TreeMesh.cpp:156
std::vector< VegetationPresetCommand > commands
float m[16]
static Diagnostic error(DiagnosticCode code, std::string message, std::string path={}, DiagnosticDetails details={}, std::string source={})
Construct an error diagnostic with the standard error severity.
Definition Diagnostic.h:125
Move-only operation result carrying either a value or Status.
Definition Result.h:155
static Result success(T value)
Construct a successful result owning value.
Definition Result.h:164
static Result failure(Status status)
Construct a failed result from a structured status.
Definition Result.h:175
void push(bool all)
Pushes .
std::unique_ptr< RenderControl > renderControl_
Definition Graphics.h:2264
CPU-side builder for GPU-driven indirect draws (stage 1).
uint32_t build(const std::vector< GpuMeshRecord > &meshTable)
Sort + merge. meshTable supplies indexCount/firstIndex/vertexBase.
void add(uint32_t seq, uint32_t meshId, uint32_t materialId, uint32_t pipelineId)
seq = monotonic instance sequence (stable order tiebreak).
const std::vector< GpuIndirectCommand > & commands() const
Commands.
const std::vector< uint32_t > & sortedInstanceOrder() const
Instance indices in the sorted order the caller must upload.
Packages shading method + surface parameters into one attachable asset.
Definition Material.h:35
GPU mesh handle (+ optional CPU morph targets).
Definition Mesh.h:25
GPU texture created via Graphics::newTexture. Owns GPU resources through an opaque backend handle.
Definition Texture.h:18
bool create(vkb::Device &device, vk::PipelineLayout layout, const std::vector< uint32_t > &spv)
Create from embedded SPIR-V words; layout must outlive the pass.
vk::Pipeline pipeline() const
Pipeline.
Definition ComputePass.h:42
void record(vk::CommandBuffer cb, uint32_t groupsX, uint32_t groupsY=1, uint32_t groupsZ=1) const
Record one dispatch (local size baked into the shader).
Per-frame GPU allocation arena (host-visible coherent).
Definition FrameArena.h:18
bool gpuDrivenCullEnabled() const
Stage 2 cull is live for this frame (GPU-written commands).
Definition Graphics.h:507
Result< void > gpuDrivenReleaseMaterialRecord(Material *material) override
Gpu driven release material record.
uint32_t gpuDrivenMeshRecord(Mesh *mesh) override
Gpu driven mesh record.
uint32_t gpuDrivenVgAssetId(Mesh *mesh) const override
Gpu driven vg asset id.
uint32_t debugMeshRecordIndex(Mesh *mesh) const
Debug mesh record index.
bool gpuDrivenVgSetInstance(uint32_t vgAssetId, const glm::mat4 &model, uint32_t materialId) override
Gpu driven vg set instance.
void gpuDrivenVgComputeSection(const glm::mat4 &viewProj, const glm::vec3 &eye, float fovYDeg, float nearZ, float farZ) override
Gpu driven vg compute section.
bool gpuDrivenMaterialUsable(Material *material) override
Gpu driven material usable.
GpuResidentSubmitStatus gpuDrivenSubmitResident(const GpuResidentInstanceBatch &batch) override
Gpu driven submit resident.
void syncMaterialTable()
Upload all registered material records to the GPU table.
uint32_t debugGpuDrivenCulledDrawCount() const
Debug gpu driven culled draw count.
uint32_t debugGpuDrivenVgVisibleCount() const
Debug readback: visible VG clusters from the last cull.
uint32_t materialTableGetOrCreate(eve::graphics::Material *material)
Get (or lazily create) the GPU material-table slot for a material.
bool gpuDrivenCullBegin(const GpuInstance *instances, uint32_t instanceCount)
Gpu driven cull begin.
void gpuDrivenResolve() override
Gpu driven resolve.
bool gpuDrivenResolveWanted() const override
Stage 3: vis+resolve live for this frame (opt-in, 1x scene pass).
void gpuDrivenDrawOpaque()
Gpu driven draw opaque.
bool gpuDrivenEnabled() const override
Gpu driven enabled.
Definition Graphics.h:450
uint32_t gpuDrivenVgUpload(const GpuVgAssetUpload &asset) override
Gpu driven vg upload.
void gpuDrivenOpenScenePass()
Gpu driven open scene pass.
bool gpuDrivenScenePassPending() const
Scene color pass is deferred until after the compute cull section.
Definition Graphics.h:511
uint32_t debugBindlessIndex(Texture *tex) const
Test/debug helpers (valid when the GPU-driven path is live).
uint32_t debugGpuDrivenVisibleCount() const
Debug readback: visible instances / non-empty buckets from the last cull.
void gpuDrivenRecordVisPass() override
Gpu driven record vis pass.
bool gpuDrivenVgAttachToMesh(Mesh *mesh, uint32_t vgAssetId) override
Gpu driven vg attach to mesh.
bool gpuDrivenSubmitOpaque(const GpuInstance *instances, uint32_t instanceCount) override
Gpu driven submit opaque.
void gpuDrivenCullEmit(const glm::mat4 &viewProj, const glm::vec3 &eye, float fovYDeg, float nearZ, float farZ)
Gpu driven cull emit.
uint32_t gpuDrivenReflectionProbeSlot(Texture *cubemap) override
Gpu driven reflection probe slot.
FrameArena & currentFrameArena()
Per-frame arena for the current swapchain frame slot.
std::vector< ParamSpec > params
constexpr uint32_t kMaxGpuDrivenBuckets
Definition GpuDriven.h:24
eve::graphics::GpuIndirectCommand GpuIndirectCommand
Definition GpuDriven.h:18
constexpr uint32_t kMaxHzbMips
Definition GpuDriven.h:25
constexpr uint32_t kMaxBindlessCubemaps
Definition GpuDriven.h:20
constexpr uint32_t kMaxGpuDrivenInstances
Definition GpuDriven.h:23
constexpr uint32_t kInvalidBindlessSlot
Definition GpuDriven.h:22
constexpr uint32_t kHZBHeaderWords
Definition GpuDriven.h:26
constexpr uint32_t kMaxBindlessTextures
Definition GpuDriven.h:19
eve::graphics::GpuMeshRecord GpuMeshRecord
GPU-driven rendering shared constants + std430 GPU layouts.
Definition GpuDriven.h:15
eve::graphics::GpuMaterialRecord GpuMaterialRecord
Definition GpuDriven.h:16
eve::graphics::GpuInstance GpuInstance
Definition GpuDriven.h:17
constexpr uint32_t kInvalidGpuDrivenSlot
GPU-driven rendering shared constants + std430 GPU layouts.
GpuResidentSubmitStatus
Structured result for direct resident-instance submission.
size_t elementSize(OnnxElement e)
Element size.
constexpr uint64_t kGpuResidentStorageOffsetAlignment
Portable alignment required for resident storage-buffer slice offsets.
Indirect draw command; layout identical to VkDrawIndexedIndirectCommand.
Per-instance GPU record (std430). Mirrors GLSL GpuInstance.
GPU material table record (std430). Mirrors GLSL GpuMaterialRecord.
GPU mesh table record (std430). Mirrors GLSL GpuMeshRecord.
Direct-render description for a GPU-authored array of GpuInstance records. @ownership buckets and buf...
const GpuResidentInstanceBucket * buckets
One contiguous mesh/material bucket in a sorted resident instance buffer.
Neutral GPU upload for one virtual-geometry asset. Raw arrays so the graphics module does not depend ...
const std::uint32_t * triangles
GPU-packed cluster node (std430, 4 x uvec4). Mirrors the virtualgeometry module's VgGpuCluster layout...
static constexpr int kMaxLights
Definition Light.h:173
Light3DGpu lights[kMaxLights]
Definition Light.h:175
bool gpuDrivenAvailable() const
Gpu driven available.
Definition GpuDriven.h:41
bool gpuDrivenCullAvailable() const
Stage 2: GPU frustum/HZB cull + GPU-written indirect commands.
Definition GpuDriven.h:56
GpuMesh public API.
Definition Graphics.h:289
GpuMeshRecord record
CPU-side record for this mesh (bounds/ranges); uploaded by registerMeshRecord.
Definition Graphics.h:311
GpuTexture public API.
Definition Graphics.h:257
uint32_t bindlessIndex2D
Bindless texture-array slot (stage 0 GPU-driven path).
Definition Graphics.h:276
uint32_t bucket
glm::uvec4 info