19 return feature ==
"grid_topology" || feature ==
"distance_constraints" || feature ==
"self_collision" ||
20 feature ==
"bounds_collision" || feature ==
"interaction_force";
38const char *kClothKernel = R
"glsl(#version 450
39layout(local_size_x = 64) in;
41layout(set = 0, binding = 0) buffer Pos {
42 vec4 posIn[]; // x, y, prevX, prevY (read)
44layout(set = 0, binding = 1) buffer PosOut {
45 vec4 posOut[]; // x, y, prevX, prevY (write)
47layout(set = 0, binding = 2) buffer Links {
48 vec4 links[]; // otherIndex, restLen, 0, 0 per particle slot
50layout(set = 0, binding = 3) buffer Flags {
51 vec4 flags[]; // x = 1.0 when pinned
53layout(set = 0, binding = 4) buffer CellHead {
54 uint cellHead[]; // per-cell linked-list head; 0xFFFFFFFF = empty
56layout(set = 0, binding = 5) buffer CellNext {
57 uint cellNext[]; // slot -> next slot in the same cell
59layout(set = 0, binding = 6) buffer CellItems {
60 uint cellItems[]; // slot -> particle index
62layout(set = 0, binding = 7) buffer CellSlot {
63 uint cellSlot[1]; // global slot counter for hash building
66layout(push_constant) uniform PC {
70// Keep two non-adjacent particles at least minDist apart. Pinned endpoints are
71// handled by their own thread being skipped: the free side takes the full
72// correction; two free sides split it.
73void separatePair(uint i, uint j, inout vec4 p) {
75 const uint K = uint(pc.data[18]);
77 for (uint k = 0u; k < K; ++k) {
78 vec4 l = links[i * K + k];
79 if (int(l.x) < 0) break;
80 if (int(l.x) == int(j)) {
86 bool jPinned = flags[j].x > 0.5;
90 float d2 = dx * dx + dy * dy;
91 float minDist = pc.data[20];
92 if (d2 >= minDist * minDist || d2 < 1e-8) return;
94 float corr = min(2.0, (minDist - d) / d);
95 float w = jPinned ? 1.0 : 0.5;
101 // Reset the hash: every cell head becomes empty, the global slot counter
102 // goes back to zero. Runs once per substep before building.
103 if (int(pc.data[19]) == 4) {
104 uint cid = gl_GlobalInvocationID.x;
105 if (cid < uint(pc.data[24])) cellHead[cid] = 0xFFFFFFFFu;
106 if (cid == 0u) cellSlot[0] = 0u;
110 uint i = gl_GlobalInvocationID.x;
111 uint n = uint(pc.data[5]);
114 const float dt = pc.data[0];
115 const float damp = 1.0 - pc.data[6];
116 const bool pinned = flags[i].x > 0.5;
117 const uint K = uint(pc.data[18]);
118 const int mode = int(pc.data[19]);
121 if (mode == 0 && !pinned) {
123 float vx = (p.x - p.z) * damp;
124 float vy = (p.y - p.w) * damp;
127 float ax = pc.data[1] + pc.data[12];
128 float ay = pc.data[2] + pc.data[13];
129 p.x += vx + ax * dt * dt;
130 p.y += vy + ay * dt * dt;
132 // Pointer field (Fluid2D-style attract/repel).
133 float ir = pc.data[16];
134 if (ir > 0.0 && abs(pc.data[17]) > 1e-9) {
135 float dx = pc.data[14] - p.x;
136 float dy = pc.data[15] - p.y;
137 float r2 = dx * dx + dy * dy;
138 if (r2 < ir * ir && r2 > 1e-6) {
140 float w = 1.0 - r / ir;
141 p.x += (dx / r) * pc.data[17] * w * dt * dt;
142 p.y += (dy / r) * pc.data[17] * w * dt * dt;
147 // Distance-constraint relaxation over this particle's link slots. The
148 // owning thread never moves a pinned particle: a pinned endpoint leaves
149 // the full correction to the free partner's thread. Runs in a separate
150 // dispatch so integration results are visible before solving.
151 if (mode == 1 && !pinned) {
152 // One Jacobi pass per dispatch: each thread reads the previous pass's
153 // positions (posIn) and writes its own corrected position (posOut).
154 // The outer loop in update() repeats the dispatch to converge.
155 for (uint k = 0u; k < K; ++k) {
156 vec4 l = links[i * K + k];
157 int other = int(l.x);
158 if (other < 0) break;
159 vec4 q = posIn[uint(other)];
160 bool otherPinned = flags[uint(other)].x > 0.5;
161 float dx = q.x - p.x;
162 float dy = q.y - p.y;
163 float dist = sqrt(dx * dx + dy * dy);
164 if (dist < 1e-5) continue;
165 float diff = (dist - l.y) / dist * pc.data[3];
170 p.x += dx * diff * 0.5;
171 p.y += dy * diff * 0.5;
176 // Build the spatial hash: each particle atomically claims a global slot and
177 // prepends it to its cell's linked list (overflow-free).
178 if (mode == 2 && pc.data[21] > 0.5) {
179 int nx = int(pc.data[22]);
180 int ny = int(pc.data[23]);
181 if (nx > 0 && ny > 0) {
182 float cellSize = pc.data[20];
183 float bx = pc.data[7];
184 float by = pc.data[8];
185 int cx = clamp(int(floor((p.x - bx) / cellSize)), 0, nx - 1);
186 int cy = clamp(int(floor((p.y - by) / cellSize)), 0, ny - 1);
187 uint cid = uint(cy * nx + cx);
188 uint slot = atomicAdd(cellSlot[0], 1u);
189 if (slot >= n) return;
190 cellNext[slot] = atomicExchange(cellHead[cid], slot);
195 // Particle-level self-collision via the hash (3x3 neighbor cells) or an
196 // O(n) scan fallback when no bounds define a hash grid.
197 if (mode == 3 && pc.data[21] > 0.5 && !pinned) {
198 int nx = int(pc.data[22]);
199 int ny = int(pc.data[23]);
200 if (nx > 0 && ny > 0 && pc.data[20] > 0.0) {
201 float cellSize = pc.data[20];
202 float bx = pc.data[7];
203 float by = pc.data[8];
204 int cx = clamp(int(floor((p.x - bx) / cellSize)), 0, nx - 1);
205 int cy = clamp(int(floor((p.y - by) / cellSize)), 0, ny - 1);
206 for (int oy = -1; oy <= 1; ++oy) {
207 for (int ox = -1; ox <= 1; ++ox) {
208 int ncx = clamp(cx + ox, 0, nx - 1);
209 int ncy = clamp(cy + oy, 0, ny - 1);
210 uint cid = uint(ncy * nx + ncx);
211 uint s = cellHead[cid];
212 while (s != 0xFFFFFFFFu) {
213 separatePair(i, cellItems[s], p);
219 for (uint j = 0u; j < n; ++j) separatePair(i, j, p);
223 // Axis-aligned bounds with a soft bounce (mirrors the CPU cloth).
224 if (mode == 1 && pc.data[11] > 0.5 && !pinned) {
225 float bx = pc.data[7];
226 float by = pc.data[8];
227 float bw = pc.data[9];
228 float bh = pc.data[10];
230 float vx = p.x - p.z;
232 p.z = p.x + vx * 0.35;
233 } else if (p.x > bx + bw) {
234 float vx = p.x - p.z;
236 p.z = p.x + vx * 0.35;
239 float vy = p.y - p.w;
241 p.w = p.y + vy * 0.35;
242 } else if (p.y > by + bh) {
243 float vy = p.y - p.w;
245 p.w = p.y + vy * 0.35;
259 if (!gpgpu_)
throw Exception(
"ClothGPU: Gpgpu module required");
260 if (cols_ < 2 || rows_ < 2)
throw Exception(
"ClothGPU: cols and rows must be >= 2");
261 if (spacing_ <= 0.f)
throw Exception(
"ClothGPU: spacing must be > 0");
264 posCpu_.assign(
static_cast<size_t>(
count) * 4, 0.f);
265 flagsCpu_.assign(
static_cast<size_t>(
count), 0.f);
266 for (
int r = 0;
r < rows_; ++
r) {
267 for (
int c = 0;
c < cols_; ++
c) {
268 const size_t i =
static_cast<size_t>(
r * cols_ +
c);
269 posCpu_[i * 4 + 0] = originX_ + float(
c) * spacing_;
270 posCpu_[i * 4 + 1] = originY_ + float(
r) * spacing_;
271 posCpu_[i * 4 + 2] = posCpu_[i * 4 + 0];
272 posCpu_[i * 4 + 3] = posCpu_[i * 4 + 1];
278 for (
int c = 0; c < cols_; ++c) flagsCpu_[static_cast<size_t>(
c)] = 1.f;
280 posBuf_ = gpgpu_->
newBuffer(
count *
int(4 *
sizeof(
float)),
"storage");
281 posBufB_ = gpgpu_->
newBuffer(
count *
int(4 *
sizeof(
float)),
"storage");
284 flagBuf_ = gpgpu_->
newBuffer(
count *
int(4 *
sizeof(
float)),
"storage");
285 staging_ = gpgpu_->
newBuffer(
count *
int(4 *
sizeof(
float)),
"staging");
286 if (!posBuf_ || !posBufB_ || !linkBuf_ || !flagBuf_ || !staging_)
287 throw Exception(
"ClothGPU: failed to allocate storage buffers");
288 uploadInitialState();
290 shader_ = gpgpu_->
newShader(kClothKernel);
291 if (!shader_)
throw Exception(
"ClothGPU: compute shader compile failed");
294 throw Exception(
"ClothGPU: command sequence unavailable");
300 if (destroyed_)
return;
317 cellHeadBuf_ =
nullptr;
319 cellNextBuf_ =
nullptr;
320 delete cellItemsBuf_;
321 cellItemsBuf_ =
nullptr;
323 cellSlotBuf_ =
nullptr;
326void ClothGPU::rebuildLinks() {
330 const auto findFreeSlot = [&](
int i) {
333 if (
s.other < 0)
return k;
337 auto add = [&](
int r,
int c,
int nr,
int nc) {
338 if (nr < 0 || nc < 0 || nr >= rows_ || nc >= cols_)
return;
339 const int i =
r * cols_ +
c;
340 const int j = nr * cols_ + nc;
344 const float dx = (float(nc) - float(
c)) * spacing_;
345 const float dy = (float(nr) - float(
r)) * spacing_;
346 slot.rest = std::sqrt(
dx *
dx +
dy *
dy);
349 for (
int r = 0;
r < rows_; ++
r) {
350 for (
int c = 0;
c < cols_; ++
c) {
355 add(
r,
c,
r + 1,
c + 1);
356 add(
r,
c,
r + 1,
c - 1);
357 add(
r,
c,
r - 1,
c + 1);
358 add(
r,
c,
r - 1,
c - 1);
366 for (
size_t i = 0; i < links_.size(); ++i) {
367 const Link &link = links_[i];
368 linkCpu_[i * 4 + 0] = float(link.other);
369 linkCpu_[i * 4 + 1] = link.rest;
370 linkCpu_[i * 4 + 2] = 0.f;
371 linkCpu_[i * 4 + 3] = 0.f;
375void ClothGPU::uploadInitialState() {
383void ClothGPU::uploadPinned() {
385 std::vector<float> buf(
static_cast<size_t>(
count) * 4, 0.f);
386 for (
int i = 0; i < count; ++i) buf[static_cast<size_t>(i) * 4] = flagsCpu_[
static_cast<size_t>(i)];
390void ClothGPU::ensureHashBuffers() {
391 const float minDist = std::max(1e-3f, particleSize_ * 2.f);
392 if (!selfCollision_ || !hasBounds_) {
393 if (hashNx_ != 0 || hashNy_ != 0) {
395 cellHeadBuf_ =
nullptr;
397 cellNextBuf_ =
nullptr;
398 delete cellItemsBuf_;
399 cellItemsBuf_ =
nullptr;
401 cellSlotBuf_ =
nullptr;
402 hashNx_ = hashNy_ = 0;
406 const int nx = std::max(1,
int(std::ceil(boundW_ / minDist)));
407 const int ny = std::max(1,
int(std::ceil(boundH_ / minDist)));
408 const int nCells =
nx *
ny;
410 if (
nx == hashNx_ &&
ny == hashNy_)
return;
412 cellHeadBuf_ =
nullptr;
414 cellNextBuf_ =
nullptr;
415 delete cellItemsBuf_;
416 cellItemsBuf_ =
nullptr;
418 cellSlotBuf_ =
nullptr;
419 cellHeadBuf_ = gpgpu_->
newBuffer(nCells *
int(
sizeof(uint32_t)),
"storage");
420 cellNextBuf_ = gpgpu_->
newBuffer(
count *
int(
sizeof(uint32_t)),
"storage");
421 cellItemsBuf_ = gpgpu_->
newBuffer(
count *
int(
sizeof(uint32_t)),
"storage");
422 cellSlotBuf_ = gpgpu_->
newBuffer(
int(
sizeof(uint32_t)),
"storage");
433 stiffness_ = std::clamp(stiffness, 0.f, 1.f);
441 damping_ = std::clamp(damping, 0.f, 1.f);
445 particleSize_ = std::max(1.f,
size);
451 if (
w <= 0.f ||
h <= 0.f) {
466 throw Exception(
"ClothGPU.pin: index out of range");
467 flagsCpu_[
static_cast<size_t>(
index)] = 1.f;
473 throw Exception(
"ClothGPU.unpin: index out of range");
474 flagsCpu_[
static_cast<size_t>(
index)] = 0.f;
479 for (
int c = 0; c < cols_; ++c) flagsCpu_[static_cast<size_t>(
c)] = 1.f;
485 return flagsCpu_[
static_cast<size_t>(
index)] > 0.5f;
496 interactRadius_ = std::max(0.f,
radius);
509 return posCpu_[
static_cast<size_t>(
index) * 4 + 0];
514 return posCpu_[
static_cast<size_t>(
index) * 4 + 1];
518 for (
int r = 0;
r < rows_; ++
r) {
519 for (
int c = 0;
c < cols_; ++
c) {
520 const size_t i =
static_cast<size_t>(
r * cols_ +
c);
521 posCpu_[i * 4 + 0] = originX_ + float(
c) * spacing_;
522 posCpu_[i * 4 + 1] = originY_ + float(
r) * spacing_;
523 posCpu_[i * 4 + 2] = posCpu_[i * 4 + 0];
524 posCpu_[i * 4 + 3] = posCpu_[i * 4 + 1];
527 std::fill(flagsCpu_.begin(), flagsCpu_.end(), 0.f);
529 forceX_ = forceY_ = 0.f;
530 interactStrength_ = 0.f;
531 uploadInitialState();
535 if (destroyed_)
return;
536 if (dt < 0.f) dt = 0.f;
537 if (dt > 0.05f) dt = 0.05f;
538 auto result = stepGpu(dt, 2);
539 result.ignore(
"legacy ClothGPU::update cannot return a structured result");
543 if (destroyed_ || !shader_ || !gpgpu_ || !seq_)
545 "GPU cloth resources are not available",
546 "physics.clothGpu.step"));
547 if (substeps < 1 || substeps > 1024)
549 "GPU cloth substep count must be in [1, 1024]",
550 "physics.clothGpu.step.substeps"));
554 const float h = dt / float(substeps);
557 if (cellHeadBuf_ && cellNextBuf_ && cellItemsBuf_ && cellSlotBuf_) {
567 shader_->
setFloat(4,
float(iterations_));
574 shader_->
setFloat(11, hasBounds_ ? 1.f : 0.f);
579 shader_->
setFloat(16, interactRadius_);
580 shader_->
setFloat(17, interactStrength_);
582 shader_->
setFloat(20, particleSize_ * 2.f);
583 shader_->
setFloat(21, selfCollision_ ? 1.f : 0.f);
584 shader_->
setFloat(22,
float(hashNx_));
585 shader_->
setFloat(23,
float(hashNy_));
586 const int nCells = hashNx_ > 0 && hashNy_ > 0 ? hashNx_ * hashNy_ : 0;
587 shader_->
setFloat(24,
float(nCells));
594 GpuBuffer *in = posBuf_;
595 GpuBuffer *out = posBufB_;
596 const auto pass = [&](
float mode) {
604 for (
int s = 0;
s < substeps; ++
s) {
606 if (selfCollision_ && nCells > 0) {
613 for (
int it = 0; it < iterations_; ++it) pass(1.f);
614 if (selfCollision_) {
615 for (
int sc = 0; sc < 2; ++sc)
626 interactStrength_ = 0.f;
634 auto valid = detail::validateSimulationStep(stepValue,
settings, observation_);
636 auto next = detail::advanceSimulationObservation(observation_, stepValue);
640 if (!applied)
return applied;
641 }
catch (
const std::exception &
error) {
643 std::string(
"GPU cloth step failed: ") +
error.what(),
644 "physics.clothGpu.step"));
649 observation_ = std::move(next).takeValue();
654 auto valid = detail::validateSimulationObservation(
observation,
"physics.clothGpu.restoreObservation");
658 "Cannot restore a destroyed GPU cloth",
659 "physics.clothGpu.restoreObservation"));
665 if (!gfx || destroyed_)
return;
666 const Color linkColor(colorR_, colorG_, colorB_, colorA_ * 0.75f);
667 const Color nodeColor(colorR_, colorG_, colorB_, colorA_);
669 for (
int i = 0; i <
count; ++i) {
672 if (link.other < 0)
break;
673 const float x1 = posCpu_[
static_cast<size_t>(i) * 4 + 0];
674 const float y1 = posCpu_[
static_cast<size_t>(i) * 4 + 1];
675 const float x2 = posCpu_[
static_cast<size_t>(link.other) * 4 + 0];
676 const float y2 = posCpu_[
static_cast<size_t>(link.other) * 4 + 1];
677 const float dx = x2 - x1;
678 const float dy = y2 - y1;
679 const float len = std::sqrt(
dx *
dx +
dy *
dy);
680 const int steps = std::max(1,
int(len / 4.f));
682 const float t = float(
s) / float(
steps);
687 for (
int i = 0; i <
count; ++i) {
688 const float s = flagsCpu_[
static_cast<size_t>(i)] > 0.5f
689 ? std::max(5.f, particleSize_ + 2.f)
691 gfx->
drawSolidRect(posCpu_[
static_cast<size_t>(i) * 4 + 0] -
s * 0.5f,
692 posCpu_[
static_cast<size_t>(i) * 4 + 1] -
s * 0.5f,
s,
s,
TerrainThermalSettings settings
static Diagnostic error(DiagnosticCode code, std::string message, std::string path={}, DiagnosticDetails details={}, std::string source={})
Construct an error diagnostic with the standard error severity.
double seconds() const noexcept
Return this duration as seconds for legacy/presentation APIs.
EVENGINE_API_FOUNDATION public API.
Move-only operation result carrying either a value or Status.
static Result success(T value)
Construct a successful result owning value.
static Result failure(Status status)
Construct a failed result from a structured status.
static Status success(StatusCode code=StatusCode::Ok)
Construct a successful status with an explicit non-error outcome.
virtual void setFloat(int index, float value)=0
Sets the float.
virtual void bindBuffer(int binding, GpuBuffer *buffer)=0
Bind a storage buffer to set=0 binding. binding in [0, kMaxBindings).
GPGPU module — compute shaders + storage buffers via the active Graphics backend. Uses the graphics q...
ComputeShader * newShader(const std::string &source)
Compatibility-only raw-owning shader factory (Vulkan: GLSL; WebGPU: WGSL). Vulkan delegates to the ch...
GpuBuffer * newBuffer(int byteSize, const std::string &usage="storage")
Allocate a GPU buffer. usage: "storage" (SSBO, device-local) | "staging" (host-visible transfer).
Sequence * newSequence()
Create a Kompute-style command Sequence: record buffer transfers and compute dispatches into one comm...
Backend-agnostic GPU buffer for compute (storage) or CPU staging transfers. Squirrel-owned; derived c...
virtual void writeFloat32s(const float *data, int count, int startIndex=0)=0
Bulk float upload/download (one transfer). startIndex is in floats.
virtual void downloadBytes(void *dst, uint64_t nbytes, uint64_t srcOffset=0) const =0
Downloads bytes.
void recordDownload(GpuBuffer *src, GpuBuffer *staging, uint64_t nbytes, uint64_t srcOffset=0)
Record download.
void recordDispatch(ComputeShader *shader, int groupsX, int groupsY=1, int groupsZ=1)
Record dispatch.
void begin()
Begins begin.
bool isAvailable() const
True when available.
virtual void drawSolidRect(float x, float y, float w, float h, float r, float g, float b, float a=1.f)
RGBA-float overload matching the script-facing drawSolidRect name.
void draw(graphics::Graphics *gfx)
Draw links + particles from the latest GPU readback.
void unpin(int index)
Unpin.
eve::Result< void > step(const eve::SimulationStep &step, const SimulationSettings &settings) override
Advances the production GPU cloth with the shared ticked contract.
float getParticleY(int index) const
Returns the particle y.
void pinTopRow()
Pin top row.
void setColor(float r, float g, float b, float a=1.f)
Sets the color.
int getParticleCount() const
Returns the particle count.
void setSelfCollision(bool on)
Enable proximity-based self-collision between non-adjacent particles Default is false....
eve::Result< void > restoreObservation(const SimulationObservation &observation) override
Restores tick/progress metadata after an owner-level restore.
void setDamping(float damping)
Damping applied to Verlet velocity [0,1] (default 0.01).
void update(float dt)
Updates .
void setBounds(float x, float y, float w, float h)
Axis-aligned walls; particles are clamped (with a small bounce).
void setParticleSize(float size)
Particle draw size in pixels (default 3).
bool supportsFeature(const std::string &feature) const
Query a stable cloth feature name; unsupported features never silently fall back.
static constexpr int kMaxLinksPerParticle
void setIterations(int iterations)
Constraint solver iterations per substep (default 4).
void setGravity(float gx, float gy)
Sets the gravity.
void applyForce(float fx, float fy)
Applies force.
bool isPinned(int index) const
True when pinned.
void reset()
Restore the flat grid pose (top row pinned) and re-upload state.
float getParticleX(int index) const
Returns the particle x.
void clearBounds()
Clears bounds.
void interactAt(float x, float y, float radius, float strength)
Pointer-field interaction like Fluid2D::interactAt: positive strength attracts, negative repels withi...
ClothGPU(eve::gpgpu::Gpgpu *gpgpu, int cols, int rows, float spacing, float originX, float originY)
Cloth gpu.
SimulationObservation observation() const noexcept override
Returns completed tick/time observables.
void setStiffness(float stiffness)
Constraint relaxation strength in [0,1] (default 0.85).
eve::Color Color
RGBA color used by every graphics draw call. Lives inside eve::graphics so including a graphics heade...
Optional physics backend for vehicle mobility and body attach.
glm::vec4 Color
Render-neutral RGBA color shared by graphics-facing modules.
One deterministic fixed-step emitted by SimulationClock.
Duration delta
Fixed simulation duration for this step.
Observable backend progress shared by CPU and accelerator providers.
Validated solver policy for one simulation step.