载入中...
搜索中...
未找到
OnnxGpuKernels.cpp
浏览该文件的文档.
2#include <sstream>
5namespace {
6std::string prefix(bool as, bool bs) {
7 std::ostringstream s;
8 s << "#version 450\nlayout(local_size_x=64) in;\n"
9 "layout(std430,binding=0) readonly buffer A {uint a[];};\n"
10 "layout(std430,binding=1) readonly buffer B {uint b[];};\n"
11 "layout(std430,binding=2) readonly buffer Z {int zp[];};\n"
12 "layout(std430,binding=3) writeonly buffer Y {int y[];};\n"
13 "int av(uint i){int v=int((a[i/4]>>((i%4)*8))&255u);return "
14 << (as ? "v>=128?v-256:v" : "v")
15 << ";}\n"
16 "int bv(uint i){int v=int((b[i/4]>>((i%4)*8))&255u);return "
17 << (bs ? "v>=128?v-256:v" : "v") << ";}\n";
18 return s.str();
19}
20// Compatibility host-byte calls use the same resident kernel generators. The
21// adapter below requests completion through the provider's synchronous contract.
22class Synchronous final : public OnnxCompute {
23 OnnxCompute& target;
24
25public:
26 explicit Synchronous(OnnxCompute& t) : target(t) {}
27 Result<std::vector<uint8_t>> dispatch(const OnnxKernel& k) override { return target.dispatch(k); }
28};
29OnnxBuffer hostBuffer(std::span<const uint8_t> bytes) {
30 auto p = std::make_shared<const std::vector<uint8_t>>(bytes.begin(), bytes.end());
31 return {p, {}, p->size()};
32}
33std::vector<int32_t> integerOutput(const OnnxBuffer& result) {
34 if (!result.host || result.host->size() % 4) throw Failure("Invalid synchronous integer output");
35 std::vector<int32_t> out(result.host->size() / 4);
36 std::memcpy(out.data(), result.host->data(), result.host->size());
37 return out;
38}
39} // namespace
40std::vector<int32_t> gpuMatmul(OnnxCompute& d, affine::ByteView a, affine::ByteView b, size_t m, size_t k, size_t n,
41 int az, std::span<const int32_t> bz) {
42 Synchronous sync(d);
43 return integerOutput(gpuMatmulResident(sync, hostBuffer(a.bytes), hostBuffer(b.bytes), a.signedValues,
44 b.signedValues, m, k, n, az, bz));
45}
47 std::span<const int32_t> wz) {
48 Synchronous sync(d);
49 return integerOutput(
50 gpuConvResident(sync, hostBuffer(x.bytes), hostBuffer(w.bytes), x.signedValues, w.signedValues, c, xz, wz));
51}
53 std::span<const int32_t> z, size_t size, size_t terms) {
54 if (terms > INT32_MAX / 65025 || !size || size > 64u * 65535u)
55 throw Failure("GPU integer kernel exceeds exact accumulator or dispatch limit", DiagnosticCode::Unsupported);
56 std::vector<uint8_t> bytes(z.size_bytes());
57 std::memcpy(bytes.data(), z.data(), bytes.size());
58 auto host = std::make_shared<const std::vector<uint8_t>>(std::move(bytes));
59 const std::vector<OnnxBuffer> inputs{a, b, {host, {}, host->size()}};
60 auto r = d.enqueue(source, inputs, size * 4, static_cast<uint32_t>(size));
61 if (!r.ok()) throw Failure(r.error()->message(), r.error()->code());
62 return std::move(r.value());
63}
64OnnxBuffer gpuMatmulResident(OnnxCompute& d, OnnxBuffer a, OnnxBuffer b, bool as, bool bs, size_t m, size_t k, size_t n,
65 int az, std::span<const int32_t> bz, size_t bOffset) {
66 if (bz.size() != 1 && bz.size() != n) throw Failure("GPU matmul zero-point extent mismatch");
67 std::ostringstream s;
68 s << prefix(as, bs) << "void main(){uint i=gl_GlobalInvocationID.x;if(i>=" << m * n << "u)return;"
69 << "uint r=i/" << n << "u,c=i%" << n << "u;int v=0;for(uint j=0;j<" << k << "u;++j)"
70 << "v+=(av(r*" << k << "u+j)-(" << az << "))*(bv(" << bOffset << "u+j*" << n << "u+c)-zp["
71 << (bz.size() == 1 ? "0" : "c") << "]);y[i]=v;}";
72 return enqueueInteger(d, s.str(), a, b, bz, m * n, k);
73}
74
76 int xz, std::span<const int32_t> wz) {
77 const int64_t oh =
78 (int64_t(c.height) + c.padTop + c.padBottom - int64_t(c.dilationH) * (c.kernelH - 1) - 1) / c.strideH + 1;
79 const int64_t ow =
80 (int64_t(c.width) + c.padLeft + c.padRight - int64_t(c.dilationW) * (c.kernelW - 1) - 1) / c.strideW + 1;
81 if ((oh - 1) * c.strideH > INT32_MAX || (ow - 1) * c.strideW > INT32_MAX ||
82 (oh - 1) * c.strideH - c.padTop + int64_t(c.kernelH - 1) * c.dilationH > INT32_MAX ||
83 (ow - 1) * c.strideW - c.padLeft + int64_t(c.kernelW - 1) * c.dilationW > INT32_MAX)
84 throw Failure("GPU convolution coordinate overflow", DiagnosticCode::Unsupported);
85 const size_t size = size_t(c.batch) * c.outputs * oh * ow;
86 const int cg = c.channels / c.groups;
87 std::ostringstream s;
88 s << prefix(xs, ws) << "void main(){uint i=gl_GlobalInvocationID.x;if(i>=" << size << "u)return;"
89 << "int xx=int(i%" << ow << "u),yy=int(i/" << ow << "u%" << oh << "u),o=int(i/" << ow * oh << "u%" << c.outputs
90 << "u),batch=int(i/" << ow * oh * c.outputs << "u);int v=0;"
91 << "for(int ch=0;ch<" << cg << ";++ch)for(int ky=0;ky<" << c.kernelH << ";++ky)for(int kx=0;kx<" << c.kernelW
92 << ";++kx){"
93 << "int iy=yy*" << c.strideH << "-" << c.padTop << "+ky*" << c.dilationH << ",ix=xx*" << c.strideW << "-"
94 << c.padLeft << "+kx*" << c.dilationW << ";"
95 << "if(iy<0||ix<0||iy>=" << c.height << "||ix>=" << c.width << ")continue;"
96 << "uint ai=uint(((batch*" << c.channels << "+o/" << c.outputs / c.groups << "*" << cg << "+ch)*" << c.height
97 << "+iy)*" << c.width << "+ix);"
98 << "uint bi=uint(((o*" << cg << "+ch)*" << c.kernelH << "+ky)*" << c.kernelW << "+kx);"
99 << "v+=(av(ai)-(" << xz << "))*(bv(bi)-zp[" << (wz.size() == 1 ? "0" : "o") << "]);}y[i]=v;}";
100 return enqueueInteger(d, s.str(), x, w, wz, size, size_t(cg) * c.kernelH * c.kernelW);
101}
102} // namespace eve::tensor::onnx_detail
LogicalId target
float w
Definition AnimClip.cpp:738
float x
Definition AnimClip.cpp:738
float z
Definition AnimClip.cpp:738
const std::string & s
int bz
Definition CaveMesh.cpp:114
int az
Definition CaveMesh.cpp:113
glm::vec4 p[6]
glm::vec3 n
Definition Grass.cpp:63
int inputs
Definition GridGraph.cpp:23
double r
std::int32_t c
std::uint64_t bytes
MeleePoint3 b
Definition MeleeHit.cpp:41
MeleePoint3 a
Definition MeleeHit.cpp:40
float d
float t
float size
Definition TreeMesh.cpp:156
UIHostHandle host
const UnitySourceAsset & source
float m[16]
float wz
GPU execution boundary for native ONNX; retains no model and retains compiled resources for the lifet...
Definition OnnxCompute.h:25
OnnxBuffer gpuConvResident(OnnxCompute &d, OnnxBuffer x, OnnxBuffer w, bool xs, bool ws, const affine::ConvShape &c, int xz, std::span< const int32_t > wz)
Gpu conv resident.
std::vector< int32_t > gpuConv(OnnxCompute &d, affine::ByteView x, affine::ByteView w, const affine::ConvShape &c, int xz, std::span< const int32_t > wz)
Gpu conv.
std::vector< int32_t > gpuMatmul(OnnxCompute &d, affine::ByteView a, affine::ByteView b, size_t m, size_t k, size_t n, int az, std::span< const int32_t > bz)
Gpu matmul.
OnnxBuffer gpuMatmulResident(OnnxCompute &d, OnnxBuffer a, OnnxBuffer b, bool as, bool bs, size_t m, size_t k, size_t n, int az, std::span< const int32_t > bz, size_t bOffset)
Gpu matmul resident.
OnnxBuffer enqueueInteger(OnnxCompute &d, const std::string &source, OnnxBuffer a, OnnxBuffer b, std::span< const int32_t > z, size_t size, size_t terms)
Immutable buffer shared by graph aliases; exactly one storage is authoritative.
Definition OnnxStorage.h:19
Borrowed packed 8-bit values, valid for the duration of a synchronous call.
Definition AffineQuant.h:24
Explicit 1D/2D NCHW convolution geometry; missing 1D height is one.
Definition AffineQuant.h:65