载入中...
搜索中...
未找到
OnnxGpuKernels.cpp
浏览该文件的文档.
27 Result<std::vector<uint8_t>> dispatch(const OnnxKernel& k) override { return target.dispatch(k); }
34 if (!result.host || result.host->size() % 4) throw Failure("Invalid synchronous integer output");
40std::vector<int32_t> gpuMatmul(OnnxCompute& d, affine::ByteView a, affine::ByteView b, size_t m, size_t k, size_t n,
43 return integerOutput(gpuMatmulResident(sync, hostBuffer(a.bytes), hostBuffer(b.bytes), a.signedValues,
46std::vector<int32_t> gpuConv(OnnxCompute& d, affine::ByteView x, affine::ByteView w, const affine::ConvShape& c, int xz,
55 throw Failure("GPU integer kernel exceeds exact accumulator or dispatch limit", DiagnosticCode::Unsupported);
64OnnxBuffer gpuMatmulResident(OnnxCompute& d, OnnxBuffer a, OnnxBuffer b, bool as, bool bs, size_t m, size_t k, size_t n,
75OnnxBuffer gpuConvResident(OnnxCompute& d, OnnxBuffer x, OnnxBuffer w, bool xs, bool ws, const affine::ConvShape& c,
78 (int64_t(c.height) + c.padTop + c.padBottom - int64_t(c.dilationH) * (c.kernelH - 1) - 1) / c.strideH + 1;
80 (int64_t(c.width) + c.padLeft + c.padRight - int64_t(c.dilationW) * (c.kernelW - 1) - 1) / c.strideW + 1;
88 s << prefix(xs, ws) << "void main(){uint i=gl_GlobalInvocationID.x;if(i>=" << size << "u)return;"
89 << "int xx=int(i%" << ow << "u),yy=int(i/" << ow << "u%" << oh << "u),o=int(i/" << ow * oh << "u%" << c.outputs
91 << "for(int ch=0;ch<" << cg << ";++ch)for(int ky=0;ky<" << c.kernelH << ";++ky)for(int kx=0;kx<" << c.kernelW
93 << "int iy=yy*" << c.strideH << "-" << c.padTop << "+ky*" << c.dilationH << ",ix=xx*" << c.strideW << "-"
GPU execution boundary for native ONNX; retains no model and retains compiled resources for the lifet...
Definition OnnxCompute.h:25
@ Failure
Definition OnnxByteStorage.h:6
OnnxBuffer gpuConvResident(OnnxCompute &d, OnnxBuffer x, OnnxBuffer w, bool xs, bool ws, const affine::ConvShape &c, int xz, std::span< const int32_t > wz)
Gpu conv resident.
Definition OnnxGpuKernels.cpp:75
std::vector< int32_t > gpuConv(OnnxCompute &d, affine::ByteView x, affine::ByteView w, const affine::ConvShape &c, int xz, std::span< const int32_t > wz)
Gpu conv.
Definition OnnxGpuKernels.cpp:46
std::vector< int32_t > gpuMatmul(OnnxCompute &d, affine::ByteView a, affine::ByteView b, size_t m, size_t k, size_t n, int az, std::span< const int32_t > bz)
Gpu matmul.
Definition OnnxGpuKernels.cpp:40
OnnxBuffer gpuMatmulResident(OnnxCompute &d, OnnxBuffer a, OnnxBuffer b, bool as, bool bs, size_t m, size_t k, size_t n, int az, std::span< const int32_t > bz, size_t bOffset)
Gpu matmul resident.
Definition OnnxGpuKernels.cpp:64
OnnxBuffer enqueueInteger(OnnxCompute &d, const std::string &source, OnnxBuffer a, OnnxBuffer b, std::span< const int32_t > z, size_t size, size_t terms)
Definition OnnxGpuKernels.cpp:52
@ Unsupported
Immutable buffer shared by graph aliases; exactly one storage is authoritative.
Definition OnnxStorage.h:19
Borrowed packed 8-bit values, valid for the duration of a synchronous call.
Definition AffineQuant.h:24
Explicit 1D/2D NCHW convolution geometry; missing 1D height is one.
Definition AffineQuant.h:65