载入中...
搜索中...
未找到
OnnxQuant.cpp
浏览该文件的文档.
28std::vector<RuntimeTensor> executeQuant(const Node& n, const std::vector<const RuntimeTensor*>& in,
45 auto zeroPoint = make(OnnxElement::UInt8, {}, std::vector<uint8_t>{static_cast<uint8_t>(values[1])});
49 "x[];};layout(std430,binding=1)readonly buffer P{float p[];};layout(std430,binding=2)writeonly buffer "
52 "u)return;uint packed=0u;for(uint c=0u;c<4u;++c){uint j=word*4u+c;if(j<" + std::to_string(total) +
54 "q=isnan(v)?0u:uint(clamp(roundEven(v/p[0])+p[1],0.0,255.0));packed|=q<<(c*8u);}}y[word]=packed;}";
56 auto q = checked(compute->enqueue(source, inputs, total, static_cast<uint32_t>((total + 3) / 4)));
76 if (required(in, 1).shape.size() > 1) throw Failure("Quantization scale must be scalar or vector");
79 const auto type = n.op == "DequantizeLinear" ? x.element : (zp ? zp->element : OnnxElement::UInt8);
90 if (static_cast<size_t>(x.shape[a]) != scales.size()) throw Failure("Scale channel count mismatch");
94 return {make(OnnxElement::Float32, x.shape, checked(affine::dequantize(bytes(x), scales, zeros, inner)))};
99 auto q = checked(affine::quantize(std::span(values).subspan(i, inner), scales[channel], zeros[channel],
110 if (k != static_cast<size_t>(b.shape[0]) || k == 0) throw Failure("Invalid MatMulInteger inner dimension");
125 checked(affine::matmul(bytes(x), bytes(b), rows, k, columns, zero(in, 2, x), zero(in, 3, b), compute)))};
133 n.attrs.at("kernel_shape").integers != std::vector<int64_t>(w.shape.begin() + 2, w.shape.end()))
176 checked(affine::detail::validateConv(x.bytes.size(), w.bytes.size(), x.element == OnnxElement::Int8,
189 (s.height + s.padTop + s.padBottom - int64_t(s.dilationH) * (s.kernelH - 1) - 1) / s.strideH + 1);
GPU execution boundary for native ONNX; retains no model and retains compiled resources for the lifet...
Definition OnnxCompute.h:25
@ Failure
Result< ConvExtent > validateConv(size_t xSize, size_t wSize, bool xSigned, bool wSigned, const ConvShape &, int, std::span< const int32_t >)
Validate conv.
Definition AffineQuant.cpp:109
Result< std::vector< int32_t > > conv(ByteView x, ByteView w, const ConvShape &s, int xZero, std::span< const int32_t > wZeros, OnnxCompute *compute)
Integer Conv, supporting groups, asymmetric padding and dilation.
Definition AffineQuant.cpp:159
Result< std::vector< uint8_t > > quantize(std::span< const float > input, float scale, int zero, bool sign)
Affine quantization to int8/uint8 bytes, saturating and rounding ties to even.
Definition AffineQuant.cpp:24
Result< std::vector< float > > dequantize(ByteView input, std::span< const float > scales, std::span< const int32_t > zeros, size_t inner)
Affine dequantization using scalar or per-axis scale/zero point.
Definition AffineQuant.cpp:62
Result< std::vector< int32_t > > matmul(ByteView a, ByteView b, size_t m, size_t k, size_t n, int aZero, int bZero, OnnxCompute *compute)
Integer row-major [M,K] x [K,N], subtracting scalar zero points.
Definition AffineQuant.cpp:80
Result< QuantizedActivation > dynamicQuantize(std::span< const float > input)
Quantize finite FP32 activations with ONNX DynamicQuantizeLinear semantics.
Definition AffineQuant.cpp:41
Definition OnnxByteStorage.h:6
OnnxBuffer gpuConvResident(OnnxCompute &d, OnnxBuffer x, OnnxBuffer w, bool xs, bool ws, const affine::ConvShape &c, int xz, std::span< const int32_t > wz)
Gpu conv resident.
Definition OnnxGpuKernels.cpp:75
std::vector< int64_t > attrs(const Node &n, const char *key, std::vector< int64_t > fallback)
Attrs.
Definition OnnxInternal.h:132
std::vector< RuntimeTensor > executeQuant(const Node &node, const std::vector< const RuntimeTensor * > &inputs, OnnxCompute *compute=nullptr)
Execute quant.
Definition OnnxQuant.cpp:28
RuntimeTensor make(OnnxElement e, std::vector< int64_t > shape, const std::vector< T > &values)
Make.
Definition OnnxInternal.h:91
RuntimeTensor dispatchFloat(OnnxCompute &, const std::vector< const RuntimeTensor * > &, const std::vector< int64_t > &, const std::string &, size_t work=0)
Dispatches float.
Definition OnnxNumeric.cpp:33
OnnxBuffer gpuMatmulResident(OnnxCompute &d, OnnxBuffer a, OnnxBuffer b, bool as, bool bs, size_t m, size_t k, size_t n, int az, std::span< const int32_t > bz, size_t bOffset)
Gpu matmul resident.
Definition OnnxGpuKernels.cpp:64
@ Float32
@ Unsupported
Explicit 1D/2D NCHW convolution geometry; missing 1D height is one.
Definition AffineQuant.h:65
std::vector< int64_t > shape
Definition OnnxByteStorage.h:88