载入中...
搜索中...
未找到
AffineQuant.cpp
浏览该文件的文档.
24Result<std::vector<uint8_t>> quantize(std::span<const float> input, float scale, int zero, bool sign) {
54 Diagnostic::error(DiagnosticCode::InvalidArgument, "Dynamic quantization scale is not representable"));
62Result<std::vector<float>> dequantize(ByteView input, std::span<const float> scales, std::span<const int32_t> zeros,
80Result<std::vector<int32_t>> matmul(ByteView a, ByteView b, size_t m, size_t k, size_t n, int aZero, int bZero,
82 if (!validZero(aZero, a.signedValues) || !validZero(bZero, b.signedValues) || !product(m, k, SIZE_MAX) ||
83 !product(k, n, SIZE_MAX) || !product(m, n, INT32_MAX) || a.bytes.size() != m * k || b.bytes.size() != k * n)
85 Diagnostic::error(DiagnosticCode::InvalidArgument, "Invalid integer matmul dimensions or zero points"));
92 return Result<std::vector<int32_t>>::failure(Diagnostic::error(DiagnosticCode::Failed, e.what()));
109Result<detail::ConvExtent> detail::validateConv(size_t xSize, size_t wSize, bool xSigned, bool wSigned,
111 for (int d : {s.batch, s.channels, s.height, s.width, s.outputs, s.kernelH, s.kernelW, s.strideH, s.strideW,
116 if (s.channels % s.groups || s.outputs % s.groups || s.padTop < 0 || s.padBottom < 0 || s.padLeft < 0 ||
120 Diagnostic::error(DiagnosticCode::InvalidArgument, "Invalid convolution groups, padding or zero points"));
156 Diagnostic::error(DiagnosticCode::InvalidArgument, "Convolution buffer size mismatch or output too large"));
162 detail::validateConv(x.bytes.size(), w.bytes.size(), x.signedValues, w.signedValues, s, xZero, wZeros);
168 return Result<std::vector<int32_t>>::success(onnx_detail::gpuConv(*compute, x, w, s, xZero, wZeros));
170 return Result<std::vector<int32_t>>::failure(Diagnostic::error(DiagnosticCode::Failed, e.what()));
static Diagnostic error(DiagnosticCode code, std::string message, std::string path={}, DiagnosticDetails details={}, std::string source={})
Construct an error diagnostic with the standard error severity.
Definition Diagnostic.h:125
static Result failure(Status status)
Construct a failed result from a structured status.
Definition Result.h:175
GPU execution boundary for native ONNX; retains no model and retains compiled resources for the lifet...
Definition OnnxCompute.h:25
Result< ConvExtent > validateConv(size_t xSize, size_t wSize, bool xSigned, bool wSigned, const ConvShape &, int, std::span< const int32_t >)
Validate conv.
Definition AffineQuant.cpp:109
Definition AffineQuant.cpp:9
Result< std::vector< int32_t > > conv(ByteView x, ByteView w, const ConvShape &s, int xZero, std::span< const int32_t > wZeros, OnnxCompute *compute)
Integer Conv, supporting groups, asymmetric padding and dilation.
Definition AffineQuant.cpp:159
Result< std::vector< uint8_t > > quantize(std::span< const float > input, float scale, int zero, bool sign)
Affine quantization to int8/uint8 bytes, saturating and rounding ties to even.
Definition AffineQuant.cpp:24
Result< std::vector< float > > dequantize(ByteView input, std::span< const float > scales, std::span< const int32_t > zeros, size_t inner)
Affine dequantization using scalar or per-axis scale/zero point.
Definition AffineQuant.cpp:62
Result< std::vector< int32_t > > matmul(ByteView a, ByteView b, size_t m, size_t k, size_t n, int aZero, int bZero, OnnxCompute *compute)
Integer row-major [M,K] x [K,N], subtracting scalar zero points.
Definition AffineQuant.cpp:80
Result< QuantizedActivation > dynamicQuantize(std::span< const float > input)
Quantize finite FP32 activations with ONNX DynamicQuantizeLinear semantics.
Definition AffineQuant.cpp:41
std::vector< int32_t > gpuConv(OnnxCompute &d, affine::ByteView x, affine::ByteView w, const affine::ConvShape &c, int xz, std::span< const int32_t > wz)
Gpu conv.
Definition OnnxGpuKernels.cpp:46
std::vector< int32_t > gpuMatmul(OnnxCompute &d, affine::ByteView a, affine::ByteView b, size_t m, size_t k, size_t n, int az, std::span< const int32_t > bz)
Gpu matmul.
Definition OnnxGpuKernels.cpp:40
@ InvalidArgument
@ Failed
Borrowed packed 8-bit values, valid for the duration of a synchronous call.
Definition AffineQuant.h:24
Explicit 1D/2D NCHW convolution geometry; missing 1D height is one.
Definition AffineQuant.h:65
ONNX unsigned activation quantization; owns bytes and scalar parameters.
Definition AffineQuant.h:17
std::vector< uint8_t > values
Definition AffineQuant.h:18