载入中...
搜索中...
未找到
AffineQuant.h
浏览该文件的文档.
1#pragma once
2#include "common/Export.h"
3
4
5#include "common/Result.h"
6
7#include <cstdint>
8#include <span>
9#include <vector>
10
11namespace eve::tensor {
12class OnnxCompute;
13}
14namespace eve::tensor::affine {
15
18 std::vector<uint8_t> values;
19 float scale = 1;
20 uint8_t zeroPoint = 0;
21};
22
24struct ByteView {
25 std::span<const uint8_t> bytes;
26 bool signedValues = false;
27};
28
35[[nodiscard]] EVENGINE_API_DOMAINS Result<QuantizedActivation> dynamicQuantize(std::span<const float> input);
36
42[[nodiscard]] EVENGINE_API_DOMAINS Result<std::vector<uint8_t>> quantize(std::span<const float> input, float scale,
43 int zeroPoint, bool signedValues);
44
51[[nodiscard]] Result<std::vector<float>> dequantize(ByteView input, std::span<const float> scales,
52 std::span<const int32_t> zeros, size_t inner = 1);
53
61 size_t n, int aZero = 0, int bZero = 0,
62 OnnxCompute* compute = nullptr);
63
65struct ConvShape {
66 int batch = 1, channels = 1, height = 1, width = 1;
67 int outputs = 1, kernelH = 1, kernelW = 1;
68 int strideH = 1, strideW = 1, dilationH = 1, dilationW = 1;
69 int padTop = 0, padLeft = 0, padBottom = 0, padRight = 0, groups = 1;
70};
71
79 int xZero, std::span<const int32_t> wZeros,
80 OnnxCompute* compute = nullptr);
81} // namespace eve::tensor::affine
float w
Definition AnimClip.cpp:738
float x
Definition AnimClip.cpp:738
EvpackChunkInput input
Definition Evpack.cpp:170
#define EVENGINE_API_DOMAINS
Definition Export.h:110
ShaderImageInput shape
glm::vec3 n
Definition Grass.cpp:63
std::array< float, 3 > scale
MeleePoint3 b
Definition MeleeHit.cpp:41
MeleePoint3 a
Definition MeleeHit.cpp:40
std::vector< float > scales
Definition OnnxLstm.cpp:27
std::vector< int64_t > zeros
Definition OnnxLstm.cpp:28
OnnxCompute * compute
Move-only, checked operation results for the common layer.
float m[16]
Move-only operation result carrying either a value or Status.
Definition Result.h:155
GPU execution boundary for native ONNX; retains no model and retains compiled resources for the lifet...
Definition OnnxCompute.h:25
Result< std::vector< int32_t > > conv(ByteView x, ByteView w, const ConvShape &s, int xZero, std::span< const int32_t > wZeros, OnnxCompute *compute)
Integer Conv, supporting groups, asymmetric padding and dilation.
Result< std::vector< uint8_t > > quantize(std::span< const float > input, float scale, int zero, bool sign)
Affine quantization to int8/uint8 bytes, saturating and rounding ties to even.
Result< std::vector< float > > dequantize(ByteView input, std::span< const float > scales, std::span< const int32_t > zeros, size_t inner)
Affine dequantization using scalar or per-axis scale/zero point.
Result< std::vector< int32_t > > matmul(ByteView a, ByteView b, size_t m, size_t k, size_t n, int aZero, int bZero, OnnxCompute *compute)
Integer row-major [M,K] x [K,N], subtracting scalar zero points.
Result< QuantizedActivation > dynamicQuantize(std::span< const float > input)
Quantize finite FP32 activations with ONNX DynamicQuantizeLinear semantics.
Borrowed packed 8-bit values, valid for the duration of a synchronous call.
Definition AffineQuant.h:24
std::span< const uint8_t > bytes
Definition AffineQuant.h:25
Explicit 1D/2D NCHW convolution geometry; missing 1D height is one.
Definition AffineQuant.h:65
ONNX unsigned activation quantization; owns bytes and scalar parameters.
Definition AffineQuant.h:17