12 std::shared_ptr<std::vector<uint8_t>> writable_;
13 const std::vector<uint8_t>& get()
const {
15 if (!buffer_.
device)
throw std::runtime_error(
"Missing ONNX storage");
16 auto r = buffer_.
device->readback();
17 if (!
r.ok())
throw std::runtime_error(
r.error()->message());
18 if (
r.value().size() != buffer_.
size)
throw std::runtime_error(
"GPU byte count mismatch");
19 buffer_.
host = std::make_shared<const std::vector<uint8_t>>(std::move(
r.value()));
23 std::vector<uint8_t>& write() {
25 if (writable_ && buffer_.
host.use_count() == 2)
return *writable_;
26 auto v = std::make_shared<std::vector<uint8_t>>(get());
27 buffer_ = {
v, {},
v->
size()};
37 buffer_.size = buffer_.host->size();
46 size_t size()
const {
return buffer_.size; }
52 const uint8_t*
data()
const {
return get().data(); }
56 uint8_t*
data() {
return write().data(); }
62 auto begin() {
return write().begin(); }
64 auto end() {
return write().end(); }
66 auto begin()
const {
return get().begin(); }
68 auto end()
const {
return get().end(); }
82 operator std::span<const uint8_t>()
const {
return get(); }
83 operator std::vector<uint8_t>()
const {
return get(); }
const OnnxBuffer & buffer() const
Buffer.
void resize(size_t n)
Resize.
uint8_t & operator[](size_t i)
Operator [].
auto end() const
Ends end.
void assign(I a, I b)
Copy borrowed iterators into owning CPU storage. @lifetime Input iterators are borrowed only for this...
uint8_t * data()
Borrow mutable CPU data until mutation/destruction; detaches graph aliases before writing....
size_t size() const
Returns the size of size.
uint8_t operator[](size_t i) const
Operator [].
ByteStorage(OnnxBuffer b)
Constructs a ByteStorage.
auto begin()
Begins begin.
auto begin() const
Begins begin.
ByteStorage()
Constructs a ByteStorage.
ByteStorage(std::vector< uint8_t > v)
Constructs a ByteStorage.
void retainAcrossRuns()
Retain across runs.
const uint8_t * data() const
Borrow CPU data until this storage is mutated or destroyed; device-thread only for lazy data....
OnnxElement
ONNX wire element types; distinct from block-quantized Tensor storage.
Immutable buffer shared by graph aliases; exactly one storage is authoritative.
std::shared_ptr< OnnxDeviceStorage > device
std::shared_ptr< const std::vector< uint8_t > > host
RuntimeTensor public API.
std::vector< int64_t > shape