13 std::shared_ptr<const RuntimeTensor>
owned;
16 const RuntimeTensor&
tensor()
const {
19 throw Failure(
"Expected tensor, received sequence");
22using Values = std::unordered_map<std::string, Value>;
30Value hold(RuntimeTensor
t, Context&
c) {
33 for (
size_t i = 0; i <
count(
t.shape); ++i)
34 if (!std::isfinite(read<float>(
t, i)))
throw Failure(
"Nonfinite float output");
35 const auto size =
t.bytes.size();
36 if (
size > 512u * 1024u * 1024u -
c.liveBytes)
throw Failure(
"ONNX live tensor memory limit exceeded");
38 auto* raw =
new RuntimeTensor(std::move(
t));
41 std::shared_ptr<const RuntimeTensor>(raw,
42 [&
c,
size](
const RuntimeTensor*
p) {
48std::set<std::string> captures(
const ModelData&
graph);
49std::set<std::string> dependencies(
const Node&
n) {
50 std::set<std::string> result;
52 if (!i.empty()) result.insert(i);
55 auto names = captures(*
a.graph);
56 result.insert(names.begin(), names.end());
60std::set<std::string> captures(
const ModelData&
graph) {
61 std::set<std::string> locals, result;
65 for (const auto& out :
n.outputs)
66 if (!out.empty()) locals.insert(out);
68 auto deps = dependencies(
n);
69 for (
const auto&
d : deps)
70 if (!locals.contains(
d)) result.insert(
d);
72 for (
const auto& out :
graph.
info.outputs)
73 if (!locals.contains(out)) result.insert(out);
76void admit(
const Node&
n) {
84std::vector<size_t> plan(
const ModelData&
g,
const std::vector<std::string>& requested) {
85 std::unordered_map<std::string, size_t> producers;
86 for (
size_t i = 0; i <
g.nodes.size(); ++i)
89 std::vector<bool> selected(
g.nodes.size(),
false);
90 std::vector<std::string>
pending = requested;
94 auto p = producers.find(
name);
95 if (
p == producers.end() || selected[
p->second])
continue;
96 selected[
p->second] =
true;
97 admit(
g.nodes[
p->second]);
98 auto deps = dependencies(
g.nodes[
p->second]);
101 std::vector<size_t> indegree(
g.nodes.size());
102 std::vector<std::vector<size_t>> users(
g.nodes.size());
104 std::priority_queue<size_t, std::vector<size_t>, std::greater<size_t>> ready;
105 for (
size_t i = 0; i <
g.nodes.size(); ++i)
108 std::set<size_t> parents;
109 for (
const auto&
name : dependencies(
g.
nodes[i])) {
110 auto p = producers.find(
name);
111 if (
p != producers.end()) parents.insert(
p->second);
113 indegree[i] = parents.size();
114 for (
auto p : parents) users[
p].push_back(i);
115 if (parents.empty()) ready.push(i);
117 std::vector<size_t>
order;
118 while (!ready.empty()) {
119 size_t i = ready.top();
122 for (
auto user : users[i])
123 if (--indegree[user] == 0) ready.push(user);
128std::vector<Value>
run(
const ModelData&, Values,
const std::vector<std::string>&, Context&,
size_t);
129std::vector<Value> control(
const Node&
n,
const std::vector<Value>& in,
const Values&
values, Context&
c,
131 auto scalar = [&](
size_t i) {
132 const auto&
t = in.at(i).tensor();
133 if (
count(
t.shape) != 1)
throw Failure(
"Control input must contain one element");
136 auto seq = [&](
size_t i) ->
const std::vector<Value>& {
137 if (!in.at(i).sequence)
throw Failure(
"Expected sequence input");
138 return *in[i].sequence;
141 return Value{
nullptr, {}, std::make_shared<std::vector<Value>>(std::move(
v)),
type};
143 if (
n.op ==
"SequenceEmpty") {
148 if (
n.op ==
"SequenceAt") {
149 const auto&
s = seq(0);
150 int64_t i = scalar(1);
151 if (i < 0) i +=
static_cast<int64_t
>(
s.size());
152 if (i < 0 ||
static_cast<size_t>(i) >=
s.size())
throw Failure(
"SequenceAt out of bounds");
155 if (
n.op ==
"SequenceInsert") {
157 if (
s.size() >= 100000)
throw Failure(
"Sequence length limit exceeded");
158 int64_t i = in.size() > 2 ? scalar(2) : static_cast<int64_t>(
s.
size());
159 if (i < 0) i +=
static_cast<int64_t
>(
s.size());
160 if (i < 0 ||
static_cast<size_t>(i) >
s.size())
throw Failure(
"SequenceInsert out of bounds");
161 const auto&
t = in.at(1).tensor();
163 s.insert(
s.begin() + i, in[1]);
166 if (
n.op ==
"SplitToSequence") {
167 const auto&
x = in.at(0).tensor();
169 std::vector<int64_t> lengths;
170 if (in.size() > 1 && (in[1].borrowed || in[1].owned)) {
171 const auto&
split = in[1].tensor();
173 if (
split.shape.empty()) {
174 if (lengths[0] <= 0)
throw Failure(
"Invalid split size");
175 int64_t
width = lengths[0];
177 for (int64_t i = 0; i <
x.shape[
a]; i +=
width) {
178 if (lengths.size() >= 100000)
throw Failure(
"Sequence length limit exceeded");
179 lengths.push_back(std::min(
width,
x.shape[
a] - i));
183 lengths.assign(
x.shape[
a], 1);
184 if (lengths.size() > 100000)
throw Failure(
"Sequence length limit exceeded");
186 for (
auto size : lengths) {
187 if (size < 0 || size >
x.shape[
a] - total)
throw Failure(
"Invalid split lengths");
190 if (total !=
x.shape[
a])
throw Failure(
"Split lengths do not cover axis");
192 for (
int j = 0; j <
a; ++j) outer *=
x.shape[j];
193 for (
size_t j =
a + 1; j <
x.shape.size(); ++j) inner *=
x.shape[j];
194 std::vector<Value>
s;
196 for (
auto length : lengths) {
199 if (!
attr(
n,
"keepdims", 1)) {
200 if (in.size() > 1 && (in[1].borrowed || in[1].owned))
205 for (
size_t o = 0; o < outer; ++o)
207 std::memcpy(out.bytes.data() + o *
length * inner,
209 s.push_back(hold(std::move(out),
c));
214 if (
n.op ==
"ConcatFromSequence") {
215 const auto&
s = seq(0);
216 if (
s.empty())
throw Failure(
"Cannot concatenate empty sequence");
217 std::vector<RuntimeTensor> tensors;
218 std::vector<const RuntimeTensor*>
inputs;
219 for (
const auto&
v :
s) {
220 tensors.push_back(
v.tensor());
221 if (
attr(
n,
"new_axis", 0)) {
222 const int a =
axis(
attr(
n,
"axis", 0), tensors.back().shape.size() + 1);
223 tensors.back().shape.insert(tensors.back().shape.begin() +
a, 1);
226 for (
const auto&
t : tensors)
inputs.push_back(&
t);
230 if (!out)
throw Failure(
"ConcatFromSequence dispatch failed");
231 return {hold(std::move(*out),
c)};
235 auto key = scalar(0) ?
"then_branch" :
"else_branch";
236 const auto&
g =
n.attrs.at(
key).graph;
237 if (!
g)
throw Failure(
"Missing If branch");
240 if (
n.op ==
"Loop") {
241 const auto&
body =
n.attrs.at(
"body").graph;
243 if (in.size() < 2 ||
body->info.inputs.size() != in.size() ||
body->info.outputs.size() != in.size() - 1 ||
244 n.outputs.size() != in.size() - 2)
246 const int64_t trips = (in[0].borrowed || in[0].owned) ? scalar(0) : 10000;
247 if (trips < 0 || trips > 10000)
throw Failure(
"Loop trip budget exceeded");
248 bool condition = (in[1].borrowed || in[1].owned) ? scalar(1) != 0 :
true;
249 std::vector<Value>
state(in.begin() + 2, in.end());
250 for (int64_t i = 0; i < trips &&
condition; ++i) {
255 for (
size_t j = 0; j <
state.size(); ++j)
scope[
body->info.inputs[j + 2]] = state[j];
257 const auto& cond = result[0].tensor();
259 throw Failure(
"Loop returned invalid condition");
261 state.assign(result.begin() + 1, result.end());
265 if (
n.op ==
"RandomUniformLike" ||
n.op ==
"RandomNormalLike") {
266 const auto&
x = in.at(0).tensor();
267 if (
attr(
n,
"dtype",
static_cast<int>(
x.element)) != 1)
269 uint64_t
seed =
c.seed;
270 for (
unsigned char ch :
n.
name)
seed = (
seed ^ ch) * 1099511628211ull;
271 if (
n.attrs.contains(
"seed")) {
273 std::memcpy(&bits, &
n.attrs.at(
"seed").real, 4);
276 auto [stream, inserted] =
c.randomStreams.try_emplace(&
n,
static_cast<uint32_t
>(
seed ^ (
seed >> 32)));
277 auto& rng = stream->second;
278 auto uniform = [&]() {
return (
double(rng()) + .5) / 4294967296.; };
279 std::vector<float>
v(
count(
x.shape));
280 const float low =
n.attrs.contains(
"low") ?
n.attrs.at(
"low").real : 0,
281 high =
n.attrs.contains(
"high") ?
n.attrs.at(
"high").real : 1,
282 mean =
n.attrs.contains(
"mean") ?
n.attrs.at(
"mean").real : 0,
283 scale =
n.attrs.contains(
"scale") ?
n.attrs.at(
"scale").real : 1;
284 if (!std::isfinite(low) || !std::isfinite(high) || !std::isfinite(mean) || !std::isfinite(
scale) ||
285 high < low ||
scale < 0)
286 throw Failure(
"Invalid random distribution parameters");
288 f =
n.op ==
"RandomUniformLike" ? static_cast<float>(low + (high - low) * uniform())
289 : static_cast<float>(mean +
scale *
std::sqrt(-2 *
std::log(uniform())) *
290 std::cos(6.283185307179586 * uniform()));
293 std::vector<const RuntimeTensor*> tensors;
295 auto outputs =
execute(
n, tensors,
c.compute);
296 std::vector<Value> result;
297 for (
auto&
t : outputs) result.push_back(hold(
std::move(
t),
c));
300std::vector<Value>
run(
const ModelData&
g, Values
values,
const std::vector<std::string>& requested, Context&
c,
302 if (
depth > 16)
throw Failure(
"Execution graph nesting limit exceeded");
303 for (
const auto& [
name,
t] :
g.constants)
305 auto order = plan(
g, requested);
306 std::unordered_map<std::string, size_t> uses;
308 for (const auto& dep : dependencies(
g.
nodes[i])) ++uses[dep];
309 for (
const auto&
name : requested) ++uses[
name];
310 for (
auto i :
order) {
311 const auto&
n =
g.nodes[i];
312 if (++
c.steps > 1000000)
throw Failure(
"ONNX execution step budget exceeded");
314 std::vector<Value> in;
321 if (it ==
values.end())
throw Failure(
"Missing feed or lexical capture: " +
name);
322 in.push_back(it->second);
325 if (outputs.size() <
n.outputs.size())
throw Failure(
"Operator output arity mismatch");
326 for (
size_t j = 0; j <
n.outputs.size(); ++j)
327 if (!
n.outputs[j].empty() && uses[
n.outputs[j]])
values[
n.outputs[j]] = std::move(outputs[j]);
328 for (
const auto& dep : dependencies(
n))
329 if (--uses[dep] == 0)
values.erase(dep);
330 }
catch (
const Failure& e) {
331 throw Failure(
n.name +
": " + e.what(), e.code, e.path.empty() ?
n.name : e.
path);
334 std::vector<Value> out;
335 for (
const auto&
name : requested) {
338 out.push_back(it->second);
348 for (
const auto&
f : feeds)
inputs[
f.name] = hold({
f.tensor.element,
f.tensor.shape,
f.tensor.bytes},
c);
350 std::vector<OnnxNamedTensor> result;
351 size_t bytes =
c.liveBytes;
352 for (
size_t i = 0; i < outputs.size(); ++i) {
353 const auto&
t = outputs[i].tensor();
354 if (
t.bytes.size() > 512u * 1024u * 1024u -
bytes)
throw Failure(
"ONNX output memory limit exceeded");
356 result.push_back({requested[i], {
t.element,
t.shape,
static_cast<std::vector<uint8_t>
>(
t.bytes)}});
std::vector< QuestEvent > pending
std::map< std::string, Var > values
std::array< float, 3 > scale
std::unique_ptr< gpgpu::Sequence > sequence
const RuntimeTensor & tensor
std::unordered_map< const Node *, std::mt19937 > randomStreams
OnnxElement sequenceElement
const RuntimeTensor * borrowed
std::map< std::string, std::vector< std::string > > graph
const SquirrelValueOptions & options
GPU execution boundary for native ONNX; retains no model and retains compiled resources for the lifet...
std::variant< std::monostate, std::int64_t, double, std::string, bool > Value
void concat(const float *const *ins, const int *const *inDims, const int *inRanks, int n, int axis, float *out, const int *outDims, int outRank)
Concat.
int64_t integer(const RuntimeTensor &v, size_t i=0)
Integer.
std::vector< int64_t > ints(const RuntimeTensor &v)
Ints.
int64_t attr(const Node &n, const char *key, int64_t fallback)
Attr.
std::vector< int64_t > attrs(const Node &n, const char *key, std::vector< int64_t > fallback)
Attrs.
RuntimeTensor make(OnnxElement e, std::vector< int64_t > shape, const std::vector< T > &values)
Make.
int axis(int64_t a, size_t rank)
Axis.
std::vector< OnnxNamedTensor > evaluate(const ModelData &, std::span< const OnnxNamedTensor >, const std::vector< std::string > &, OnnxCompute *, OnnxRunOptions options)
Evaluate.
bool isSupported(const Node &n)
True when supported.
void validate(const RuntimeTensor &v)
Validate.
size_t elementSize(OnnxElement e)
Element size.
std::optional< RuntimeTensor > executeShape(const Node &, const std::vector< const RuntimeTensor * > &)
Execute shape.
std::vector< RuntimeTensor > execute(const Node &n, const std::vector< const RuntimeTensor * > &in, OnnxCompute *compute)
Execute.
OnnxElement
ONNX wire element types; distinct from block-quantized Tensor storage.
WidgetDesc child(std::string id, std::vector< WidgetDesc > children, float width, float height)
Scrollable child region with an explicit size.
Per-call deterministic RNG and optional strict finite-output diagnostic.