20 auto z = (state_ += 0x9e3779b97f4a7c15ULL);
21 z = (
z ^ (
z >> 30)) * 0xbf58476d1ce4e5b9ULL;
22 z = (
z ^ (
z >> 27)) * 0x94d049bb133111ebULL;
24 return double((
z ^ (
z >> 31)) >> 11) * 0x1.0p-53;
35 return hidden * (
inputs + 1) + hidden * (hidden + 1) +
actions * (hidden + 1);
39inline std::vector<double>
layer(
const std::vector<double>&
x,
const Policy&
p, std::size_t
offset, std::size_t outputs,
42 std::vector<double>
y(outputs);
43 for (std::size_t j = 0; j < outputs; ++j) {
45 double sum =
p.weights[
start +
x.size()];
46 for (std::size_t i = 0; i <
x.size(); ++i) sum +=
x[i] *
p.weights[
start + i];
47 y[j] = activate ? std::tanh(sum) : sum;
54 double maximum = -std::numeric_limits<double>::infinity();
57 std::vector<double> probabilities(logits.size());
68 auto h1 =
layer(
x,
p, 0,
p.hiddenWidth,
true);
69 const auto second =
p.hiddenWidth * (
p.featureCount + 1);
71 const auto third =
second +
p.hiddenWidth * (
p.hiddenWidth + 1);
80 p.actionCount =
c.actionCount;
81 p.hiddenWidth =
c.hiddenWidth;
82 p.weights.resize(
weightCount(
c.featureCount,
c.hiddenWidth,
c.actionCount));
85 for (
auto&
w :
p.weights)
w = (
random.unit() * 2 - 1) / std::sqrt(
double(
c.hiddenWidth));
93 auto h1 =
layer(
x,
p, 0,
p.hiddenWidth,
true);
94 const auto second =
p.hiddenWidth * (
p.featureCount + 1);
96 const auto third =
second +
p.hiddenWidth * (
p.hiddenWidth + 1);
100 std::vector<double> d2(
p.hiddenWidth), d1(
p.hiddenWidth);
101 for (std::size_t i = 0; i <
p.hiddenWidth; ++i) {
102 for (std::size_t j = 0; j <
p.actionCount; ++j) d2[i] += d3[j] *
p.weights[
third + j * (
p.hiddenWidth + 1) + i];
103 d2[i] *= 1 - h2[i] * h2[i];
105 for (std::size_t i = 0; i <
p.hiddenWidth; ++i) {
106 for (std::size_t j = 0; j <
p.hiddenWidth; ++j)
107 d1[i] += d2[j] *
p.weights[
second + j * (
p.hiddenWidth + 1) + i];
108 d1[i] *= 1 - h1[i] * h1[i];
110 auto update = [&](std::size_t
offset,
const auto&
inputs,
const auto& gradient) {
111 for (std::size_t j = 0; j < gradient.size(); ++j) {
113 for (std::size_t i = 0; i <
inputs.size(); ++i)
114 p.weights[
start + i] -=
rate * std::clamp(gradient[j] *
inputs[i], -1.0, 1.0);
115 p.weights[
start +
inputs.size()] -=
rate * std::clamp(gradient[j], -1.0, 1.0);
119 update(
third, h2, d3);
std::vector< ActionSpec > actions
Random(std::uint64_t seed)
Constructs a Random.
std::size_t index(std::size_t count)
Index.
Policy makePolicy(const Config &c)
Make policy.
void train(Policy &p, const Observation &o, std::uint32_t action, double rate)
Train.
std::vector< double > forward(const Policy &p, const Observation &o)
Forward.
std::vector< double > softmax(std::vector< double > logits, const Observation &o)
Softmax.
std::size_t weightCount(std::size_t inputs, std::size_t hidden, std::size_t actions)
Weight count.
Bounded search configuration; seed streams for environment, search and learning are separate.
Owning state projection; action IDs index a fixed domain action catalogue.
std::vector< std::uint32_t > legalActions
std::vector< float > features
Owning version-1 portable network weights; import validates the entire value before use.
std::uint32_t featureCount