Stream 2 of the clean-slate rewrite: nisps/ml/ replaces src/memlp/ with a header-only, heap-free MLP that satisfies nisps::core::MLEngine. Files (nisps/ml/): - activations.hpp — ReLU (leaky 0.01 for parity), sigmoid, tanh - loss.hpp — MSE per-sample (fixes meml-ues double-scaling: returns the sample's MSE without an extra 1/N multiplication; the training loop averages explicitly) - init.hpp — uniform/Xavier/spread-aware weight init - training.hpp — gradient clip helper (±10.0 matches legacy) - rl.hpp — move_weights with per-layer Xavier scaling, weight decay (10% * spread), gaussian noise via the deterministic Rng (matches the legacy JS sum-of-three-uniforms shape); draw_weights also spread-aware - stats.hpp — per-layer mean/max/dead/saturating diagnostics - mlp.hpp — 4-layer (3 hidden + sigmoid output) MLP class with std::array-backed weights, biases, gradient accumulators, dataset ring buffer (default 128 examples), loss history (default 4096 iters). Bias is a separate per-layer parameter — no input-vector mutation. Flat get_weights/set_weights layout: weights all layers (row-major, layer order), then biases all layers. Tests (tests/cpp/, all 50 passing under -Wall -Wextra -Werror -Wpedantic): - test_mlp_init.cpp — deterministic seeding, spread regimes, static_assert MLEngine concept satisfied - test_mlp_inference.cpp — golden hand-computed forward pass match, sigmoid output range, set_input bounds - test_mlp_training.cpp — XOR convergence (loss < 0.01 in <2k iters), ring-buffer eviction - test_mlp_loss.cpp — meml-ues regression test: reported loss equals hand-computed average MSE without extra 1/N scaling; sample weights honoured - test_mlp_rl.cpp — move_weights respects output_pin_mask (final-layer rows + biases preserved); spread regimes; grad clear after draw_weights - test_mlp_serialize.cpp — get_weights/set_weights round-trip preserves inference exactly; eval_loss is non-mutating; infer_batch matches individual inference Verification: - Clean build, no warnings - 50 tests pass (22 prior + 28 new) - No std::vector / new / malloc in nisps/ml/ - All float literals .f-suffixed in code (comments excepted)
92 lines
2.9 KiB
C++
92 lines
2.9 KiB
C++
// tests/cpp/test_mlp_training.cpp — convergence test on XOR.
|
|
//
|
|
// XOR is the classic minimum non-linear problem: a 2-layer linear net
|
|
// cannot solve it; an MLP with one hidden layer (and a non-linear
|
|
// activation) can. Our 4-layer MLP with sigmoid output is more than enough.
|
|
//
|
|
// We check loss < 0.01 within a generous iteration budget. If this test
|
|
// regresses to taking >1000 iterations, something is wrong with the
|
|
// gradient or weight-update path.
|
|
|
|
#include <array>
|
|
|
|
#include "test_helpers.hpp"
|
|
|
|
#include "../../nisps/ml/mlp.hpp"
|
|
|
|
namespace {
|
|
|
|
NISPS_TEST(mlp_xor_converges) {
|
|
// Modest network: 2 inputs, [4, 4, 4] hidden, 1 output.
|
|
using M = nisps::ml::MLP<2, 4, 4, 4, 1, 4, 1024>;
|
|
M m(7ull);
|
|
m.draw_weights(1.f); // Xavier-ish; needed for sigmoid output to start reasonable
|
|
|
|
// XOR truth table.
|
|
std::array<std::array<float, 2>, 4> X = {{
|
|
{0.f, 0.f},
|
|
{0.f, 1.f},
|
|
{1.f, 0.f},
|
|
{1.f, 1.f},
|
|
}};
|
|
std::array<std::array<float, 1>, 4> Y = {{
|
|
{0.f},
|
|
{1.f},
|
|
{1.f},
|
|
{0.f},
|
|
}};
|
|
|
|
for (std::size_t i = 0; i < 4u; ++i) {
|
|
m.add_example(std::span<const float>(X[i]), std::span<const float>(Y[i]));
|
|
}
|
|
NISPS_EXPECT(m.example_count() == 4u);
|
|
|
|
// Train. Higher LR + more iterations is fine — the test is "did it
|
|
// converge AT ALL within a generous budget".
|
|
const float final_loss = m.train(/*lr=*/0.5f, /*max_iter=*/2000u, /*min_err=*/0.01f);
|
|
NISPS_EXPECT(final_loss < 0.01f);
|
|
|
|
// Sanity: outputs should be near labels for each input.
|
|
for (std::size_t i = 0; i < 4u; ++i) {
|
|
m.set_input(0, X[i][0]);
|
|
m.set_input(1, X[i][1]);
|
|
m.process();
|
|
const float o = m.outputs()[0];
|
|
NISPS_EXPECT_NEAR(o, Y[i][0], 0.2);
|
|
}
|
|
|
|
// Loss history should record at least one entry.
|
|
NISPS_EXPECT(m.loss_history().size() >= 1u);
|
|
}
|
|
|
|
NISPS_TEST(mlp_train_with_no_examples_returns_zero) {
|
|
using M = nisps::ml::MLP<2, 4, 4, 4, 1, 4, 8>;
|
|
M m(0ull);
|
|
const float loss = m.train(0.5f, 100u, 0.001f);
|
|
NISPS_EXPECT(loss == 0.f);
|
|
}
|
|
|
|
NISPS_TEST(mlp_clear_examples_works) {
|
|
using M = nisps::ml::MLP<2, 4, 4, 4, 1, 4, 8>;
|
|
M m(0ull);
|
|
std::array<float, 2> f{0.f, 1.f};
|
|
std::array<float, 1> l{0.5f};
|
|
m.add_example(std::span<const float>(f), std::span<const float>(l));
|
|
NISPS_EXPECT(m.example_count() == 1u);
|
|
m.clear_examples();
|
|
NISPS_EXPECT(m.example_count() == 0u);
|
|
}
|
|
|
|
NISPS_TEST(mlp_dataset_ring_buffer_evicts_oldest) {
|
|
// NMaxExamples=4, add 6 examples; oldest 2 should be evicted.
|
|
using M = nisps::ml::MLP<1, 2, 2, 2, 1, 4, 8>;
|
|
M m(0ull);
|
|
for (int i = 0; i < 6; ++i) {
|
|
std::array<float, 1> f{static_cast<float>(i)};
|
|
std::array<float, 1> l{static_cast<float>(i) * 0.1f};
|
|
m.add_example(std::span<const float>(f), std::span<const float>(l));
|
|
}
|
|
NISPS_EXPECT(m.example_count() == 4u);
|
|
}
|
|
|
|
} // namespace
|