train-kit / test / dqn.test.mjs
 1
 2
 3
 4
 5
 6
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
// Frozen-target regression sets and the replay cap.
import test from "node:test";
import assert from "node:assert/strict";
import { netQs, dqnTargets, capReplay } from "../dist/dqn.js";

// identity-ish net: y = x0 (one linear layer)
const net = { sizes: [2, 1], layers: [{ w: [1, 0], b: [0] }] };

test("netQs is one forward pass per candidate", () => {
  assert.deepEqual(netQs(net, [[3, 9], [5, 9], [-2, 9]]), [3, 5, -2]);
});

test("dqnTargets: terminal = r, non-terminal = r + γ·max V(f2), x = f[a]", () => {
  const batch = [
    { g: "toy", f: [[1, 0], [2, 0]], a: 1, r: 0.5, f2: [[4, 0], [7, 0]], done: false },
    { g: "toy", f: [[3, 0]], a: 0, r: -1, f2: null, done: true },
  ];
  const { xs, ys } = dqnTargets(net, batch, 0.9);
  assert.deepEqual(xs, [[2, 0], [3, 0]]);
  assert.deepEqual(ys, [0.5 + 0.9 * 7, -1]);
});

test("capReplay keeps the newest", () => {
  const r = [1, 2, 3, 4, 5];
  capReplay(r, 3);
  assert.deepEqual(r, [3, 4, 5]);
  capReplay(r, 10);
  assert.deepEqual(r, [3, 4, 5], "under the cap is untouched");
});

static mirror of HEAD · about · clone: git clone https://git.ardegazu.ro/train-kit.git