train-kit / examples / train-gym.ts
  1
  2
  3
  4
  5
  6
  7
  8
  9
 10
 11
 12
 13
 14
 15
 16
 17
 18
 19
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
/**
 * A miniature of the fleet's real trainer: valley-blocks self-play in
 * peer-kit's gym, frozen-target DQN fits, and a promotion gate against the
 * stock Dellacherie brain. Everything writes into a scratch model dir.
 *
 *   node --experimental-strip-types examples/train-gym.ts
 *
 * Needs the dev deps: ardegazu-peer-kit (the gym) and @tensorflow/tfjs-node
 * (the fit — the kit's one optional peer, loaded only here).
 */

import { mkdtempSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { ValleyBlocksGym, boardMetrics, stockPick, mulberry32, type VBCandidate } from "ardegazu-peer-kit";
import {
  mlpInit,
  netQs,
  pickAction,
  dqnTargets,
  fitNet,
  capReplay,
  resolveModel,
  gatePromotion,
  isMlpModel,
  type MlpModel,
  type FeatureTransition,
} from "../dist/index.js";

const GAME = "valley-blocks";
const FEATURES = 12;
const GAMMA = 0.95;
const FIT_ROUNDS = 3;
const EPISODES_PER_ROUND = 12;
const EVAL_EPISODES = 30;

/** The fleet's real feature map: classic board metrics of each afterstate. */
function features(candidates: VBCandidate[]): number[][] {
  return candidates.map((c) => {
    const m = boardMetrics(c.grid);
    return [
      c.cleared / 4,
      m.aggregateHeight / 200,
      m.maxHeight / 22,
      m.holes / 40,
      m.bumpiness / 40,
      m.wells / 20,
      m.rowTransitions / 80,
      m.colTransitions / 80,
      c.landingRow / 22,
      c.erodedCells / 8,
      c.useHold ? 1 : 0,
      1,
    ];
  });
}

const modelDir = mkdtempSync(join(tmpdir(), "train-gym-"));
const resumed = resolveModel(modelDir, GAME);
const model: MlpModel<typeof GAME> =
  resumed && isMlpModel(resumed, GAME)
    ? resumed
    : { v: 1, game: GAME, episodes: 0, updatedAt: 0, net: mlpInit([FEATURES, 32, 1]) };

const replay: FeatureTransition<typeof GAME>[] = [];

for (let round = 0; round < FIT_ROUNDS; round++) {
  const gym = new ValleyBlocksGym({ players: 2, seed: 1000 + round, target: 1200, maxPlacements: 150 });
  for (let ep = 0; ep < EPISODES_PER_ROUND; ep++) {
    // seat 0 learns ε-greedy; seat 1 is the stock baseline — real opposition
    let views = gym.reset();
    for (;;) {
      const feats0 = views[0].over ? null : features(views[0].candidates);
      const a0 = feats0 ? pickAction(netQs(model.net, feats0), 0.15) : null;
      const step = gym.step([a0, views[1].over ? null : stockPick(views[1].candidates)]);
      if (feats0 && a0 !== null) {
        const v2 = step.views[0];
        const terminal = step.done || v2.over;
        replay.push({
          g: GAME,
          f: feats0,
          a: a0,
          r: step.rewards[0],
          f2: terminal || !v2.candidates.length ? null : features(v2.candidates),
          done: terminal,
        });
      }
      if (step.done) break;
      views = step.views;
    }
    model.episodes++;
  }
  const { xs, ys } = dqnTargets(model.net, replay, GAMMA);
  model.net = await fitNet(model.net, xs, ys, { lr: 0.001, epochs: 2 });
  capReplay(replay, 20_000);
  console.log(`round ${round + 1}/${FIT_ROUNDS}: ${replay.length} transitions in replay`);
}

// the gate: greedy net vs the stock brain, seeded jitter on both seats so the
// deterministic pair doesn't just replay one episode
const rng = mulberry32(31337);
let wins = 0;
let decided = 0;
const gym = new ValleyBlocksGym({ players: 2, seed: 424242, target: 1200, maxPlacements: 150 });
for (let ep = 0; ep < EVAL_EPISODES; ep++) {
  const netSeat = ep % 2;
  let views = gym.reset();
  for (;;) {
    const acts = views.map((v, seat) => {
      if (v.over || !v.candidates.length) return null;
      if (rng() < 0.03) return Math.floor(rng() * v.candidates.length);
      return seat === netSeat ? pickAction(netQs(model.net, features(v.candidates)), 0) : stockPick(v.candidates);
    });
    const step = gym.step(acts);
    if (step.done) {
      if (step.winner !== null) {
        decided++;
        if (step.winner === netSeat) wins++;
      }
      break;
    }
    views = step.views;
  }
}
const winRate = decided ? wins / decided : 0;
const promoted = winRate >= 0.5 && decided >= 20;
model.updatedAt = Date.now();
const file = gatePromotion(modelDir, GAME, model, promoted);
console.log(
  `gate ${(winRate * 100).toFixed(0)}% vs the stock brain → ${promoted ? "PROMOTED" : "candidate"} (${model.episodes} episodes)`,
);
console.log(`wrote ${file}`);

static mirror of HEAD · about · clone: git clone https://git.ardegazu.ro/train-kit.git