train-kit / src / ardegazu / train / tabular.cljs
 1
 2
 3
 4
 5
 6
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
;; ported-from: src/tabular.ts @ v1.0.0 (extracted-from: bot/src/rl/train.ts @ fa686ee,
;;   seanceQ / seanceUpdate)
;;
;; Tabular Q-learning over a string-encoded state space — the right tool when
;; the whole space is a few hundred cells.
(ns ardegazu.train.tabular)

(defn ensure-row
  "The Q row for a state, created zeroed on first sight (q[s] ??= zeros)."
  [^js model s n-actions]
  (let [q (.-q model)
        row (unchecked-get q s)]
    (if (some? row)
      row
      (let [fresh (.fill (js/Array. n-actions) 0)]
        (unchecked-set q s fresh)
        fresh))))

(defn q-update
  "One Bellman update: q[a] += α·(r + γ·max q(s') − q[a]); terminal next = 0."
  [model ^js t ^js opts]
  (let [n-actions (.-nActions opts)
        q (ensure-row model (.-s t) n-actions)
        next-v (if ^boolean (js* "(~{} || !~{})" (.-done t) (.-s2 t))
                 0
                 (.apply js/Math.max nil (ensure-row model (.-s2 t) n-actions)))
        a (.-a t)]
    (unchecked-set q a
                   (+ (unchecked-get q a)
                      (* (.-alpha opts)
                         (- (+ (.-r t) (* (.-gamma opts) next-v))
                            (unchecked-get q a)))))
    js/undefined))

static mirror of HEAD · about · clone: git clone https://git.ardegazu.ro/train-kit.git