;; ported-from: src/tabular.ts @ v1.0.0 (extracted-from: bot/src/rl/train.ts @ fa686ee,
;; seanceQ / seanceUpdate)
;;
;; Tabular Q-learning over a string-encoded state space — the right tool when
;; the whole space is a few hundred cells.
(ns ardegazu.train.tabular)
(defn ensure-row
"The Q row for a state, created zeroed on first sight (q[s] ??= zeros)."
[^js model s n-actions]
(let [q (.-q model)
row (unchecked-get q s)]
(if (some? row)
row
(let [fresh (.fill (js/Array. n-actions) 0)]
(unchecked-set q s fresh)
fresh))))
(defn q-update
"One Bellman update: q[a] += α·(r + γ·max q(s') − q[a]); terminal next = 0."
[model ^js t ^js opts]
(let [n-actions (.-nActions opts)
q (ensure-row model (.-s t) n-actions)
next-v (if ^boolean (js* "(~{} || !~{})" (.-done t) (.-s2 t))
0
(.apply js/Math.max nil (ensure-row model (.-s2 t) n-actions)))
a (.-a t)]
(unchecked-set q a
(+ (unchecked-get q a)
(* (.-alpha opts)
(- (+ (.-r t) (* (.-gamma opts) next-v))
(unchecked-get q a)))))
js/undefined))