bot / deploy / ardegazu-bot@.service
 1
 2
 3
 4
 5
 6
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
# systemd template unit for the ardegazu bot fleet — one instance per bot.
#   install: cp deploy/ardegazu-bot@.service /etc/systemd/system/
#            mkdir -p /etc/ardegazu-bot   (one <name>.json per instance)
#            systemctl enable --now ardegazu-bot@duhul
# A static user (not DynamicUser) so every instance and the trainer share the
# models/ checkpoint directory: useradd -r -s /usr/sbin/nologin ardegazu-bot
[Unit]
Description=ardegazu suite bot (%i)
After=network-online.target
Wants=network-online.target

[Service]
User=ardegazu-bot
Group=ardegazu-bot
# Explicit heap cap (also on the workers, forked with --max-old-space-size=384
# in bot.rooms): without one, EACH of the unit's three node processes sizes
# its V8 old-space from the shared cgroup MemoryMax — ~850M apiece against a
# 2000M cap — and whoever allocates last is OOM-killed every couple of hours.
# 576 + 2×384 = 1344M of granted heap leaves native/buffer headroom under Max.
ExecStart=/usr/bin/node --max-old-space-size=576 /opt/ardegazu-bot/stack-a/dist/main.js /etc/ardegazu-bot/%i.json
StateDirectory=ardegazu-bot/%i ardegazu-bot/models
Environment=NODE_ENV=production
Restart=always
RestartSec=5
# MemoryMax is the backstop only; heap sizing is explicit (see ExecStart and
# the worker fork) because V8's own cgroup-derived sizing over-commits a
# multi-process unit. On an 8 GB host three bots at 2000M leave room for the
# three oneshot trainers (1G each, staggered 20 min).
#
# NO MemoryHigh — ever. Over-high reclaim throttling stalls whichever thread
# is allocating, and on 2026-08-31 that was a node-datachannel RTC worker
# holding a libdatachannel lock: the main thread futex-waited on it and two
# bots froze SILENTLY for 20 h (event loop dead, sockets unread, no OOM, no
# restart). A hard MemoryMax kill + Restart=always recovers in seconds; a
# throttle-freeze never recovers on its own.
MemoryMax=2000M
# The last line of defense against any such wedge: main.js pets the watchdog
# every 30 s (via systemd-notify, hence NotifyAccess=all — the ping comes from
# a short-lived child, not the main PID); a frozen event loop stops petting
# and systemd aborts + restarts the unit.
WatchdogSec=180
NotifyAccess=all
NoNewPrivileges=yes
ProtectSystem=strict
ProtectHome=yes

[Install]
WantedBy=multi-user.target

static mirror of HEAD · about · clone: git clone https://git.ardegazu.ro/bot.git