File size: 3,313 Bytes
a72139f
 
 
 
 
 
 
 
 
 
 
 
 
f5a38e0
 
 
 
 
 
 
a72139f
f5a38e0
a72139f
f5a38e0
a72139f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f5a38e0
 
 
 
 
 
 
 
 
 
a72139f
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
# =============================================================================
# deepseek-v4-flash.service   (stage 2: MTP speculative decoding)
#
# systemd unit for the DeepSeek-V4-Flash vLLM server on 4x RTX 6000 Ada
# (SM89). serve.sh (generated by setup_deepseek_v4_sm89.sh) is the single
# source of truth for environment variables, model-snapshot resolution, and
# launch flags; this unit only supervises it. systemd's clean service
# environment additionally guarantees no stale shell-profile exports can
# leak in.
#
# Measured 2026-07-21: 115-133 tok/s sustained generation (~1.5-1.65x over
# the ~80 tok/s plain-decoding baseline), 96.6-100% draft acceptance.
#
# Incident 2026-07-23: a Triton JIT module load OOM'd and killed a worker
# mid-run; the API server then exited *cleanly* (main PID rc 0), so the old
# Restart=on-failure never fired and the port stayed dead. Hence Restart=always
# below -- for an inference server a crash that looks like a clean exit is
# still a crash; only 'systemctl stop' should keep it down. ExecStartPost runs
# warmup.sh after every (re)start so Triton specializations compile up front.
#
# NOTE: the setup script also generates a copy of this unit inside the
# project folder with User/paths already baked in. EDIT the four
# /home/sinclair/deepseek-v4-serve references and User/Group below if your
# layout differs (three ExecStart* lines + WorkingDirectory).
#
# Install:
#   sudo cp deepseek-v4-flash.service /etc/systemd/system/
#   sudo systemctl daemon-reload
#   sudo systemctl enable --now deepseek-v4-flash
# Logs:
#   journalctl -fu deepseek-v4-flash
# =============================================================================

[Unit]
Description=vLLM OpenAI API server - DeepSeek-V4-Flash + MTP on 4x RTX 6000 Ada (SM89)
After=network-online.target
Wants=network-online.target
# A boot loop on a 159 GB model is expensive: allow 3 failed starts per 10 min,
# then stay down until 'systemctl reset-failed deepseek-v4-flash'.
StartLimitIntervalSec=600
StartLimitBurst=3

[Service]
Type=exec
User=sinclair
Group=sinclair
WorkingDirectory=/home/sinclair/deepseek-v4-serve

# Fail fast (and retry via Restart=) if the NVIDIA driver is not up yet,
# e.g. when racing device initialization right after boot.
ExecStartPre=/usr/bin/nvidia-smi

ExecStart=/home/sinclair/deepseek-v4-serve/serve.sh

# Warm the Triton shape specializations after EVERY start (incl. auto-restart).
# warmup.sh waits for the API itself and is bounded, so it cannot hang the
# unit; '-' prefix + timeout mean a wedged warmup never fails the start.
ExecStartPost=-/usr/bin/timeout 900 /home/sinclair/deepseek-v4-serve/warmup.sh
# Budget must cover ExecStartPre + weight load + bounded warmup.
TimeoutStartSec=1200

# Restart=always, NOT on-failure: a JIT-OOM crash can present as a clean exit
# (see incident note above); only an explicit 'systemctl stop' should end it.
Restart=always
# Weight load alone is ~80 s (+ ~5 s MTP drafter); don't hammer restarts.
RestartSec=15
# Give the engine time to tear down 4 TP workers and NCCL cleanly on stop.
TimeoutStopSec=90
# SIGTERM to the API server first, then SIGKILL the whole cgroup
# (EngineCore + 4 worker processes) if anything lingers.
KillMode=mixed
LimitNOFILE=65535
LimitMEMLOCK=infinity

[Install]
WantedBy=multi-user.target