| # ============================================================================= | |
| # deepseek-v4-flash.service (stage 2: MTP speculative decoding) | |
| # | |
| # systemd unit for the DeepSeek-V4-Flash vLLM server on 4x RTX 6000 Ada | |
| # (SM89). serve.sh (generated by setup_deepseek_v4_sm89.sh) is the single | |
| # source of truth for environment variables, model-snapshot resolution, and | |
| # launch flags; this unit only supervises it. systemd's clean service | |
| # environment additionally guarantees no stale shell-profile exports can | |
| # leak in. | |
| # | |
| # Measured 2026-07-21: 115-133 tok/s sustained generation (~1.5-1.65x over | |
| # the ~80 tok/s plain-decoding baseline), 96.6-100% draft acceptance. | |
| # | |
| # Incident 2026-07-23: a Triton JIT module load OOM'd and killed a worker | |
| # mid-run; the API server then exited *cleanly* (main PID rc 0), so the old | |
| # Restart=on-failure never fired and the port stayed dead. Hence Restart=always | |
| # below -- for an inference server a crash that looks like a clean exit is | |
| # still a crash; only 'systemctl stop' should keep it down. ExecStartPost runs | |
| # warmup.sh after every (re)start so Triton specializations compile up front. | |
| # | |
| # NOTE: the setup script also generates a copy of this unit inside the | |
| # project folder with User/paths already baked in. EDIT the four | |
| # /home/sinclair/deepseek-v4-serve references and User/Group below if your | |
| # layout differs (three ExecStart* lines + WorkingDirectory). | |
| # | |
| # Install: | |
| # sudo cp deepseek-v4-flash.service /etc/systemd/system/ | |
| # sudo systemctl daemon-reload | |
| # sudo systemctl enable --now deepseek-v4-flash | |
| # Logs: | |
| # journalctl -fu deepseek-v4-flash | |
| # ============================================================================= | |
| [Unit] | |
| Description=vLLM OpenAI API server - DeepSeek-V4-Flash + MTP on 4x RTX 6000 Ada (SM89) | |
| After=network-online.target | |
| Wants=network-online.target | |
| # A boot loop on a 159 GB model is expensive: allow 3 failed starts per 10 min, | |
| # then stay down until 'systemctl reset-failed deepseek-v4-flash'. | |
| StartLimitIntervalSec=600 | |
| StartLimitBurst=3 | |
| [Service] | |
| Type=exec | |
| User=sinclair | |
| Group=sinclair | |
| WorkingDirectory=/home/sinclair/deepseek-v4-serve | |
| # Fail fast (and retry via Restart=) if the NVIDIA driver is not up yet, | |
| # e.g. when racing device initialization right after boot. | |
| ExecStartPre=/usr/bin/nvidia-smi | |
| ExecStart=/home/sinclair/deepseek-v4-serve/serve.sh | |
| # Warm the Triton shape specializations after EVERY start (incl. auto-restart). | |
| # warmup.sh waits for the API itself and is bounded, so it cannot hang the | |
| # unit; '-' prefix + timeout mean a wedged warmup never fails the start. | |
| ExecStartPost=-/usr/bin/timeout 900 /home/sinclair/deepseek-v4-serve/warmup.sh | |
| # Budget must cover ExecStartPre + weight load + bounded warmup. | |
| TimeoutStartSec=1200 | |
| # Restart=always, NOT on-failure: a JIT-OOM crash can present as a clean exit | |
| # (see incident note above); only an explicit 'systemctl stop' should end it. | |
| Restart=always | |
| # Weight load alone is ~80 s (+ ~5 s MTP drafter); don't hammer restarts. | |
| RestartSec=15 | |
| # Give the engine time to tear down 4 TP workers and NCCL cleanly on stop. | |
| TimeoutStopSec=90 | |
| # SIGTERM to the API server first, then SIGKILL the whole cgroup | |
| # (EngineCore + 4 worker processes) if anything lingers. | |
| KillMode=mixed | |
| LimitNOFILE=65535 | |
| LimitMEMLOCK=infinity | |
| [Install] | |
| WantedBy=multi-user.target | |