fix(deploy): 群晖没有 pgrep,那三处「守卫」一直是空转

上一版给 push.sh 加的 pid 比对刚跑就露馅了:`sh: pgrep: command not found`,
于是 BEFORE 和 AFTER 都是空字符串,`[ -n "$BEFORE" ]` 直接跳过检查——我刚加
的安全网自己就是个摆设。

顺手查了一圈,stop.sh 和 start.sh 里同样的 pgrep 也一直在空转:
- stop.sh 的 `pgrep ... || exit 0` 每次都以 127 失败,永远匹配不到那个提前
  返回,所以每次停服都白等满 15 秒,也从没真正确认过残留 worker 已经没了。
- start.sh 的等待循环同理,白等 20 秒。

DSM 有 pkill 没有 pgrep,而且非特权的 ps 看不见 root 起的进程(服务是开机
以 root 启动的),两者任一都会让检查静默返回「没有」——这正是最危险的答案,
因为它长得和「已经停干净了」一模一样。

- push.sh 改用 `sudo ps -eo pid,args | grep`;并且 AFTER 为空时直接报错退出,
  不再把「看不见」当成「通过」
- stop.sh / start.sh 改用 `ps -eo args | grep -q`

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
ericwyuan
2026-09-01 14:34:27 +08:00
parent 570d1590eb
commit 14b6bbecb4
3 changed files with 30 additions and 5 deletions

View File

@@ -79,7 +79,15 @@ $TAR -C "$REPO/client/build" . | sh_ "tar xzf - -C '$STATIC'"
# Hence sudo, matching how the service actually runs. # Hence sudo, matching how the service actually runs.
SUDO="echo '$PASS' | sudo -S" SUDO="echo '$PASS' | sudo -S"
echo "==> restart (sudo: the service runs as root, started at boot)" echo "==> restart (sudo: the service runs as root, started at boot)"
BEFORE=$(sh_ "pgrep -f '$APP/backend/.venv/bin/gunicorn' | tr '\n' ',' " || true) # DSM has no pgrep, and an unprivileged `ps` cannot see a root-owned process
# — either one silently returns nothing, which would make the pid comparison
# below always "pass" and put the check right back where it started.
pids_() {
sh_ "echo '$PASS' | sudo -S ps -eo pid,args 2>/dev/null \
| grep '$APP/backend/.venv/bin/gunicorn' | grep -v grep \
| awk '{print \$1}' | sort -n | tr '\n' ','"
}
BEFORE=$(pids_ || true)
sh_ "cd '$APP' && $SUDO sh deploy/stop.sh >/dev/null 2>&1; sleep 3; $SUDO sh deploy/start.sh" \ sh_ "cd '$APP' && $SUDO sh deploy/stop.sh >/dev/null 2>&1; sleep 3; $SUDO sh deploy/start.sh" \
|| { echo "restart failed" >&2; exit 1; } || { echo "restart failed" >&2; exit 1; }
@@ -89,8 +97,13 @@ sh_ "curl -sf -m 5 -o /dev/null -w 'local api: %{http_code}\n' \
# A 200 alone proves nothing: it is exactly what a surviving old master # A 200 alone proves nothing: it is exactly what a surviving old master
# returns. The master pid must have changed for the new code to be loaded. # returns. The master pid must have changed for the new code to be loaded.
AFTER=$(sh_ "pgrep -f '$APP/backend/.venv/bin/gunicorn' | tr '\n' ',' " || true) AFTER=$(pids_ || true)
if [ -n "$BEFORE" ] && [ "$BEFORE" = "$AFTER" ]; then if [ -z "$AFTER" ]; then
echo "cannot see any gunicorn process - the pid check could not run." >&2
echo "Verify by hand before trusting this deploy." >&2
exit 1
fi
if [ "$BEFORE" = "$AFTER" ]; then
echo "the gunicorn pids did not change ($AFTER) - the old process is still" >&2 echo "the gunicorn pids did not change ($AFTER) - the old process is still" >&2
echo "serving and your changes are NOT live. Check $APP/logs/error.log." >&2 echo "serving and your changes are NOT live. Check $APP/logs/error.log." >&2
exit 1 exit 1

View File

@@ -11,9 +11,14 @@ GUNICORN="$APP/backend/.venv/bin/gunicorn"
# the new master then started, reported success, and served nothing — the # the new master then started, reported success, and served nothing — the
# site was down while every log line looked normal. # site was down while every log line looked normal.
"$APP/deploy/stop.sh" >/dev/null 2>&1 "$APP/deploy/stop.sh" >/dev/null 2>&1
# `ps | grep`, not pgrep: DSM has no pgrep, so this loop used to fail with 127
# every second and wait the full 20s whether or not anything was still running.
alive() { ps -eo args 2>/dev/null | grep -q "^$GUNICORN"; }
i=0 i=0
while [ $i -lt 20 ]; do while [ $i -lt 20 ]; do
pgrep -f "$GUNICORN" >/dev/null 2>&1 || break alive || break
sleep 1 sleep 1
i=$((i + 1)) i=$((i + 1))
done done

View File

@@ -9,9 +9,16 @@ fi
# The master's workers do not always go with it, and a surviving worker keeps # The master's workers do not always go with it, and a surviving worker keeps
# :8124 bound — which looks exactly like a healthy start that serves nothing. # :8124 bound — which looks exactly like a healthy start that serves nothing.
#
# `ps | grep` rather than pgrep: DSM does not ship pgrep, so the original
# check failed with 127 on every iteration — never matching the `|| exit 0`
# early return, always burning the full 15 seconds, and never actually
# confirming anything.
alive() { ps -eo args 2>/dev/null | grep -q "^$GUNICORN"; }
i=0 i=0
while [ $i -lt 15 ]; do while [ $i -lt 15 ]; do
pgrep -f "$GUNICORN" >/dev/null 2>&1 || exit 0 alive || exit 0
sleep 1 sleep 1
i=$((i + 1)) i=$((i + 1))
done done