diff --git a/packaging/docker/bin/entrypoint.sh b/packaging/docker/bin/entrypoint.sh index f18a6d4a4492..bd02832f23bb 100755 --- a/packaging/docker/bin/entrypoint.sh +++ b/packaging/docker/bin/entrypoint.sh @@ -53,16 +53,69 @@ fi NEEDS_INITDB=0 +# PIDs of every long-running child we start. Used by the watchdog at the bottom +# of this script to take the whole container down if any single one of them +# dies (#34179). Without this, taosd being OOM-killed would leave the container +# alive (taosadapter still up) and Docker's --restart would never fire. +PIDS="" + +# Install cleanup traps early so that *any* exit path (signal, set -e, watchdog) +# still tears down the background children we are about to start. Without an +# EXIT trap, a `set -e` failure during initdb / create-dnode / create-snode +# would leave orphans behind for tini to reap, and the container would just +# blink instead of going through the normal "restart" loop. +cleanup_children() { + # Use the saved PIDS first so we don't accidentally signal anything else + # tini may have adopted. pkill -P $$ is the safety net for anything + # backgrounded that we forgot to record. + for pid in $PIDS; do + kill -TERM "$pid" 2>/dev/null || true + done + pkill -P $$ 2>/dev/null || true +} +trap 'echo "Received stop signal, killing children"; cleanup_children; exit 0' SIGINT SIGTERM +# EXIT trap: capture the script's exit code BEFORE running cleanup (pkill etc. +# would otherwise overwrite $?), then re-exit with that same code so Docker +# sees the real cause instead of the trap helpers' last status. +trap 'rc=$?; if [ "$rc" -ne 0 ]; then echo "entrypoint exiting abnormally (rc=$rc), killing children"; cleanup_children; fi; exit "$rc"' EXIT + +# Wait up to ~10s for a service's metrics endpoint to come up. If the service +# process itself dies in the meantime (immediate crash on bad config, port +# in use, etc.) fail fast instead of burning the full timeout in silence. +# Per gemini-code-assist review on PR #35368. +wait_for_metric_or_die() { + local svc=$1 pid=$2 url=$3 + for _ in $(seq 1 20); do + if ! kill -0 "$pid" 2>/dev/null; then + echo "$svc (pid $pid) died during startup; aborting entrypoint" >&2 + exit 1 + fi + curl -sf "$url" && return 0 + sleep 0.5 + done +} + # if dnode has been created or has mnode ep set or the host is first ep or not for cluster, just start. if [ -f "$DATA_DIR/dnode/dnode.json" ] || [ -f "$DATA_DIR/dnode/mnodeEpSet.json" ] || [ "$FQDN" = "$FIRST_EP_HOST" ]; then echo "start taosd with mnode ep set" taosd & + TAOSD_PID=$! + PIDS="$PIDS $TAOSD_PID" while true; do - es=$(taos -h $FIRST_EP_HOST -P $FIRST_EP_PORT --check | grep "^[0-9]*:") + # If taosd died during startup, fail fast instead of looping forever + # waiting for `taos --check` to succeed against a dead server. + if ! kill -0 "$TAOSD_PID" 2>/dev/null; then + echo "taosd (pid $TAOSD_PID) died during startup; aborting entrypoint" >&2 + exit 1 + fi + # `|| true` keeps a transient connection error from tripping set -e while + # we're still polling for taosd to come up; the kill -0 above is the + # authoritative "did the process die" check. + es=$(taos -h $FIRST_EP_HOST -P $FIRST_EP_PORT --check 2>/dev/null | grep "^[0-9]*:" || true) echo ${es} - if [ "${es%%:*}" -eq 2 ]; then + if [ -n "$es" ] && [ "${es%%:*}" = "2" ]; then # Initialization scripts should only work in first node. if [ "$FQDN" = "$FIRST_EP_HOST" ]; then @@ -85,14 +138,26 @@ if [ -f "$DATA_DIR/dnode/dnode.json" ] || done # others will first wait the first ep ready. else + TAOSD_PID="" if [ "$TAOS_FIRST_EP" = "" ]; then echo "run TDengine with single node." taosd & + TAOSD_PID=$! + PIDS="$PIDS $TAOSD_PID" fi while true; do - es=$(taos -h $FIRST_EP_HOST -P $FIRST_EP_PORT --check | grep "^[0-9]*:") + # If we started taosd locally and it died early, fail fast. + # Cluster non-first nodes (TAOSD_PID empty) wait for the remote first EP. + if [ -n "$TAOSD_PID" ] && ! kill -0 "$TAOSD_PID" 2>/dev/null; then + echo "taosd (pid $TAOSD_PID) died during startup; aborting entrypoint" >&2 + exit 1 + fi + # `|| true` keeps a transient connection error from tripping set -e while + # we're still polling for taosd to come up; the kill -0 above is the + # authoritative "did the process die" check. + es=$(taos -h $FIRST_EP_HOST -P $FIRST_EP_PORT --check 2>/dev/null | grep "^[0-9]*:" || true) echo ${es} - if [ "${es%%:*}" -eq 2 ]; then + if [ -n "$es" ] && [ "${es%%:*}" = "2" ]; then echo "execute create dnode" sh -c "taos -p'$TAOS_ROOT_PASSWORD' -h $FIRST_EP_HOST -P $FIRST_EP_PORT -s 'create dnode \"$FQDN:$SERVER_PORT\";'" break @@ -101,32 +166,26 @@ else done fi -if [ "$DISABLE_ADAPTER" = "0" ]; then - which taosadapter >/dev/null && taosadapter & - # wait for 6041 port ready - for _ in $(seq 1 20); do - curl -sf http://localhost:6041/metrics && break - sleep 0.5 - done +if [ "$DISABLE_ADAPTER" = "0" ] && which taosadapter >/dev/null; then + taosadapter & + TAOSADAPTER_PID=$! + PIDS="$PIDS $TAOSADAPTER_PID" + wait_for_metric_or_die taosadapter "$TAOSADAPTER_PID" http://localhost:6041/metrics fi -if [ "$DISABLE_KEEPER" = "0" ]; then +if [ "$DISABLE_KEEPER" = "0" ] && which taoskeeper >/dev/null; then sleep 3 - which taoskeeper >/dev/null && taoskeeper & - # wait for 6043 port ready - for _ in $(seq 1 20); do - curl -sf http://localhost:6043/metrics && break - sleep 0.5 - done + taoskeeper & + TAOSKEEPER_PID=$! + PIDS="$PIDS $TAOSKEEPER_PID" + wait_for_metric_or_die taoskeeper "$TAOSKEEPER_PID" http://localhost:6043/metrics fi -if [ "$DISABLE_EXPLORER" = "0" ]; then - which taos-explorer >/dev/null && taos-explorer & - # wait for 6060 port ready - for _ in $(seq 1 20); do - curl -sf http://localhost:6060/metrics && break - sleep 0.5 - done +if [ "$DISABLE_EXPLORER" = "0" ] && which taos-explorer >/dev/null; then + taos-explorer & + TAOSEXPLORER_PID=$! + PIDS="$PIDS $TAOSEXPLORER_PID" + wait_for_metric_or_die taos-explorer "$TAOSEXPLORER_PID" http://localhost:6060/metrics fi if [ "$NEEDS_INITDB" = "1" ]; then @@ -158,5 +217,34 @@ fi sh -c "taos -p'$TAOS_ROOT_PASSWORD' -h $FIRST_EP_HOST -P $FIRST_EP_PORT -s 'create snode on dnode 1;'" -trap 'echo "Received stop signal, killing children"; pkill -P $$ || true; exit 0' SIGINT SIGTERM -wait \ No newline at end of file +# Watchdog: if any of taosd / taosadapter / taoskeeper / taos-explorer dies +# (e.g. taosd OOM-killed), tear the whole container down so Docker's --restart +# can bring it back. Before this, only taosd dying would silently leave +# taosadapter accepting requests it could not fulfill (#34179). +# +# Polling instead of `wait -n` keeps this compatible with bash versions that +# do not support `-n`; the 2s tick is negligible CPU. +# Cleanup traps were installed at the top of this script — both the SIGINT/ +# SIGTERM and the EXIT trap call cleanup_children, so we don't need to repeat +# the pkill plumbing here. + +if [ -z "$PIDS" ]; then + # cluster non-first node with every sidecar disabled — nothing to watch. + # Stay alive so the SIGINT/SIGTERM trap can still stop the container. + echo "No background TDengine processes to watch; idling" + while true; do sleep 3600; done +fi + +echo "All services started, watching child processes: $PIDS" +while true; do + for pid in $PIDS; do + if ! kill -0 "$pid" 2>/dev/null; then + echo "child process $pid exited; shutting container down so Docker can restart it" + # cleanup_children runs via the EXIT trap; just give children a + # beat to actually wind down before we let the container exit. + sleep 1 + exit 1 + fi + done + sleep 2 +done \ No newline at end of file