#!/usr/bin/env bash ## entrypoint-shared-ols.sh — PID 1 for the shared-ols tier. ## ## One OpenLiteSpeed container fronting MANY tenants' detached cac-lsphp ## sidecars (the OLS analogue of the shared-httpd container). Webserver ONLY — ## it runs NO PHP locally (render-shared-ols-config.sh strips the stock local ## lsphp; every site's PHP goes to its own sidecar over LSAPI). HAProxy stays ## the TLS/WAF/SNI edge and routes OLS-type hostnames here on :443. ## ## Reuses cac-litespeed's hard-won DAEMON-MODE supervision (NOT `openlitespeed ## -n` + wait): OLS self-restarts on QUIC.cloud IP refresh would otherwise exit ## PID 1 cleanly and tear the container down. See entrypoint-litespeed.sh and ## feedback_ols_quiccloud_restart_kills_container. set -euo pipefail : "${environment:=PROD}" export CONTAINER_ROLE="shared_ols" LSWS_CONF=/usr/local/lsws/conf CERT_DIR="$LSWS_CONF/cert" HEALTH_DIR=/usr/local/lsws/shared-ols-health export SITES_ROOT="${SITES_ROOT:-$LSWS_CONF/shared-sites}" export LSCACHE_ROOT="${LSCACHE_ROOT:-/var/lscache}" export CERT_FILE="$CERT_DIR/shared-ols.crt" export KEY_FILE="$CERT_DIR/shared-ols.key" mkdir -p "$SITES_ROOT" "$LSCACHE_ROOT" "$CERT_DIR" "$HEALTH_DIR/html" ## ---- self-signed cert for the :443 listener (HAProxy verifies none) ---- if [ ! -f "$CERT_FILE" ]; then openssl req -x509 -newkey rsa:2048 -nodes -days 3650 \ -keyout "$KEY_FILE" -out "$CERT_FILE" -subj "/CN=shared-ols" 2>/dev/null fi ## ---- health vhost (catch-all): valid server with zero customer sites + ## answers HAProxy health checks that hit by IP / unknown Host with a 200 ---- cat > "$HEALTH_DIR/vhconf.conf" <<'EOF' docRoot $VH_ROOT/html enableScript 0 context / { allowBrowse 1 location $DOC_ROOT/ } EOF printf 'ok\n' > "$HEALTH_DIR/html/healthz" printf 'shared-ols\n' > "$HEALTH_DIR/html/index.html" ## ---- ownership: OLS reads conf/ as lsadm. chown the base conf dir + health dir ## NON-recursively (the per-site files under conf/shared-sites are written by the ## panel and are world-readable; a recursive chown here would be O(N-sites) on ## every container (re)start, delaying first-listen after a crash). The render ## script chowns the httpd_config.conf it produces. ---- chown lsadm:nogroup "$LSWS_CONF" "$HEALTH_DIR" "$HEALTH_DIR/html" 2>/dev/null || true chown lsadm:nogroup "$HEALTH_DIR/vhconf.conf" "$HEALTH_DIR/html/healthz" "$HEALTH_DIR/html/index.html" 2>/dev/null || true ## ---- assemble httpd_config.conf from the panel's per-site files ---- /scripts/render-shared-ols-config.sh ## ---- stream OLS logs to PID-1 stdout (follows across restarts) ---- mkdir -p /usr/local/lsws/logs touch /usr/local/lsws/logs/error.log /usr/local/lsws/logs/access.log tail -F /usr/local/lsws/logs/error.log /usr/local/lsws/logs/access.log 2>/dev/null & ## ---- .htaccess watcher (required; spec 5.3). Background; the panel monitors ## that it stays alive (its death silently stops rewrite changes applying). ---- /scripts/ols-htaccess-watcher.sh & WATCHER_PID=$! ## ---- supervise OLS in DAEMON mode (verbatim model from entrypoint-litespeed.sh) ---- STOP_REQUESTED=0 term_handler() { STOP_REQUESTED=1 kill "$WATCHER_PID" 2>/dev/null || true /usr/local/lsws/bin/lswsctrl stop >/dev/null 2>&1 || true } trap term_handler TERM INT ## NOT `lswsctrl status` (unlike the otherwise-identical function in ## entrypoint-litespeed.sh). `lswsctrl` appends a timestamped line to ## logs/lsrestart.log on EVERY invocation it makes, including `status` — and ## this loop polls every 3s forever. Measured on whp01: lsrestart.log is 96 MB, ## holding 1,819,286 `status` lines against 2,429 real `restart` lines; at one ## poll per 3s that's ~63 days of continuous polling, which is exactly the ## file's age, and it isn't rotated on any host (whp01/whp02/sdbees all growing ## at ~1.5 MB/day). So: check liveness directly instead of shelling out to a ## tool whose logging is a side effect we don't want on a fixed timer. ## ## Verified (docker run litespeedtech/openlitespeed:1.8.4-lsphp83, the exact ## base this image is built FROM — see Dockerfile.shared-ols): the running main ## process shows in `ps` as `openlitespeed (lshttpd - main)`, one PID, always ## present while OLS is up and absent the instant it is killed (checked via ## `ps aux` immediately after `kill -9` on the main PID). `pgrep -f` matches ## against the full command line, and no other process on this image's `ps` ## output contains that string, so this cannot cross-match an unrelated ## process. It also cannot self-match: pgrep excludes its own PID by default, ## and the invoking process here is bash executing this script file, whose own ## argv never contains the pattern text (only the *source lines* of this script ## do, which `pgrep -f` never sees). ## ## Deliberately NOT the pidfile (/tmp/lshttpd/lshttpd.pid, confirmed present in ## the same probe): pidfiles are known to go stale across a crash (verified — ## after `kill -9` the file still held the dead PID), and treating a stale PID ## as "alive" if the kernel ever reuses that number is a false positive this ## supervisor cannot afford (see below). `pgrep -f` reads the live process ## table, so there is no staleness window to reason about. ## ## Conservative on both failure directions, which matters because this is a ## supervisor predicate, not a metric: a false negative makes start_ols() run ## `lswsctrl start` against an already-running OLS — verified against the same ## probe base image, that is NOT a no-op, it sends SIGUSR1 to the live main ## process, i.e. the same graceful self-restart QUIC.cloud IP refreshes trigger ## (see entrypoint-litespeed.sh's note on that handoff) — a brief, zero- ## downtime blip at worst. A false positive is worse: it leaves a genuinely ## dead OLS un-revived until some later poll happens to notice. So if this ## predicate is ever in doubt it should err toward reporting "not running", not ## "running". ols_running() { pgrep -f 'lshttpd - main' >/dev/null 2>&1 } MAX_STARTS=5 WINDOW=60 starts="" start_ols() { /usr/local/lsws/bin/lswsctrl start >/dev/null 2>&1 || true for _ in $(seq 1 20); do ols_running && return 0 sleep 0.5 done return 1 } if ! start_ols; then echo "entrypoint-shared-ols: OLS failed to start (not running after 10s)." >&2 exit 1 fi echo "entrypoint-shared-ols: OLS started in daemon mode — $(/usr/local/lsws/bin/lswsctrl status 2>/dev/null || true)" while true; do if ols_running; then sleep 3 continue fi sleep 2 if [ "$STOP_REQUESTED" -eq 0 ] && ols_running; then continue fi if [ "$STOP_REQUESTED" -eq 1 ]; then echo "entrypoint-shared-ols: SIGTERM received, OLS stopped — exiting." exit 0 fi now=$(date +%s) starts="$starts $now" pruned="" for t in $starts; do [ $((now - t)) -lt "$WINDOW" ] && pruned="$pruned $t" done starts="$pruned" n=$(echo $starts | wc -w) echo "entrypoint-shared-ols: OLS not running — relaunching (attempt $n/$MAX_STARTS within ${WINDOW}s)." >&2 if [ "$n" -ge "$MAX_STARTS" ]; then echo "entrypoint-shared-ols: OLS crash-looping — bailing for Docker restart policy / monitor." >&2 exit 1 fi start_ols || true done