2026-06-10 01:22:14 -07:00
|
|
|
#!/usr/bin/env bash
|
|
|
|
|
## entrypoint-shared-ols.sh — PID 1 for the shared-ols tier.
|
|
|
|
|
##
|
|
|
|
|
## One OpenLiteSpeed container fronting MANY tenants' detached cac-lsphp
|
|
|
|
|
## sidecars (the OLS analogue of the shared-httpd container). Webserver ONLY —
|
|
|
|
|
## it runs NO PHP locally (render-shared-ols-config.sh strips the stock local
|
|
|
|
|
## lsphp; every site's PHP goes to its own sidecar over LSAPI). HAProxy stays
|
|
|
|
|
## the TLS/WAF/SNI edge and routes OLS-type hostnames here on :443.
|
|
|
|
|
##
|
|
|
|
|
## Reuses cac-litespeed's hard-won DAEMON-MODE supervision (NOT `openlitespeed
|
|
|
|
|
## -n` + wait): OLS self-restarts on QUIC.cloud IP refresh would otherwise exit
|
|
|
|
|
## PID 1 cleanly and tear the container down. See entrypoint-litespeed.sh and
|
|
|
|
|
## feedback_ols_quiccloud_restart_kills_container.
|
|
|
|
|
set -euo pipefail
|
|
|
|
|
|
|
|
|
|
: "${environment:=PROD}"
|
|
|
|
|
export CONTAINER_ROLE="shared_ols"
|
|
|
|
|
|
|
|
|
|
LSWS_CONF=/usr/local/lsws/conf
|
|
|
|
|
CERT_DIR="$LSWS_CONF/cert"
|
|
|
|
|
HEALTH_DIR=/usr/local/lsws/shared-ols-health
|
|
|
|
|
export SITES_ROOT="${SITES_ROOT:-$LSWS_CONF/shared-sites}"
|
|
|
|
|
export LSCACHE_ROOT="${LSCACHE_ROOT:-/var/lscache}"
|
|
|
|
|
export CERT_FILE="$CERT_DIR/shared-ols.crt"
|
|
|
|
|
export KEY_FILE="$CERT_DIR/shared-ols.key"
|
|
|
|
|
|
|
|
|
|
mkdir -p "$SITES_ROOT" "$LSCACHE_ROOT" "$CERT_DIR" "$HEALTH_DIR/html"
|
|
|
|
|
|
|
|
|
|
## ---- self-signed cert for the :443 listener (HAProxy verifies none) ----
|
|
|
|
|
if [ ! -f "$CERT_FILE" ]; then
|
|
|
|
|
openssl req -x509 -newkey rsa:2048 -nodes -days 3650 \
|
|
|
|
|
-keyout "$KEY_FILE" -out "$CERT_FILE" -subj "/CN=shared-ols" 2>/dev/null
|
|
|
|
|
fi
|
|
|
|
|
|
2026-08-22 19:44:01 -07:00
|
|
|
## ---- health vhost (catch-all) ----
|
|
|
|
|
## This vhost is mapped `map _health *` by render-shared-ols-config.sh, so it
|
|
|
|
|
## answers EVERY Host that no customer vhost claims. It exists so the server is
|
|
|
|
|
## valid with zero customer sites and so local/edge health probes get a 200.
|
|
|
|
|
##
|
|
|
|
|
## IT MUST NOT ANSWER 200 FOR AN UNMAPPED CUSTOMER HOST.
|
|
|
|
|
## It used to serve html/index.html ("shared-ols", 11 bytes) with HTTP 200 to
|
|
|
|
|
## anything that fell through. Measured 2026-08: three live customer sites
|
|
|
|
|
## (their vhost had silently stopped being rendered) served that 200 for ~2
|
|
|
|
|
## months and no monitor noticed, because every uptime check asks "is it 200?"
|
|
|
|
|
## and the answer was yes. A hostname this server cannot serve now gets
|
|
|
|
|
## 421 Misdirected Request -- semantically exact (RFC 7540 s9.1.2: the server is
|
|
|
|
|
## not able to produce a response for the combination of scheme and authority in
|
|
|
|
|
## the request URI) and unambiguous to monitoring in a way 404 is not, since a
|
|
|
|
|
## 404 is a perfectly normal answer from a real, working site.
|
|
|
|
|
##
|
|
|
|
|
## THE DISCRIMINATOR: request path /healthz AND an INTERNAL client address.
|
|
|
|
|
## * Path alone is not enough -- anyone can request /healthz.
|
|
|
|
|
## * REMOTE_ADDR is the half an outside caller cannot choose, BECAUSE of
|
|
|
|
|
## `useIpInProxyHeader 1` in httpd_config_base.tpl: OLS resolves the client
|
|
|
|
|
## IP from X-Forwarded-For, and HAProxy -- the only thing that can reach
|
|
|
|
|
## this tier, which has no host-published ports and sits on client-net --
|
|
|
|
|
## SETS (not appends) that header:
|
|
|
|
|
## `http-request set-header X-Forwarded-For %[var(txn.real_ip)]` in
|
|
|
|
|
## haproxy-manager-base/templates/hap_backend.tpl, which DISCARDS whatever
|
|
|
|
|
## the client sent. So a request arriving from outside carries the real
|
|
|
|
|
## public client IP. Verified on the lab: `-H 'X-Forwarded-For: 8.8.8.8'`
|
|
|
|
|
## on /healthz returns 421.
|
|
|
|
|
## * MEASURED LIMIT OF THE IP GATE, stated plainly rather than assumed away:
|
|
|
|
|
## OLS takes the FIRST element of a multi-value X-Forwarded-For as
|
|
|
|
|
## REMOTE_ADDR. `X-Forwarded-For: 10.0.0.1, 8.8.8.8` returns 200 on /healthz
|
|
|
|
|
## here, and anchoring the pattern ^...$ does NOT change that (tested both
|
|
|
|
|
## ways) -- because by the time the rule sees REMOTE_ADDR it is already the
|
|
|
|
|
## single token `10.0.0.1`. The anchors are kept because they are correct
|
|
|
|
|
## and free, not because they close that hole. What closes it is HAProxy:
|
|
|
|
|
## `http-request set-header X-Forwarded-For %[var(txn.real_ip)]` REPLACES
|
|
|
|
|
## whatever the client sent with one value.
|
|
|
|
|
## * AND THE GATE IS NOT LOAD-BEARING ANYWAY. It only guards /healthz. `/`,
|
|
|
|
|
## and every other path, is 421 UNCONDITIONALLY -- no header, source
|
|
|
|
|
## address or Host can talk this vhost into a 200 there. So even a total
|
|
|
|
|
## bypass of the IP gate buys an attacker a 3-byte `ok` on /healthz, never
|
|
|
|
|
## a "the site is up" answer on the URL a monitor actually requests. That
|
|
|
|
|
## is the property this change exists to guarantee, and it does not rest on
|
|
|
|
|
## anything spoofable.
|
|
|
|
|
## * The probes that MUST keep passing all originate inside: the Docker
|
|
|
|
|
## HEALTHCHECK (`curl -sfk https://127.0.0.1/healthz` in Dockerfile.shared-ols,
|
|
|
|
|
## overridden by WHP's setup-shared-ols.sh to `https://localhost/healthz`)
|
|
|
|
|
## connects over loopback and sends no X-Forwarded-For, so REMOTE_ADDR falls
|
|
|
|
|
## back to the peer, 127.0.0.1. An edge/host probe of the container IP comes
|
|
|
|
|
## from the docker gateway (172.16/12), also allowed.
|
|
|
|
|
##
|
|
|
|
|
## `/` is 421 for EVERY client, internal ones included -- there is deliberately
|
|
|
|
|
## no "internal clients still get the old 200 page" escape hatch, because that
|
|
|
|
|
## is exactly the response that hid the outage. Anything probing this tier for
|
|
|
|
|
## liveness must ask for /healthz.
|
|
|
|
|
##
|
|
|
|
|
## WHY REWRITE AND NOT A REDIRECT CONTEXT: `context / { type redirect
|
|
|
|
|
## statusCode 421 }` was measured on this image (OLS 1.8.4) and does NOT work --
|
|
|
|
|
## 421 is not in OLS's accepted status-code list, so it silently degrades to a
|
|
|
|
|
## 302 with a literal, unexpanded `Location: $DOC_ROOT/?`. A rewrite `[R=421,L]`
|
|
|
|
|
## does emit a real 421.
|
|
|
|
|
##
|
|
|
|
|
## WHY THE THE_REQUEST GUARD ON THE ERROR PAGE: a bare [R=421] has no body, and
|
|
|
|
|
## a bare 421 with no explanation is a support ticket. `errorpage 421` supplies
|
|
|
|
|
## the body, but OLS fetches that URL as a fresh internal request that runs
|
|
|
|
|
## through these same rules -- without an exception it is itself 421'd and the
|
|
|
|
|
## body comes back empty (measured: content-length 0). %{IS_SUBREQ} and
|
|
|
|
|
## %{ENV:REDIRECT_STATUS} are NOT populated by OLS's rewrite engine (both
|
|
|
|
|
## measured, both no-ops), but %{THE_REQUEST} keeps the ORIGINAL request line
|
|
|
|
|
## across the internal fetch. So: serve misdirected.html when the client did not
|
|
|
|
|
## itself ask for it, which lets the error page render while a direct external
|
|
|
|
|
## GET /misdirected.html still gets 421 -- no path on this catch-all answers 200
|
|
|
|
|
## to an outside caller.
|
|
|
|
|
##
|
|
|
|
|
## The body is deliberately generic: no branding, no customer names, nothing
|
|
|
|
|
## that reveals which hostnames this server does serve. Every unmapped Host and
|
|
|
|
|
## every path gets the byte-identical 421, so the response cannot be used to
|
|
|
|
|
## enumerate configured vs unconfigured hostnames.
|
2026-06-10 01:22:14 -07:00
|
|
|
cat > "$HEALTH_DIR/vhconf.conf" <<'EOF'
|
|
|
|
|
docRoot $VH_ROOT/html
|
|
|
|
|
enableScript 0
|
2026-08-22 19:44:01 -07:00
|
|
|
|
|
|
|
|
errorpage 421 {
|
|
|
|
|
url /misdirected.html
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
rewrite {
|
|
|
|
|
enable 1
|
|
|
|
|
rules <<<END_rules
|
|
|
|
|
RewriteCond %{THE_REQUEST} !\s/+misdirected\.html
|
|
|
|
|
RewriteRule ^/?misdirected\.html$ - [L]
|
|
|
|
|
RewriteCond %{REMOTE_ADDR} ^(127\.0\.0\.1|::1|10\.[0-9.]+|192\.168\.[0-9.]+|172\.(1[6-9]|2[0-9]|3[01])\.[0-9.]+)$
|
|
|
|
|
RewriteRule ^/?healthz$ - [L]
|
|
|
|
|
RewriteRule .* - [R=421,L]
|
|
|
|
|
END_rules
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-10 01:22:14 -07:00
|
|
|
context / {
|
|
|
|
|
allowBrowse 1
|
|
|
|
|
location $DOC_ROOT/
|
|
|
|
|
}
|
|
|
|
|
EOF
|
|
|
|
|
printf 'ok\n' > "$HEALTH_DIR/html/healthz"
|
2026-08-22 19:44:01 -07:00
|
|
|
cat > "$HEALTH_DIR/html/misdirected.html" <<'EOF'
|
|
|
|
|
<!DOCTYPE html>
|
|
|
|
|
<html lang="en">
|
|
|
|
|
<head><meta charset="utf-8"><title>421 Misdirected Request</title></head>
|
|
|
|
|
<body>
|
|
|
|
|
<h1>421 Misdirected Request</h1>
|
|
|
|
|
<p>This hostname is not configured on this server.</p>
|
|
|
|
|
<p>If you own this domain, check that its DNS points to the correct server and
|
|
|
|
|
that the site is active in your hosting control panel.</p>
|
|
|
|
|
</body>
|
|
|
|
|
</html>
|
|
|
|
|
EOF
|
|
|
|
|
## The old catch-all index.html ("shared-ols") is gone on purpose, and actively
|
|
|
|
|
## removed so an in-place upgrade of a long-lived container cannot leave it
|
|
|
|
|
## behind. If these rewrite rules were ever to stop applying, `context /` would
|
|
|
|
|
## fall back to serving the docRoot index -- with no index.html that is a 403,
|
|
|
|
|
## which is wrong-but-loud, instead of a 200 that is wrong-and-silent.
|
|
|
|
|
rm -f "$HEALTH_DIR/html/index.html"
|
2026-06-10 01:22:14 -07:00
|
|
|
|
2026-06-10 08:34:55 -07:00
|
|
|
## ---- ownership: OLS reads conf/ as lsadm. chown the base conf dir + health dir
|
|
|
|
|
## NON-recursively (the per-site files under conf/shared-sites are written by the
|
|
|
|
|
## panel and are world-readable; a recursive chown here would be O(N-sites) on
|
|
|
|
|
## every container (re)start, delaying first-listen after a crash). The render
|
|
|
|
|
## script chowns the httpd_config.conf it produces. ----
|
|
|
|
|
chown lsadm:nogroup "$LSWS_CONF" "$HEALTH_DIR" "$HEALTH_DIR/html" 2>/dev/null || true
|
2026-08-22 19:44:01 -07:00
|
|
|
chown lsadm:nogroup "$HEALTH_DIR/vhconf.conf" "$HEALTH_DIR/html/healthz" "$HEALTH_DIR/html/misdirected.html" 2>/dev/null || true
|
2026-06-10 08:34:55 -07:00
|
|
|
|
2026-06-10 01:22:14 -07:00
|
|
|
## ---- assemble httpd_config.conf from the panel's per-site files ----
|
|
|
|
|
/scripts/render-shared-ols-config.sh
|
|
|
|
|
|
|
|
|
|
## ---- stream OLS logs to PID-1 stdout (follows across restarts) ----
|
|
|
|
|
mkdir -p /usr/local/lsws/logs
|
|
|
|
|
touch /usr/local/lsws/logs/error.log /usr/local/lsws/logs/access.log
|
|
|
|
|
tail -F /usr/local/lsws/logs/error.log /usr/local/lsws/logs/access.log 2>/dev/null &
|
|
|
|
|
|
|
|
|
|
## ---- .htaccess watcher (required; spec 5.3). Background; the panel monitors
|
|
|
|
|
## that it stays alive (its death silently stops rewrite changes applying). ----
|
|
|
|
|
/scripts/ols-htaccess-watcher.sh &
|
|
|
|
|
WATCHER_PID=$!
|
|
|
|
|
|
|
|
|
|
## ---- supervise OLS in DAEMON mode (verbatim model from entrypoint-litespeed.sh) ----
|
|
|
|
|
STOP_REQUESTED=0
|
|
|
|
|
term_handler() {
|
|
|
|
|
STOP_REQUESTED=1
|
|
|
|
|
kill "$WATCHER_PID" 2>/dev/null || true
|
|
|
|
|
/usr/local/lsws/bin/lswsctrl stop >/dev/null 2>&1 || true
|
|
|
|
|
}
|
|
|
|
|
trap term_handler TERM INT
|
|
|
|
|
|
2026-08-13 21:29:48 -07:00
|
|
|
## NOT `lswsctrl status` (unlike the otherwise-identical function in
|
|
|
|
|
## entrypoint-litespeed.sh). `lswsctrl` appends a timestamped line to
|
|
|
|
|
## logs/lsrestart.log on EVERY invocation it makes, including `status` — and
|
|
|
|
|
## this loop polls every 3s forever. Measured on whp01: lsrestart.log is 96 MB,
|
|
|
|
|
## holding 1,819,286 `status` lines against 2,429 real `restart` lines; at one
|
|
|
|
|
## poll per 3s that's ~63 days of continuous polling, which is exactly the
|
|
|
|
|
## file's age, and it isn't rotated on any host (whp01/whp02/sdbees all growing
|
|
|
|
|
## at ~1.5 MB/day). So: check liveness directly instead of shelling out to a
|
|
|
|
|
## tool whose logging is a side effect we don't want on a fixed timer.
|
|
|
|
|
##
|
|
|
|
|
## Verified (docker run litespeedtech/openlitespeed:1.8.4-lsphp83, the exact
|
|
|
|
|
## base this image is built FROM — see Dockerfile.shared-ols): the running main
|
|
|
|
|
## process shows in `ps` as `openlitespeed (lshttpd - main)`, one PID, always
|
|
|
|
|
## present while OLS is up and absent the instant it is killed (checked via
|
|
|
|
|
## `ps aux` immediately after `kill -9` on the main PID). `pgrep -f` matches
|
|
|
|
|
## against the full command line, and no other process on this image's `ps`
|
|
|
|
|
## output contains that string, so this cannot cross-match an unrelated
|
|
|
|
|
## process. It also cannot self-match: pgrep excludes its own PID by default,
|
|
|
|
|
## and the invoking process here is bash executing this script file, whose own
|
|
|
|
|
## argv never contains the pattern text (only the *source lines* of this script
|
|
|
|
|
## do, which `pgrep -f` never sees).
|
|
|
|
|
##
|
|
|
|
|
## Deliberately NOT the pidfile (/tmp/lshttpd/lshttpd.pid, confirmed present in
|
|
|
|
|
## the same probe): pidfiles are known to go stale across a crash (verified —
|
|
|
|
|
## after `kill -9` the file still held the dead PID), and treating a stale PID
|
|
|
|
|
## as "alive" if the kernel ever reuses that number is a false positive this
|
|
|
|
|
## supervisor cannot afford (see below). `pgrep -f` reads the live process
|
|
|
|
|
## table, so there is no staleness window to reason about.
|
|
|
|
|
##
|
|
|
|
|
## Conservative on both failure directions, which matters because this is a
|
|
|
|
|
## supervisor predicate, not a metric: a false negative makes start_ols() run
|
|
|
|
|
## `lswsctrl start` against an already-running OLS — verified against the same
|
|
|
|
|
## probe base image, that is NOT a no-op, it sends SIGUSR1 to the live main
|
|
|
|
|
## process, i.e. the same graceful self-restart QUIC.cloud IP refreshes trigger
|
|
|
|
|
## (see entrypoint-litespeed.sh's note on that handoff) — a brief, zero-
|
|
|
|
|
## downtime blip at worst. A false positive is worse: it leaves a genuinely
|
|
|
|
|
## dead OLS un-revived until some later poll happens to notice. So if this
|
|
|
|
|
## predicate is ever in doubt it should err toward reporting "not running", not
|
|
|
|
|
## "running".
|
2026-08-05 15:23:56 -07:00
|
|
|
ols_running() {
|
2026-08-13 21:29:48 -07:00
|
|
|
pgrep -f 'lshttpd - main' >/dev/null 2>&1
|
2026-08-05 15:23:56 -07:00
|
|
|
}
|
2026-06-10 01:22:14 -07:00
|
|
|
|
|
|
|
|
MAX_STARTS=5
|
|
|
|
|
WINDOW=60
|
|
|
|
|
starts=""
|
|
|
|
|
|
|
|
|
|
start_ols() {
|
|
|
|
|
/usr/local/lsws/bin/lswsctrl start >/dev/null 2>&1 || true
|
|
|
|
|
for _ in $(seq 1 20); do
|
|
|
|
|
ols_running && return 0
|
|
|
|
|
sleep 0.5
|
|
|
|
|
done
|
|
|
|
|
return 1
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if ! start_ols; then
|
|
|
|
|
echo "entrypoint-shared-ols: OLS failed to start (not running after 10s)." >&2
|
|
|
|
|
exit 1
|
|
|
|
|
fi
|
|
|
|
|
echo "entrypoint-shared-ols: OLS started in daemon mode — $(/usr/local/lsws/bin/lswsctrl status 2>/dev/null || true)"
|
|
|
|
|
|
|
|
|
|
while true; do
|
|
|
|
|
if ols_running; then
|
|
|
|
|
sleep 3
|
|
|
|
|
continue
|
|
|
|
|
fi
|
|
|
|
|
sleep 2
|
|
|
|
|
if [ "$STOP_REQUESTED" -eq 0 ] && ols_running; then
|
|
|
|
|
continue
|
|
|
|
|
fi
|
|
|
|
|
if [ "$STOP_REQUESTED" -eq 1 ]; then
|
|
|
|
|
echo "entrypoint-shared-ols: SIGTERM received, OLS stopped — exiting."
|
|
|
|
|
exit 0
|
|
|
|
|
fi
|
|
|
|
|
now=$(date +%s)
|
|
|
|
|
starts="$starts $now"
|
|
|
|
|
pruned=""
|
|
|
|
|
for t in $starts; do
|
|
|
|
|
[ $((now - t)) -lt "$WINDOW" ] && pruned="$pruned $t"
|
|
|
|
|
done
|
|
|
|
|
starts="$pruned"
|
|
|
|
|
n=$(echo $starts | wc -w)
|
|
|
|
|
echo "entrypoint-shared-ols: OLS not running — relaunching (attempt $n/$MAX_STARTS within ${WINDOW}s)." >&2
|
|
|
|
|
if [ "$n" -ge "$MAX_STARTS" ]; then
|
|
|
|
|
echo "entrypoint-shared-ols: OLS crash-looping — bailing for Docker restart policy / monitor." >&2
|
|
|
|
|
exit 1
|
|
|
|
|
fi
|
|
|
|
|
start_ols || true
|
|
|
|
|
done
|