Compare commits
10
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9d16151120 | ||
|
|
b892438070 | ||
|
|
2a2b9739fc | ||
|
|
7732e2a2ff | ||
|
|
89c74c10cf | ||
|
|
1b557b9931 | ||
|
|
6ced2f8797 | ||
|
|
3917b6d1ae | ||
|
|
d9cc5311de | ||
|
|
f1c1954378 |
@@ -36,11 +36,26 @@ jobs:
|
|||||||
username: shadowdao
|
username: shadowdao
|
||||||
password: ${{ secrets.GHCR_TOKEN }}
|
password: ${{ secrets.GHCR_TOKEN }}
|
||||||
|
|
||||||
|
# Read the human-readable release version from the VERSION file so every
|
||||||
|
# build is pinnable for rollback (alongside the immutable git SHA). Bump
|
||||||
|
# VERSION (YYYY.MM.N) in the same commit as a release-worthy change.
|
||||||
|
- name: Read version
|
||||||
|
id: ver
|
||||||
|
run: echo "version=$(cat VERSION)" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
- name: Build Image
|
- name: Build Image
|
||||||
uses: docker/build-push-action@v6
|
uses: docker/build-push-action@v6
|
||||||
with:
|
with:
|
||||||
platforms: linux/amd64
|
platforms: linux/amd64
|
||||||
push: true
|
push: true
|
||||||
|
build-args: |
|
||||||
|
VERSION=${{ steps.ver.outputs.version }}
|
||||||
|
# Three tags per registry: :latest (moving), :<version> (human-readable
|
||||||
|
# release), :<sha> (immutable, guaranteed-unique rollback target).
|
||||||
tags: |
|
tags: |
|
||||||
repo.anhonesthost.net/cloud-hosting-platform/haproxy-manager-base:latest
|
repo.anhonesthost.net/cloud-hosting-platform/haproxy-manager-base:latest
|
||||||
|
repo.anhonesthost.net/cloud-hosting-platform/haproxy-manager-base:${{ steps.ver.outputs.version }}
|
||||||
|
repo.anhonesthost.net/cloud-hosting-platform/haproxy-manager-base:${{ gitea.sha }}
|
||||||
ghcr.io/shadowdao/haproxy-manager-base:latest
|
ghcr.io/shadowdao/haproxy-manager-base:latest
|
||||||
|
ghcr.io/shadowdao/haproxy-manager-base:${{ steps.ver.outputs.version }}
|
||||||
|
ghcr.io/shadowdao/haproxy-manager-base:${{ gitea.sha }}
|
||||||
|
|||||||
+7
-1
@@ -14,9 +14,13 @@ FROM repo.anhonesthost.net/cloud-hosting-platform/python:3.12-slim
|
|||||||
# sidebar; pointing at the public GitHub mirror enables that linking. The
|
# sidebar; pointing at the public GitHub mirror enables that linking. The
|
||||||
# canonical source-of-truth git remote is still Gitea, but Gitea's registry
|
# canonical source-of-truth git remote is still Gitea, but Gitea's registry
|
||||||
# doesn't consume this label, so there's no contention.
|
# doesn't consume this label, so there's no contention.
|
||||||
|
# Stamped from the VERSION file by CI (build-arg) so `docker inspect` reports
|
||||||
|
# what's running on any host. Defaults to "dev" for local/manual builds.
|
||||||
|
ARG VERSION=dev
|
||||||
LABEL org.opencontainers.image.title="haproxy-manager-base" \
|
LABEL org.opencontainers.image.title="haproxy-manager-base" \
|
||||||
org.opencontainers.image.description="HAProxy management API with Let's Encrypt automation, Coraza WAF integration, and template-driven config" \
|
org.opencontainers.image.description="HAProxy management API with Let's Encrypt automation, Coraza WAF integration, and template-driven config" \
|
||||||
org.opencontainers.image.source="https://github.com/shadowdao/haproxy-manager-base" \
|
org.opencontainers.image.source="https://github.com/shadowdao/haproxy-manager-base" \
|
||||||
|
org.opencontainers.image.version="${VERSION}" \
|
||||||
org.opencontainers.image.licenses="MIT"
|
org.opencontainers.image.licenses="MIT"
|
||||||
|
|
||||||
RUN apt update -y && apt dist-upgrade -y && apt install socat haproxy cron certbot curl jq net-tools -y && apt clean && rm -rf /var/lib/apt/lists/*
|
RUN apt update -y && apt dist-upgrade -y && apt install socat haproxy cron certbot curl jq net-tools -y && apt clean && rm -rf /var/lib/apt/lists/*
|
||||||
@@ -43,7 +47,9 @@ RUN mkdir -p /var/spool/cron/crontabs && \
|
|||||||
echo '0 */12 * * * /haproxy/scripts/renew-certificates.sh >> /var/log/haproxy-manager.log 2>&1' >> /var/spool/cron/crontabs/root && \
|
echo '0 */12 * * * /haproxy/scripts/renew-certificates.sh >> /var/log/haproxy-manager.log 2>&1' >> /var/spool/cron/crontabs/root && \
|
||||||
chmod 600 /var/spool/cron/crontabs/root && \
|
chmod 600 /var/spool/cron/crontabs/root && \
|
||||||
chown root:crontab /var/spool/cron/crontabs/root
|
chown root:crontab /var/spool/cron/crontabs/root
|
||||||
EXPOSE 80 443 8000
|
# 443/udp carries HTTP/3 (QUIC). EXPOSE is documentation only — the container
|
||||||
|
# must still be run with `-p 443:443/udp` for the UDP listener to be reachable.
|
||||||
|
EXPOSE 80 443 443/udp 8000
|
||||||
# Add health check
|
# Add health check
|
||||||
HEALTHCHECK --interval=30s --timeout=10s --start-period=60s --retries=3 \
|
HEALTHCHECK --interval=30s --timeout=10s --start-period=60s --retries=3 \
|
||||||
CMD curl -sf --max-time 5 http://localhost:8000/health && curl -s --max-time 5 -o /dev/null http://localhost/ || exit 1
|
CMD curl -sf --max-time 5 http://localhost:8000/health && curl -s --max-time 5 -o /dev/null http://localhost/ || exit 1
|
||||||
|
|||||||
@@ -6,10 +6,10 @@ A Flask-based API service for managing HAProxy configurations with dynamic SSL c
|
|||||||
To run the container:
|
To run the container:
|
||||||
```bash
|
```bash
|
||||||
# Without API key authentication (default)
|
# Without API key authentication (default)
|
||||||
docker run -d -p 80:80 -p 443:443 -p 8000:8000 -v lets-encrypt:/etc/letsencrypt -v haproxy:/etc/haproxy --name haproxy-manager your-registry.example.com/cloud-hosting-platform/haproxy-manager-base:latest
|
docker run -d -p 80:80 -p 443:443 -p 443:443/udp -p 8000:8000 -v lets-encrypt:/etc/letsencrypt -v haproxy:/etc/haproxy --name haproxy-manager your-registry.example.com/cloud-hosting-platform/haproxy-manager-base:latest
|
||||||
|
|
||||||
# With API key authentication (recommended for production)
|
# With API key authentication (recommended for production)
|
||||||
docker run -d -p 80:80 -p 443:443 -p 8000:8000 -v lets-encrypt:/etc/letsencrypt -v haproxy:/etc/haproxy -e HAPROXY_API_KEY=your-secure-api-key-here --name haproxy-manager your-registry.example.com/cloud-hosting-platform/haproxy-manager-base:latest
|
docker run -d -p 80:80 -p 443:443 -p 443:443/udp -p 8000:8000 -v lets-encrypt:/etc/letsencrypt -v haproxy:/etc/haproxy -e HAPROXY_API_KEY=your-secure-api-key-here --name haproxy-manager your-registry.example.com/cloud-hosting-platform/haproxy-manager-base:latest
|
||||||
```
|
```
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
@@ -394,7 +394,7 @@ You can customize the default page by setting environment variables:
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
docker run -d \
|
docker run -d \
|
||||||
-p 80:80 -p 443:443 -p 8000:8000 \
|
-p 80:80 -p 443:443 -p 443:443/udp -p 8000:8000 \
|
||||||
-v lets-encrypt:/etc/letsencrypt \
|
-v lets-encrypt:/etc/letsencrypt \
|
||||||
-v haproxy:/etc/haproxy \
|
-v haproxy:/etc/haproxy \
|
||||||
-e HAPROXY_API_KEY=your-secure-api-key-here \
|
-e HAPROXY_API_KEY=your-secure-api-key-here \
|
||||||
@@ -411,7 +411,7 @@ docker run -d \
|
|||||||
```bash
|
```bash
|
||||||
# Start container with API key
|
# Start container with API key
|
||||||
docker run -d \
|
docker run -d \
|
||||||
-p 80:80 -p 443:443 -p 8000:8000 \
|
-p 80:80 -p 443:443 -p 443:443/udp -p 8000:8000 \
|
||||||
-v lets-encrypt:/etc/letsencrypt \
|
-v lets-encrypt:/etc/letsencrypt \
|
||||||
-v haproxy:/etc/haproxy \
|
-v haproxy:/etc/haproxy \
|
||||||
-e HAPROXY_API_KEY=your-secure-api-key-here \
|
-e HAPROXY_API_KEY=your-secure-api-key-here \
|
||||||
|
|||||||
+419
-62
@@ -12,12 +12,45 @@ from datetime import datetime, timedelta
|
|||||||
import json
|
import json
|
||||||
import ipaddress
|
import ipaddress
|
||||||
import shutil
|
import shutil
|
||||||
|
import stat
|
||||||
import tempfile
|
import tempfile
|
||||||
import threading
|
import threading
|
||||||
import time
|
import time
|
||||||
import re
|
import re
|
||||||
import fcntl
|
import fcntl
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Bounded subprocess execution (incident 2026-07-07)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Every external command this manager runs — certbot ACME issuance/renewal,
|
||||||
|
# `socat` reloads over the haproxy admin socket, `haproxy -c` validation — is a
|
||||||
|
# potential hang. The management API runs under gunicorn gthread workers, and a
|
||||||
|
# subprocess.run() with NO timeout blocks its worker thread forever if the
|
||||||
|
# command stalls (e.g. an ACME/upstream that stops responding mid-read).
|
||||||
|
# gunicorn's --timeout does not rescue this: for gthread it only kills a worker
|
||||||
|
# whose *main* thread stops heart-beating, but the main thread keeps polling
|
||||||
|
# while pool threads are wedged. Enough stalled calls exhaust the 4-thread pool
|
||||||
|
# and the whole API stops responding — "healthy" health-check, every request
|
||||||
|
# 30s-timeouts — which is exactly what stalled WHP site updates on 2026-07-07.
|
||||||
|
#
|
||||||
|
# Fix: give EVERY subprocess.run() a default timeout unless the caller passes
|
||||||
|
# one explicitly. On expiry Python kills the child and raises
|
||||||
|
# subprocess.TimeoutExpired (a subclass of Exception); the existing per-endpoint
|
||||||
|
# try/except turns that into a clean error AND releases the worker thread.
|
||||||
|
# Bounding by default (instead of editing ~30 call sites) means no site can be
|
||||||
|
# missed and any future call is protected automatically.
|
||||||
|
DEFAULT_SUBPROCESS_TIMEOUT = int(os.environ.get('HAPROXY_MGR_SUBPROCESS_TIMEOUT', '180'))
|
||||||
|
_unbounded_subprocess_run = subprocess.run
|
||||||
|
|
||||||
|
|
||||||
|
def _bounded_subprocess_run(*args, **kwargs):
|
||||||
|
if kwargs.get('timeout') is None:
|
||||||
|
kwargs['timeout'] = DEFAULT_SUBPROCESS_TIMEOUT
|
||||||
|
return _unbounded_subprocess_run(*args, **kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
subprocess.run = _bounded_subprocess_run
|
||||||
|
|
||||||
app = Flask(__name__)
|
app = Flask(__name__)
|
||||||
|
|
||||||
# Default page server (port 8080) — served to HAProxy clients whose request hit
|
# Default page server (port 8080) — served to HAProxy clients whose request hit
|
||||||
@@ -73,8 +106,18 @@ HAPROXY_CONFIG_PATH = '/etc/haproxy/haproxy.cfg'
|
|||||||
HAPROXY_BACKUP_PATH = '/etc/haproxy/haproxy.cfg.backup'
|
HAPROXY_BACKUP_PATH = '/etc/haproxy/haproxy.cfg.backup'
|
||||||
BLOCKED_IPS_MAP_PATH = '/etc/haproxy/blocked_ips.map'
|
BLOCKED_IPS_MAP_PATH = '/etc/haproxy/blocked_ips.map'
|
||||||
BLOCKED_IPS_MAP_BACKUP_PATH = '/etc/haproxy/blocked_ips.map.backup'
|
BLOCKED_IPS_MAP_BACKUP_PATH = '/etc/haproxy/blocked_ips.map.backup'
|
||||||
|
# Coraza SPOE engine file. `haproxy -c` parses this too (the frontend's
|
||||||
|
# `filter spoe engine coraza config <path>` line points at it), so it is part
|
||||||
|
# of the same restorable config set as haproxy.cfg — rolling back haproxy.cfg
|
||||||
|
# while leaving a broken coraza-spoe.cfg behind still fails validation.
|
||||||
|
CORAZA_SPOE_CONFIG_PATH = '/etc/haproxy/coraza-spoe.cfg'
|
||||||
|
CORAZA_SPOE_BACKUP_PATH = '/etc/haproxy/coraza-spoe.cfg.backup'
|
||||||
HAPROXY_SOCKET_PATH = '/var/run/haproxy.sock'
|
HAPROXY_SOCKET_PATH = '/var/run/haproxy.sock'
|
||||||
SSL_CERTS_DIR = '/etc/haproxy/certs'
|
SSL_CERTS_DIR = '/etc/haproxy/certs'
|
||||||
|
# Stable per-host secret for QUIC Retry/address-validation tokens. Lives in the
|
||||||
|
# /etc/haproxy named volume so it survives container recreates; self-healed on
|
||||||
|
# first config render. See get_or_create_cluster_secret().
|
||||||
|
CLUSTER_SECRET_PATH = '/etc/haproxy/cluster-secret'
|
||||||
API_KEY = os.environ.get('HAPROXY_API_KEY') # Optional API key for authentication
|
API_KEY = os.environ.get('HAPROXY_API_KEY') # Optional API key for authentication
|
||||||
|
|
||||||
# Setup logging
|
# Setup logging
|
||||||
@@ -807,10 +850,12 @@ def renew_certificates():
|
|||||||
# Defensive: clear any stale lock left by a SIGKILLed prior run.
|
# Defensive: clear any stale lock left by a SIGKILLed prior run.
|
||||||
clear_stale_certbot_locks()
|
clear_stale_certbot_locks()
|
||||||
|
|
||||||
# Run certbot renew
|
# Run certbot renew. Explicit long timeout (overrides the module
|
||||||
|
# default): `renew` walks every lineage and can legitimately make many
|
||||||
|
# ACME round-trips when several certs are actually due.
|
||||||
result = subprocess.run([
|
result = subprocess.run([
|
||||||
'certbot', 'renew', '--quiet'
|
'certbot', 'renew', '--quiet'
|
||||||
], capture_output=True, text=True)
|
], capture_output=True, text=True, timeout=900)
|
||||||
|
|
||||||
if result.returncode == 0:
|
if result.returncode == 0:
|
||||||
# Check if any certificates were renewed
|
# Check if any certificates were renewed
|
||||||
@@ -1687,6 +1732,45 @@ def dns_challenge_verify():
|
|||||||
log_operation('dns_challenge_verify', False, str(e))
|
log_operation('dns_challenge_verify', False, str(e))
|
||||||
return jsonify({'success': False, 'error': str(e)}), 500
|
return jsonify({'success': False, 'error': str(e)}), 500
|
||||||
|
|
||||||
|
def get_or_create_cluster_secret():
|
||||||
|
"""Return a stable secret for QUIC token derivation, generating it once.
|
||||||
|
|
||||||
|
HAProxy uses `cluster-secret` to key QUIC Retry/address-validation tokens.
|
||||||
|
Without a stable value it picks a random one each (re)start and logs a
|
||||||
|
notice; tokens then don't survive reloads. We persist one in the
|
||||||
|
/etc/haproxy named volume so it's stable across container recreates.
|
||||||
|
Exclusive-create avoids a race if two renders run concurrently. Failure to
|
||||||
|
read/write is non-fatal: we fall back to an empty string and the template
|
||||||
|
simply omits the directive (HAProxy reverts to its random-per-process
|
||||||
|
behaviour), so QUIC still works.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
if os.path.exists(CLUSTER_SECRET_PATH):
|
||||||
|
with open(CLUSTER_SECRET_PATH, 'r') as f:
|
||||||
|
secret = f.read().strip()
|
||||||
|
if secret:
|
||||||
|
return secret
|
||||||
|
# Generate and persist exclusively (0600). hex => config-safe charset.
|
||||||
|
secret = os.urandom(32).hex()
|
||||||
|
fd = os.open(CLUSTER_SECRET_PATH, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||||
|
try:
|
||||||
|
os.write(fd, secret.encode())
|
||||||
|
finally:
|
||||||
|
os.close(fd)
|
||||||
|
logger.info("Generated new QUIC cluster-secret at %s", CLUSTER_SECRET_PATH)
|
||||||
|
return secret
|
||||||
|
except FileExistsError:
|
||||||
|
# Lost the create race — another render just wrote it; read it back.
|
||||||
|
try:
|
||||||
|
with open(CLUSTER_SECRET_PATH, 'r') as f:
|
||||||
|
return f.read().strip()
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("Failed to read cluster-secret after race: %s", e)
|
||||||
|
return ''
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("Failed to get/create cluster-secret: %s", e)
|
||||||
|
return ''
|
||||||
|
|
||||||
def generate_config():
|
def generate_config():
|
||||||
try:
|
try:
|
||||||
conn = sqlite3.connect(DB_FILE)
|
conn = sqlite3.connect(DB_FILE)
|
||||||
@@ -1718,6 +1802,21 @@ def generate_config():
|
|||||||
|
|
||||||
config_parts = []
|
config_parts = []
|
||||||
|
|
||||||
|
# Snapshot the last-known-good config BEFORE anything below touches a
|
||||||
|
# file in /etc/haproxy. Everything this function writes (haproxy.cfg,
|
||||||
|
# blocked_ips.map, coraza-spoe.cfg) is validated as one set by
|
||||||
|
# `haproxy -c`, so the rollback point has to predate the first of them.
|
||||||
|
# Taking it here (rather than inside reload_haproxy_safely(), which runs
|
||||||
|
# after the writes) is what makes rollback real - see create_backup().
|
||||||
|
backup_ok, backup_status = create_backup()
|
||||||
|
if not backup_ok:
|
||||||
|
# Could not even attempt a snapshot (I/O error). Writing a new
|
||||||
|
# config now would leave us with no way back, so refuse.
|
||||||
|
raise Exception(
|
||||||
|
"Refusing to regenerate config: failed to back up the current "
|
||||||
|
"configuration, so a failed change could not be rolled back"
|
||||||
|
)
|
||||||
|
|
||||||
# Optional Coraza WAF integration. When HAPROXY_CORAZA_SPOE_BACKEND is
|
# Optional Coraza WAF integration. When HAPROXY_CORAZA_SPOE_BACKEND is
|
||||||
# set on the haproxy-manager container, we render an extra TCP backend
|
# set on the haproxy-manager container, we render an extra TCP backend
|
||||||
# pointing at a coraza-spoa sidecar AND inject a `filter spoe ...` line
|
# pointing at a coraza-spoa sidecar AND inject a `filter spoe ...` line
|
||||||
@@ -1749,7 +1848,9 @@ def generate_config():
|
|||||||
logger.error(f"Failed to create {suspended_list_path}: {e}")
|
logger.error(f"Failed to create {suspended_list_path}: {e}")
|
||||||
|
|
||||||
# Add Haproxy Default Headers
|
# Add Haproxy Default Headers
|
||||||
default_headers = template_env.get_template('hap_header.tpl').render()
|
default_headers = template_env.get_template('hap_header.tpl').render(
|
||||||
|
cluster_secret = get_or_create_cluster_secret(),
|
||||||
|
)
|
||||||
config_parts.append(default_headers)
|
config_parts.append(default_headers)
|
||||||
|
|
||||||
# Update blocked IPs map file first
|
# Update blocked IPs map file first
|
||||||
@@ -1810,7 +1911,11 @@ def generate_config():
|
|||||||
# First pass: exact domain ACLs (higher priority - evaluated first)
|
# First pass: exact domain ACLs (higher priority - evaluated first)
|
||||||
for domain in exact_domains:
|
for domain in exact_domains:
|
||||||
if not domain['backend_name']:
|
if not domain['backend_name']:
|
||||||
logger.warning(f"Skipping domain {domain['domain']} - no backend name")
|
# Expected for domains registered without a proxy backend (e.g. the
|
||||||
|
# panel's own hostname, present only for certificate management).
|
||||||
|
# Log at INFO — not WARNING — so it doesn't trip log monitors as an
|
||||||
|
# error; it recurs on every generate_config by design.
|
||||||
|
logger.info(f"Skipping domain {domain['domain']} - no proxy backend (cert/management-only)")
|
||||||
continue
|
continue
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -1829,7 +1934,8 @@ def generate_config():
|
|||||||
# Second pass: wildcard domain ACLs (lower priority - evaluated after exact matches)
|
# Second pass: wildcard domain ACLs (lower priority - evaluated after exact matches)
|
||||||
for domain in wildcard_domains:
|
for domain in wildcard_domains:
|
||||||
if not domain['backend_name']:
|
if not domain['backend_name']:
|
||||||
logger.warning(f"Skipping wildcard domain {domain['domain']} - no backend name")
|
# See note above — INFO, not WARNING; expected for cert/management-only domains.
|
||||||
|
logger.info(f"Skipping wildcard domain {domain['domain']} - no proxy backend (cert/management-only)")
|
||||||
continue
|
continue
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -1900,25 +2006,21 @@ backend default-backend
|
|||||||
# how the file was authored.
|
# how the file was authored.
|
||||||
if not coraza_spoe_cfg.endswith('\n'):
|
if not coraza_spoe_cfg.endswith('\n'):
|
||||||
coraza_spoe_cfg += '\n'
|
coraza_spoe_cfg += '\n'
|
||||||
coraza_spoe_path = '/etc/haproxy/coraza-spoe.cfg'
|
write_config_atomically(CORAZA_SPOE_CONFIG_PATH, coraza_spoe_cfg)
|
||||||
with open(coraza_spoe_path, 'w') as f:
|
logger.info(f"Coraza SPOE engine config written to "
|
||||||
f.write(coraza_spoe_cfg)
|
f"{CORAZA_SPOE_CONFIG_PATH} "
|
||||||
logger.info(f"Coraza SPOE engine config written to {coraza_spoe_path} "
|
|
||||||
f"(SPOA target: {coraza_spoe_backend})")
|
f"(SPOA target: {coraza_spoe_backend})")
|
||||||
|
|
||||||
# Write complete configuration to tmp
|
|
||||||
temp_config_path = "/etc/haproxy/haproxy.cfg"
|
|
||||||
|
|
||||||
config_content = '\n'.join(config_parts)
|
config_content = '\n'.join(config_parts)
|
||||||
logger.debug("Generated HAProxy configuration")
|
logger.debug("Generated HAProxy configuration")
|
||||||
|
|
||||||
# Write complete configuration to tmp
|
# Write new configuration to file (atomically - a truncated haproxy.cfg
|
||||||
# Write new configuration to file
|
# is as fatal as an invalid one). The rollback point was taken above,
|
||||||
with open(HAPROXY_CONFIG_PATH, 'w') as f:
|
# before this write.
|
||||||
f.write(config_content)
|
write_config_atomically(HAPROXY_CONFIG_PATH, config_content)
|
||||||
|
|
||||||
# Use safe reload with validation and rollback
|
# Use safe reload with validation and rollback
|
||||||
success, message = reload_haproxy_safely()
|
success, message = reload_haproxy_safely(backup_status=backup_status)
|
||||||
if success:
|
if success:
|
||||||
logger.info("Configuration generated and HAProxy reloaded safely")
|
logger.info("Configuration generated and HAProxy reloaded safely")
|
||||||
log_operation('generate_config', True, 'Configuration generated and HAProxy reloaded safely')
|
log_operation('generate_config', True, 'Configuration generated and HAProxy reloaded safely')
|
||||||
@@ -1935,61 +2037,304 @@ backend default-backend
|
|||||||
traceback.print_exc()
|
traceback.print_exc()
|
||||||
raise
|
raise
|
||||||
|
|
||||||
def create_backup():
|
# ---------------------------------------------------------------------------
|
||||||
"""Create backup of current config and map files"""
|
# Config backup / rollback
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Rollback only works if the backup predates the write it is supposed to undo.
|
||||||
|
# Until 2026-08 create_backup() ran from inside reload_haproxy_safely(), i.e.
|
||||||
|
# AFTER generate_config() had already overwritten haproxy.cfg — so the "backup"
|
||||||
|
# was a copy of the new (possibly broken) config and restore_backup() restored
|
||||||
|
# the same broken bytes. The advertised rollback was a no-op and a fatal
|
||||||
|
# haproxy.cfg persisted on disk, where start_haproxy() refuses to launch (the
|
||||||
|
# June 2026 missing-template incident). create_backup() must now be called by
|
||||||
|
# the writer, BEFORE the first byte is written.
|
||||||
|
|
||||||
|
# Statuses returned by create_backup() that mean a rollback target exists.
|
||||||
|
_ROLLBACK_AVAILABLE_STATUSES = ('created', 'kept_previous')
|
||||||
|
|
||||||
|
|
||||||
|
def _files_identical(path_a, path_b):
|
||||||
|
"""Byte-compare two files.
|
||||||
|
|
||||||
|
Deliberately not filecmp.cmp(): it memoises on (size, mtime), and
|
||||||
|
shutil.copy2() preserves mtime, so a stale cache entry could report a
|
||||||
|
changed config as unchanged. These files are small; read them.
|
||||||
|
"""
|
||||||
try:
|
try:
|
||||||
if os.path.exists(HAPROXY_CONFIG_PATH):
|
if os.path.getsize(path_a) != os.path.getsize(path_b):
|
||||||
shutil.copy2(HAPROXY_CONFIG_PATH, HAPROXY_BACKUP_PATH)
|
return False
|
||||||
if os.path.exists(BLOCKED_IPS_MAP_PATH):
|
with open(path_a, 'rb') as fa, open(path_b, 'rb') as fb:
|
||||||
shutil.copy2(BLOCKED_IPS_MAP_PATH, BLOCKED_IPS_MAP_BACKUP_PATH)
|
while True:
|
||||||
logger.info("Backups created successfully")
|
chunk_a = fa.read(65536)
|
||||||
return True
|
chunk_b = fb.read(65536)
|
||||||
|
if chunk_a != chunk_b:
|
||||||
|
return False
|
||||||
|
if not chunk_a:
|
||||||
|
return True
|
||||||
|
except OSError:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _config_set_matches_backup():
|
||||||
|
"""True if every live config file is byte-identical to its backup copy.
|
||||||
|
|
||||||
|
After a successful reload the live set has already been recorded as
|
||||||
|
known-good (see promote_current_config_to_backup()), which is the common
|
||||||
|
case at the start of the next generation. Recognising it lets create_backup()
|
||||||
|
skip both the re-validation and the copy - worth doing because
|
||||||
|
`haproxy -c` on an edge with hundreds of certificates is not free and
|
||||||
|
generate_config() runs synchronously inside customer-facing API calls.
|
||||||
|
"""
|
||||||
|
for live_path, backup_path in _config_backup_pairs():
|
||||||
|
if os.path.exists(live_path) != os.path.exists(backup_path):
|
||||||
|
return False
|
||||||
|
if (os.path.exists(live_path)
|
||||||
|
and not _files_identical(live_path, backup_path)):
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def _config_backup_pairs():
|
||||||
|
"""(live, backup) pairs forming one restorable config set.
|
||||||
|
|
||||||
|
Built at call time rather than at import so the module-level path constants
|
||||||
|
stay patchable (tests, alternate deployments).
|
||||||
|
"""
|
||||||
|
return (
|
||||||
|
(HAPROXY_CONFIG_PATH, HAPROXY_BACKUP_PATH),
|
||||||
|
(BLOCKED_IPS_MAP_PATH, BLOCKED_IPS_MAP_BACKUP_PATH),
|
||||||
|
(CORAZA_SPOE_CONFIG_PATH, CORAZA_SPOE_BACKUP_PATH),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def write_config_atomically(path, content):
|
||||||
|
"""Write content to path via temp file + rename.
|
||||||
|
|
||||||
|
A half-written haproxy.cfg (disk full, container killed mid-write) is just
|
||||||
|
as fatal as an invalid one and is invisible to the caller. os.replace() is
|
||||||
|
atomic within a filesystem, so the file on disk is always either the whole
|
||||||
|
old config or the whole new one — never a truncated hybrid. This also keeps
|
||||||
|
the "existing config is already broken" case from being self-inflicted.
|
||||||
|
"""
|
||||||
|
directory = os.path.dirname(path) or '.'
|
||||||
|
# Preserve the mode of the file we are replacing; mkstemp defaults to 0600
|
||||||
|
# and HAProxy config files are conventionally 0644.
|
||||||
|
try:
|
||||||
|
mode = stat.S_IMODE(os.stat(path).st_mode)
|
||||||
|
except OSError:
|
||||||
|
mode = 0o644
|
||||||
|
fd, tmp_path = tempfile.mkstemp(
|
||||||
|
dir=directory, prefix=os.path.basename(path) + '.', suffix='.tmp'
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
with os.fdopen(fd, 'w') as f:
|
||||||
|
f.write(content)
|
||||||
|
f.flush()
|
||||||
|
os.fsync(f.fileno())
|
||||||
|
os.chmod(tmp_path, mode)
|
||||||
|
os.replace(tmp_path, path)
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
os.unlink(tmp_path)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
raise
|
||||||
|
|
||||||
|
|
||||||
|
def create_backup(require_valid=True):
|
||||||
|
"""Snapshot the CURRENT on-disk config set as the rollback point.
|
||||||
|
|
||||||
|
MUST be called BEFORE the new configuration is written — see the module
|
||||||
|
comment above. Calling it afterwards silently disarms rollback.
|
||||||
|
|
||||||
|
require_valid=True (default) refuses to promote a config that HAProxy
|
||||||
|
already rejects. Backing up a broken config would make "rollback" mean
|
||||||
|
"restore a different broken config"; keeping the older, validated backup
|
||||||
|
instead means a rollback always lands on something HAProxy will actually
|
||||||
|
start with. Cost is one `haproxy -c` run per config generation.
|
||||||
|
|
||||||
|
Returns (ok, status):
|
||||||
|
ok=False, status='error' - the copy itself failed; caller decides.
|
||||||
|
status='created' - backup now holds the current config.
|
||||||
|
status='kept_previous' - current config missing or invalid; the
|
||||||
|
existing (older, good) backup was kept.
|
||||||
|
status='unavailable' - nothing to roll back to at all (first
|
||||||
|
run, or broken config and no prior
|
||||||
|
backup). Rollback is NOT possible.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
snapshot_ok = True
|
||||||
|
reason = None
|
||||||
|
|
||||||
|
if not os.path.exists(HAPROXY_CONFIG_PATH):
|
||||||
|
snapshot_ok = False
|
||||||
|
reason = 'no existing HAProxy config on disk (first run?)'
|
||||||
|
elif _config_set_matches_backup():
|
||||||
|
# The backup already IS the current config, recorded when it last
|
||||||
|
# loaded successfully. Nothing to copy and nothing to re-validate.
|
||||||
|
logger.debug("Config backup already matches the live config")
|
||||||
|
return True, 'created'
|
||||||
|
elif require_valid:
|
||||||
|
status, msg = validate_config_file(HAPROXY_CONFIG_PATH)
|
||||||
|
if status == 'invalid':
|
||||||
|
snapshot_ok = False
|
||||||
|
reason = f'current config on disk does not validate: {msg}'
|
||||||
|
elif status == 'unavailable':
|
||||||
|
# The validator itself could not run (no haproxy binary, etc).
|
||||||
|
# That is NOT evidence the config is bad, and refusing to back
|
||||||
|
# up would leave us with no rollback target at all, so fall
|
||||||
|
# back to last-written semantics and say so loudly.
|
||||||
|
logger.warning(
|
||||||
|
f"Could not verify current config before backup ({msg}); "
|
||||||
|
"backing it up unverified"
|
||||||
|
)
|
||||||
|
|
||||||
|
if not snapshot_ok:
|
||||||
|
if os.path.exists(HAPROXY_BACKUP_PATH):
|
||||||
|
logger.warning(
|
||||||
|
f"Not refreshing config backup: {reason}. Keeping the "
|
||||||
|
f"existing backup at {HAPROXY_BACKUP_PATH} as the rollback "
|
||||||
|
"target."
|
||||||
|
)
|
||||||
|
return True, 'kept_previous'
|
||||||
|
logger.error(
|
||||||
|
f"No config backup could be taken: {reason}, and no previous "
|
||||||
|
f"backup exists at {HAPROXY_BACKUP_PATH}. ROLLBACK IS NOT "
|
||||||
|
"AVAILABLE for this configuration change."
|
||||||
|
)
|
||||||
|
return True, 'unavailable'
|
||||||
|
|
||||||
|
for live_path, backup_path in _config_backup_pairs():
|
||||||
|
if os.path.exists(live_path):
|
||||||
|
shutil.copy2(live_path, backup_path)
|
||||||
|
logger.info("Backup of last-known-good config created successfully")
|
||||||
|
return True, 'created'
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(f"Failed to create backup: {e}")
|
logger.error(f"Failed to create backup: {e}")
|
||||||
return False
|
return False, 'error'
|
||||||
|
|
||||||
def restore_backup():
|
def promote_current_config_to_backup():
|
||||||
"""Restore from backup files"""
|
"""Record the live config as the known-good rollback target.
|
||||||
|
|
||||||
|
Called ONLY after the config has both validated and been loaded by HAProxy,
|
||||||
|
so "backup" really means "the last configuration this box was running".
|
||||||
|
Must never be called before a reload attempt: doing so would make the
|
||||||
|
backup a copy of the config we may still have to roll back from - the same
|
||||||
|
class of bug as backing up after the write.
|
||||||
|
|
||||||
|
Without this, a box whose very first generation succeeded has no rollback
|
||||||
|
target at all until its second successful generation, and any corruption of
|
||||||
|
haproxy.cfg in between leaves nothing to recover to.
|
||||||
|
"""
|
||||||
try:
|
try:
|
||||||
if os.path.exists(HAPROXY_BACKUP_PATH):
|
for live_path, backup_path in _config_backup_pairs():
|
||||||
shutil.copy2(HAPROXY_BACKUP_PATH, HAPROXY_CONFIG_PATH)
|
if os.path.exists(live_path):
|
||||||
if os.path.exists(BLOCKED_IPS_MAP_BACKUP_PATH):
|
shutil.copy2(live_path, backup_path)
|
||||||
shutil.copy2(BLOCKED_IPS_MAP_BACKUP_PATH, BLOCKED_IPS_MAP_PATH)
|
logger.debug("Known-good config backup updated after successful reload")
|
||||||
logger.info("Backups restored successfully")
|
|
||||||
return True
|
return True
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(f"Failed to restore backup: {e}")
|
# Non-fatal: the config is live and working, we just failed to record
|
||||||
|
# it. Loud, because the next change now has a staler rollback target.
|
||||||
|
logger.error(f"Failed to record known-good config backup: {e}")
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def validate_haproxy_config():
|
|
||||||
"""Validate HAProxy configuration file"""
|
|
||||||
try:
|
|
||||||
result = subprocess.run(['haproxy', '-c', '-f', HAPROXY_CONFIG_PATH],
|
|
||||||
capture_output=True, text=True)
|
|
||||||
if result.returncode == 0:
|
|
||||||
logger.info("HAProxy configuration validation passed")
|
|
||||||
return True, None
|
|
||||||
else:
|
|
||||||
error_msg = f"HAProxy configuration validation failed: {result.stderr}"
|
|
||||||
logger.error(error_msg)
|
|
||||||
return False, error_msg
|
|
||||||
except Exception as e:
|
|
||||||
error_msg = f"Error validating HAProxy config: {e}"
|
|
||||||
logger.error(error_msg)
|
|
||||||
return False, error_msg
|
|
||||||
|
|
||||||
def reload_haproxy_safely():
|
def restore_backup():
|
||||||
"""Safely reload HAProxy with validation and rollback"""
|
"""Restore the backed-up config set over the live files.
|
||||||
|
|
||||||
|
Returns (restored, message). restored=False means NOTHING was rolled back
|
||||||
|
and the live config is still whatever the failed change left on disk —
|
||||||
|
callers MUST surface that difference, it is the difference between "we
|
||||||
|
recovered" and "this edge is sitting on a config HAProxy will not load".
|
||||||
|
"""
|
||||||
|
if not os.path.exists(HAPROXY_BACKUP_PATH):
|
||||||
|
msg = (f"No config backup at {HAPROXY_BACKUP_PATH} - cannot roll back; "
|
||||||
|
f"{HAPROXY_CONFIG_PATH} still holds the failed configuration")
|
||||||
|
logger.critical(msg)
|
||||||
|
return False, msg
|
||||||
try:
|
try:
|
||||||
# Create backup before changes
|
for live_path, backup_path in _config_backup_pairs():
|
||||||
if not create_backup():
|
if os.path.exists(backup_path):
|
||||||
return False, "Failed to create backup"
|
shutil.copy2(backup_path, live_path)
|
||||||
|
msg = f"Configuration restored from backup ({HAPROXY_BACKUP_PATH})"
|
||||||
|
logger.info(msg)
|
||||||
|
return True, msg
|
||||||
|
except Exception as e:
|
||||||
|
msg = (f"Failed to restore backup: {e} - {HAPROXY_CONFIG_PATH} may hold "
|
||||||
|
"a broken configuration")
|
||||||
|
logger.critical(msg)
|
||||||
|
return False, msg
|
||||||
|
|
||||||
|
|
||||||
|
def validate_config_file(config_path):
|
||||||
|
"""Run `haproxy -c` against config_path.
|
||||||
|
|
||||||
|
Returns (status, message) with status one of:
|
||||||
|
'valid' - HAProxy parsed the file successfully
|
||||||
|
'invalid' - HAProxy rejected it (message carries stderr)
|
||||||
|
'unavailable' - the validator could not be run at all (binary missing,
|
||||||
|
timeout, ...). Deliberately distinct from 'invalid':
|
||||||
|
it tells us nothing about the config.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
result = subprocess.run(['haproxy', '-c', '-f', config_path],
|
||||||
|
capture_output=True, text=True)
|
||||||
|
except Exception as e:
|
||||||
|
return 'unavailable', f"Error validating HAProxy config: {e}"
|
||||||
|
if result.returncode == 0:
|
||||||
|
return 'valid', None
|
||||||
|
return 'invalid', f"HAProxy configuration validation failed: {result.stderr}"
|
||||||
|
|
||||||
|
|
||||||
|
def validate_haproxy_config():
|
||||||
|
"""Validate the live HAProxy configuration file. Returns (is_valid, error)."""
|
||||||
|
status, message = validate_config_file(HAPROXY_CONFIG_PATH)
|
||||||
|
if status == 'valid':
|
||||||
|
logger.info("HAProxy configuration validation passed")
|
||||||
|
return True, None
|
||||||
|
logger.error(message)
|
||||||
|
return False, message
|
||||||
|
|
||||||
|
def reload_haproxy_safely(backup_status=None):
|
||||||
|
"""Safely reload HAProxy with validation and rollback.
|
||||||
|
|
||||||
|
PRECONDITION: the caller must already have called create_backup() BEFORE
|
||||||
|
writing the new config, and pass the status it returned. This function runs
|
||||||
|
after the new config is on disk, so it cannot take a meaningful backup
|
||||||
|
itself — doing so is exactly the bug this contract exists to prevent.
|
||||||
|
|
||||||
|
backup_status=None means the caller did not take a pre-write backup. We do
|
||||||
|
NOT create one here (that would overwrite a genuinely good backup with the
|
||||||
|
unverified new config); we log it and fall back to whatever backup already
|
||||||
|
exists on disk.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
if backup_status is None:
|
||||||
|
logger.error(
|
||||||
|
"reload_haproxy_safely() called without a pre-write backup "
|
||||||
|
"status - rollback will fall back to whatever backup already "
|
||||||
|
"exists on disk. Callers must call create_backup() BEFORE "
|
||||||
|
"writing the new configuration."
|
||||||
|
)
|
||||||
|
elif backup_status not in _ROLLBACK_AVAILABLE_STATUSES:
|
||||||
|
logger.warning(
|
||||||
|
f"Proceeding with reload without a rollback target "
|
||||||
|
f"(backup status: {backup_status})"
|
||||||
|
)
|
||||||
|
|
||||||
# Validate new configuration
|
# Validate new configuration
|
||||||
is_valid, error_msg = validate_haproxy_config()
|
is_valid, error_msg = validate_haproxy_config()
|
||||||
if not is_valid:
|
if not is_valid:
|
||||||
# Restore backup on validation failure
|
# Restore backup on validation failure
|
||||||
restore_backup()
|
restored, restore_msg = restore_backup()
|
||||||
|
if not restored:
|
||||||
|
logger.critical(
|
||||||
|
"Config validation failed AND rollback was not possible - "
|
||||||
|
f"{HAPROXY_CONFIG_PATH} holds an invalid configuration that "
|
||||||
|
"HAProxy will refuse to start with"
|
||||||
|
)
|
||||||
|
return False, (f"Config validation failed: {error_msg} | "
|
||||||
|
f"ROLLBACK FAILED: {restore_msg}")
|
||||||
return False, f"Config validation failed: {error_msg}"
|
return False, f"Config validation failed: {error_msg}"
|
||||||
|
|
||||||
# Attempt reload
|
# Attempt reload
|
||||||
@@ -2010,20 +2355,28 @@ def reload_haproxy_safely():
|
|||||||
|
|
||||||
if reload_result.returncode == 0:
|
if reload_result.returncode == 0:
|
||||||
logger.info("HAProxy reloaded successfully")
|
logger.info("HAProxy reloaded successfully")
|
||||||
|
# Now - and only now - is this config known good.
|
||||||
|
promote_current_config_to_backup()
|
||||||
return True, "HAProxy reloaded successfully"
|
return True, "HAProxy reloaded successfully"
|
||||||
else:
|
else:
|
||||||
# Reload failed, restore backup
|
# Reload failed, restore backup
|
||||||
restore_backup()
|
restored, restore_msg = restore_backup()
|
||||||
# Try to reload with backup config
|
if restored:
|
||||||
subprocess.run('echo "reload" | socat stdio /tmp/haproxy-cli',
|
# Try to reload with the restored (known-good) config
|
||||||
shell=True, capture_output=True)
|
subprocess.run(
|
||||||
|
'echo "reload" | socat stdio /tmp/haproxy-cli',
|
||||||
|
shell=True, capture_output=True)
|
||||||
error_msg = f"HAProxy reload failed: {reload_result.stderr}"
|
error_msg = f"HAProxy reload failed: {reload_result.stderr}"
|
||||||
|
if not restored:
|
||||||
|
error_msg += f" | ROLLBACK FAILED: {restore_msg}"
|
||||||
logger.error(error_msg)
|
logger.error(error_msg)
|
||||||
return False, error_msg
|
return False, error_msg
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
# Critical error during reload, restore backup
|
# Critical error during reload, restore backup
|
||||||
restore_backup()
|
restored, restore_msg = restore_backup()
|
||||||
error_msg = f"Critical error during reload: {e}"
|
error_msg = f"Critical error during reload: {e}"
|
||||||
|
if not restored:
|
||||||
|
error_msg += f" | ROLLBACK FAILED: {restore_msg}"
|
||||||
logger.error(error_msg)
|
logger.error(error_msg)
|
||||||
return False, error_msg
|
return False, error_msg
|
||||||
else:
|
else:
|
||||||
@@ -2034,11 +2387,15 @@ def reload_haproxy_safely():
|
|||||||
check=True, capture_output=True, text=True
|
check=True, capture_output=True, text=True
|
||||||
)
|
)
|
||||||
logger.info("HAProxy started successfully")
|
logger.info("HAProxy started successfully")
|
||||||
|
# Now - and only now - is this config known good.
|
||||||
|
promote_current_config_to_backup()
|
||||||
return True, "HAProxy started successfully"
|
return True, "HAProxy started successfully"
|
||||||
except subprocess.CalledProcessError as e:
|
except subprocess.CalledProcessError as e:
|
||||||
# Start failed, restore backup
|
# Start failed, restore backup
|
||||||
restore_backup()
|
restored, restore_msg = restore_backup()
|
||||||
error_msg = f"Failed to start HAProxy: {e.stderr}"
|
error_msg = f"Failed to start HAProxy: {e.stderr}"
|
||||||
|
if not restored:
|
||||||
|
error_msg += f" | ROLLBACK FAILED: {restore_msg}"
|
||||||
logger.error(error_msg)
|
logger.error(error_msg)
|
||||||
return False, error_msg
|
return False, error_msg
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
|
|||||||
@@ -0,0 +1,52 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Idempotent haproxy liveness check — driven by the in-container supervisor loop.
|
||||||
|
|
||||||
|
Why this exists
|
||||||
|
---------------
|
||||||
|
haproxy runs as a *background child of PID 1* (gunicorn) — it is started once at
|
||||||
|
container init (scripts/init.py -> do_initial_setup -> start_haproxy) and then
|
||||||
|
left running. Nothing supervises it after that. If the haproxy master process
|
||||||
|
dies mid-life (SIGABRT -> exit 134, segfault, or an OOM of the haproxy master),
|
||||||
|
the container stays "up" because gunicorn is still PID 1, so Docker's
|
||||||
|
`--restart` policy never fires. haproxy then stays down until the *external*
|
||||||
|
host watchdog (haproxy-watchdog.sh) notices port 80 is dead for ~3 minutes and
|
||||||
|
does a full `docker restart` — which drops every in-flight connection.
|
||||||
|
|
||||||
|
This script closes that gap: called on a short interval by the supervisor loop
|
||||||
|
in start-up.sh, it re-launches haproxy *in place* within one interval.
|
||||||
|
|
||||||
|
Safety
|
||||||
|
------
|
||||||
|
start_haproxy() is guarded by `is_process_running('haproxy')` (psutil-based, so
|
||||||
|
it works in this container which has no `ps`), so calling this while haproxy is
|
||||||
|
healthy is a cheap no-op. It only ever acts when haproxy is genuinely gone.
|
||||||
|
"""
|
||||||
|
import sys
|
||||||
|
|
||||||
|
sys.path.insert(0, '/haproxy')
|
||||||
|
import haproxy_manager # noqa: E402 (sys.path manipulation must come first)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
if haproxy_manager.is_process_running('haproxy'):
|
||||||
|
return 0
|
||||||
|
|
||||||
|
haproxy_manager.logger.warning(
|
||||||
|
"[haproxy-supervisor] haproxy process not found — attempting in-place restart"
|
||||||
|
)
|
||||||
|
# start_haproxy() validates the config (and regenerates it if invalid)
|
||||||
|
# before launching, and swallows its own errors, so it will not raise here.
|
||||||
|
haproxy_manager.start_haproxy()
|
||||||
|
|
||||||
|
if haproxy_manager.is_process_running('haproxy'):
|
||||||
|
haproxy_manager.logger.info("[haproxy-supervisor] haproxy restarted in place")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
haproxy_manager.logger.error(
|
||||||
|
"[haproxy-supervisor] haproxy restart FAILED — still not running after start_haproxy()"
|
||||||
|
)
|
||||||
|
return 1
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
sys.exit(main())
|
||||||
@@ -6,22 +6,40 @@
|
|||||||
SOCKET="/tmp/haproxy-cli"
|
SOCKET="/tmp/haproxy-cli"
|
||||||
MAP_FILE="/etc/haproxy/blocked_ips.map"
|
MAP_FILE="/etc/haproxy/blocked_ips.map"
|
||||||
|
|
||||||
|
# HAProxy runs in master-worker mode here, and /tmp/haproxy-cli is the MASTER
|
||||||
|
# socket. Data-plane commands (map/table manipulation) are NOT accepted on the
|
||||||
|
# master socket — they must be routed to a worker with the "@<n>" prefix. "@1"
|
||||||
|
# targets the current active worker. (A bare "add map ..." on the master socket
|
||||||
|
# fails with "Unknown command: 'add'".)
|
||||||
|
cli() { printf '@1 %s\n' "$*" | socat stdio "$SOCKET"; }
|
||||||
|
|
||||||
|
# Map lookup in haproxy.cfg is `map_ip(...,0) -m int gt 0`, so each entry MUST be
|
||||||
|
# "<ip_or_cidr> 1" — a bare IP yields an empty value (0) and is NOT blocked once
|
||||||
|
# the map file is re-read on reload. The runtime map and the file must agree.
|
||||||
|
MAP_VALUE=1
|
||||||
|
|
||||||
# Ensure map file exists
|
# Ensure map file exists
|
||||||
if [ ! -f "$MAP_FILE" ]; then
|
if [ ! -f "$MAP_FILE" ]; then
|
||||||
touch "$MAP_FILE"
|
echo "# Blocked IPs - Format: <ip_or_cidr> 1 (one per line)" > "$MAP_FILE"
|
||||||
echo "# Blocked IPs - Format: IP_ADDRESS" > "$MAP_FILE"
|
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# Escape regex metacharacters (notably dots) in an IP/CIDR for anchored matching.
|
||||||
|
esc_re() { printf '%s' "$1" | sed 's/[.[\*^$/]/\\&/g'; }
|
||||||
|
|
||||||
case "$1" in
|
case "$1" in
|
||||||
block)
|
block)
|
||||||
if [ -z "$2" ]; then
|
if [ -z "$2" ]; then
|
||||||
echo "Usage: $0 block IP_ADDRESS"
|
echo "Usage: $0 block IP_ADDRESS"
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
# Add IP to map file
|
re="$(esc_re "$2")"
|
||||||
grep -q "^$2" "$MAP_FILE" || echo "$2" >> "$MAP_FILE"
|
# Persist (idempotent, anchored so 1.2.3.4 doesn't match 1.2.3.45),
|
||||||
# Add to runtime map
|
# always with the trailing value so the block survives a reload.
|
||||||
echo "add map /etc/haproxy/blocked_ips.map $2 1" | socat stdio "$SOCKET"
|
if ! grep -qE "^${re}([[:space:]]|$)" "$MAP_FILE"; then
|
||||||
|
echo "$2 $MAP_VALUE" >> "$MAP_FILE"
|
||||||
|
fi
|
||||||
|
# Apply at runtime immediately (no reload).
|
||||||
|
cli "add map $MAP_FILE $2 $MAP_VALUE"
|
||||||
echo "Blocked IP: $2"
|
echo "Blocked IP: $2"
|
||||||
;;
|
;;
|
||||||
|
|
||||||
@@ -30,31 +48,33 @@ case "$1" in
|
|||||||
echo "Usage: $0 unblock IP_ADDRESS"
|
echo "Usage: $0 unblock IP_ADDRESS"
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
# Remove from map file
|
re="$(esc_re "$2")"
|
||||||
sed -i "/^$2$/d" "$MAP_FILE"
|
# Remove from map file (match "<ip>" optionally followed by a value).
|
||||||
# Remove from runtime map
|
sed -i -E "/^${re}([[:space:]]|$)/d" "$MAP_FILE"
|
||||||
echo "del map /etc/haproxy/blocked_ips.map $2" | socat stdio "$SOCKET"
|
# Remove from runtime map.
|
||||||
|
cli "del map $MAP_FILE $2"
|
||||||
echo "Unblocked IP: $2"
|
echo "Unblocked IP: $2"
|
||||||
;;
|
;;
|
||||||
|
|
||||||
list)
|
list)
|
||||||
echo "Currently blocked IPs:"
|
echo "Currently blocked IPs:"
|
||||||
echo "show map /etc/haproxy/blocked_ips.map" | socat stdio "$SOCKET" | awk '{print $1}'
|
# `show map` output is "<ptr> <key> <value>" — the IP is field 2.
|
||||||
|
cli "show map $MAP_FILE" | awk 'NF>=2 {print $2}'
|
||||||
;;
|
;;
|
||||||
|
|
||||||
clear)
|
clear)
|
||||||
echo "Clearing all blocked IPs..."
|
echo "Clearing all blocked IPs..."
|
||||||
echo "clear map /etc/haproxy/blocked_ips.map" | socat stdio "$SOCKET"
|
cli "clear map $MAP_FILE"
|
||||||
echo "# Blocked IPs - Format: IP_ADDRESS" > "$MAP_FILE"
|
echo "# Blocked IPs - Format: <ip_or_cidr> 1 (one per line)" > "$MAP_FILE"
|
||||||
echo "All IPs unblocked"
|
echo "All IPs unblocked"
|
||||||
;;
|
;;
|
||||||
|
|
||||||
stats)
|
stats)
|
||||||
echo "=== HAProxy 3.0.11 Threat Intelligence Dashboard ==="
|
echo "=== HAProxy 3.0.11 Threat Intelligence Dashboard ==="
|
||||||
echo "show table web" | socat stdio "$SOCKET" | awk 'NR<=21'
|
cli "show table web" | awk 'NR<=21'
|
||||||
echo ""
|
echo ""
|
||||||
echo "=== Top Threat Scores ==="
|
echo "=== Top Threat Scores ==="
|
||||||
echo "show table web" | socat stdio "$SOCKET" | awk '
|
cli "show table web" | awk '
|
||||||
NR>1 {
|
NR>1 {
|
||||||
ip = $1
|
ip = $1
|
||||||
auth_fail = 0
|
auth_fail = 0
|
||||||
@@ -84,7 +104,7 @@ case "$1" in
|
|||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
# Add to manual blacklist using GPC(13)
|
# Add to manual blacklist using GPC(13)
|
||||||
echo "set table web key $2 data.gpc(13) 1" | socat stdio "$SOCKET"
|
cli "set table web key $2 data.gpc(13) 1"
|
||||||
echo "Manually blacklisted IP: $2 (GPC(13) = 1)"
|
echo "Manually blacklisted IP: $2 (GPC(13) = 1)"
|
||||||
;;
|
;;
|
||||||
|
|
||||||
@@ -94,7 +114,7 @@ case "$1" in
|
|||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
# Clear manual blacklist flag
|
# Clear manual blacklist flag
|
||||||
echo "set table web key $2 data.gpc(13) 0" | socat stdio "$SOCKET"
|
cli "set table web key $2 data.gpc(13) 0"
|
||||||
echo "Removed manual blacklist for IP: $2"
|
echo "Removed manual blacklist for IP: $2"
|
||||||
;;
|
;;
|
||||||
|
|
||||||
@@ -104,7 +124,7 @@ case "$1" in
|
|||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
# Add to auto-blacklist using GPC(14)
|
# Add to auto-blacklist using GPC(14)
|
||||||
echo "set table web key $2 data.gpc(14) 1" | socat stdio "$SOCKET"
|
cli "set table web key $2 data.gpc(14) 1"
|
||||||
echo "Auto-blacklisted IP: $2 (GPC(14) = 1)"
|
echo "Auto-blacklisted IP: $2 (GPC(14) = 1)"
|
||||||
;;
|
;;
|
||||||
|
|
||||||
@@ -115,14 +135,14 @@ case "$1" in
|
|||||||
fi
|
fi
|
||||||
# Show detailed threat breakdown for specific IP
|
# Show detailed threat breakdown for specific IP
|
||||||
echo "Threat analysis for $2:"
|
echo "Threat analysis for $2:"
|
||||||
echo "show table web key $2" | socat stdio "$SOCKET"
|
cli "show table web key $2"
|
||||||
;;
|
;;
|
||||||
|
|
||||||
*)
|
*)
|
||||||
echo "Usage: $0 {block|unblock|list|clear|blacklist|unblacklist|auto-blacklist|threat-score|stats} [IP_ADDRESS]"
|
echo "Usage: $0 {block|unblock|list|clear|blacklist|unblacklist|auto-blacklist|threat-score|stats} [IP_ADDRESS]"
|
||||||
echo ""
|
echo ""
|
||||||
echo "HAProxy 3.0.11 Enhanced Security Commands:"
|
echo "HAProxy 3.0.11 Enhanced Security Commands:"
|
||||||
echo " block IP - Block IP via map file (immediate)"
|
echo " block IP - Block IP via map file (immediate + persisted)"
|
||||||
echo " unblock IP - Unblock IP from map file"
|
echo " unblock IP - Unblock IP from map file"
|
||||||
echo " blacklist IP - Manual blacklist via GPC(13) array"
|
echo " blacklist IP - Manual blacklist via GPC(13) array"
|
||||||
echo " unblacklist IP - Remove manual blacklist flag"
|
echo " unblacklist IP - Remove manual blacklist flag"
|
||||||
|
|||||||
+23
-1
@@ -27,11 +27,33 @@ cron &
|
|||||||
# Phase 1: container init
|
# Phase 1: container init
|
||||||
python /haproxy/scripts/init.py
|
python /haproxy/scripts/init.py
|
||||||
|
|
||||||
|
# Phase 1.5: in-container haproxy supervisor.
|
||||||
|
# haproxy runs as a background child of PID 1 (gunicorn) with NOTHING watching
|
||||||
|
# it after init. If the haproxy master dies mid-life (e.g. SIGABRT -> exit 134,
|
||||||
|
# segfault), the container stays "up" (gunicorn is PID 1), Docker's --restart
|
||||||
|
# policy never fires, and haproxy is down until the external host watchdog
|
||||||
|
# full-restarts the whole container minutes later (dropping every connection).
|
||||||
|
# This loop revives haproxy in place within one interval. ensure_haproxy.py is
|
||||||
|
# idempotent — a cheap no-op whenever haproxy is already running.
|
||||||
|
HAPROXY_SUPERVISOR_INTERVAL="${HAPROXY_SUPERVISOR_INTERVAL:-15}"
|
||||||
|
(
|
||||||
|
while true; do
|
||||||
|
sleep "${HAPROXY_SUPERVISOR_INTERVAL}"
|
||||||
|
python /haproxy/scripts/ensure_haproxy.py 2>&1 || true
|
||||||
|
done
|
||||||
|
) &
|
||||||
|
|
||||||
# Phase 2: WSGI servers
|
# Phase 2: WSGI servers
|
||||||
# Tunable via env: HAPROXY_MGR_API_WORKERS (default 1), HAPROXY_MGR_API_TIMEOUT
|
# Tunable via env: HAPROXY_MGR_API_WORKERS (default 1), HAPROXY_MGR_API_TIMEOUT
|
||||||
# (default 120 — API can do slow ACME calls), HAPROXY_MGR_MAX_REQUESTS (default
|
# (default 120 — API can do slow ACME calls), HAPROXY_MGR_MAX_REQUESTS (default
|
||||||
# 1000 — worker recycle frequency).
|
# 1000 — worker recycle frequency).
|
||||||
API_WORKERS="${HAPROXY_MGR_API_WORKERS:-1}"
|
#
|
||||||
|
# API_WORKERS default is 2 (was 1). A single worker is a single point of
|
||||||
|
# failure: if its gthread pool ever wedges (see the 2026-07-07 subprocess-hang
|
||||||
|
# incident — now bounded by DEFAULT_SUBPROCESS_TIMEOUT in haproxy_manager.py),
|
||||||
|
# the entire management API goes dark. A second worker keeps the API answering
|
||||||
|
# (config regenerate, health, SSL) while the other recycles via --max-requests.
|
||||||
|
API_WORKERS="${HAPROXY_MGR_API_WORKERS:-2}"
|
||||||
API_TIMEOUT="${HAPROXY_MGR_API_TIMEOUT:-120}"
|
API_TIMEOUT="${HAPROXY_MGR_API_TIMEOUT:-120}"
|
||||||
MAX_REQ="${HAPROXY_MGR_MAX_REQUESTS:-1000}"
|
MAX_REQ="${HAPROXY_MGR_MAX_REQUESTS:-1000}"
|
||||||
MAX_REQ_JITTER="${HAPROXY_MGR_MAX_REQUESTS_JITTER:-100}"
|
MAX_REQ_JITTER="${HAPROXY_MGR_MAX_REQUESTS_JITTER:-100}"
|
||||||
|
|||||||
Executable
+467
@@ -0,0 +1,467 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Regression tests for HAProxy config backup / rollback ordering.
|
||||||
|
|
||||||
|
Why this file exists
|
||||||
|
--------------------
|
||||||
|
generate_config() used to write the new haproxy.cfg and only THEN call
|
||||||
|
reload_haproxy_safely() -> create_backup(), so the "backup" was a copy of the
|
||||||
|
config that had just been written. On a validation failure restore_backup()
|
||||||
|
restored the identical broken bytes: the advertised rollback was a no-op and a
|
||||||
|
fatal haproxy.cfg stayed on disk, where start_haproxy() refuses to launch.
|
||||||
|
|
||||||
|
These tests pin the ordering invariant (backup predates the write) and the
|
||||||
|
observable end-to-end behaviour (after a failed validation the file on disk is
|
||||||
|
the previous working config and HAProxy will start with it).
|
||||||
|
|
||||||
|
Running
|
||||||
|
-------
|
||||||
|
python3 scripts/test-config-rollback.py # tests the repo checkout
|
||||||
|
HAPROXY_MANAGER_DIR=/some/other/tree \
|
||||||
|
python3 scripts/test-config-rollback.py # tests another tree
|
||||||
|
|
||||||
|
The repo has no Python test framework (scripts/test-*.sh are curl-based
|
||||||
|
integration scripts against a running API), so this is a self-contained
|
||||||
|
stdlib-unittest script - no pytest, no venv, no new dependencies beyond the
|
||||||
|
application's own requirements.txt (Flask/Jinja2/psutil), which are already
|
||||||
|
present in the container image.
|
||||||
|
|
||||||
|
No HAProxy binary is required: a stub `haproxy` is put on PATH that mimics
|
||||||
|
`haproxy -c -f <file>` by rejecting any config containing the token
|
||||||
|
__BROKEN__, which is how the tests inject an invalid configuration.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import shutil
|
||||||
|
import sqlite3
|
||||||
|
import logging
|
||||||
|
import tempfile
|
||||||
|
import textwrap
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
BROKEN_TOKEN = '__BROKEN__'
|
||||||
|
|
||||||
|
MODULE_DIR = os.path.abspath(
|
||||||
|
os.environ.get('HAPROXY_MANAGER_DIR',
|
||||||
|
os.path.join(os.path.dirname(os.path.abspath(__file__)), '..'))
|
||||||
|
)
|
||||||
|
|
||||||
|
# haproxy_manager builds its Jinja2 environment from the relative path
|
||||||
|
# Path('templates'), so it has to be imported with the module dir as cwd.
|
||||||
|
os.chdir(MODULE_DIR)
|
||||||
|
sys.path.insert(0, MODULE_DIR)
|
||||||
|
|
||||||
|
# The module opens /var/log/haproxy-manager.log at import time via
|
||||||
|
# logging.FileHandler. Redirect that one call so the suite runs unprivileged.
|
||||||
|
_LOG_DIR = tempfile.mkdtemp(prefix='haproxy-mgr-test-logs-')
|
||||||
|
_real_file_handler = logging.FileHandler
|
||||||
|
logging.FileHandler = (
|
||||||
|
lambda fn, *a, **kw: _real_file_handler(
|
||||||
|
os.path.join(_LOG_DIR, os.path.basename(fn)), *a, **kw)
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
import haproxy_manager as hm
|
||||||
|
except ImportError as exc: # pragma: no cover - environment problem, not a failure
|
||||||
|
sys.stderr.write(
|
||||||
|
f"SKIP: cannot import haproxy_manager ({exc}).\n"
|
||||||
|
"Install the application requirements first: pip install -r requirements.txt\n"
|
||||||
|
)
|
||||||
|
raise SystemExit(77)
|
||||||
|
finally:
|
||||||
|
logging.FileHandler = _real_file_handler
|
||||||
|
|
||||||
|
logging.getLogger('haproxy_manager').setLevel(logging.CRITICAL)
|
||||||
|
|
||||||
|
FAKE_HAPROXY = textwrap.dedent(f"""\
|
||||||
|
#!/bin/sh
|
||||||
|
# Test stub for the haproxy binary.
|
||||||
|
# haproxy -c -f FILE -> exit 1 if FILE contains {BROKEN_TOKEN}, else 0
|
||||||
|
# haproxy -W -S ... -f FILE (start) -> same validation, then exit 0
|
||||||
|
cfg=""
|
||||||
|
while [ $# -gt 0 ]; do
|
||||||
|
case "$1" in -f) cfg="$2"; shift ;; esac
|
||||||
|
shift
|
||||||
|
done
|
||||||
|
if [ -n "$cfg" ] && grep -q '{BROKEN_TOKEN}' "$cfg" 2>/dev/null; then
|
||||||
|
echo "[ALERT] parsing [$cfg:1] : unknown keyword '{BROKEN_TOKEN}'" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
exit 0
|
||||||
|
""")
|
||||||
|
|
||||||
|
|
||||||
|
class RollbackTestCase(unittest.TestCase):
|
||||||
|
"""Base fixture: an isolated fake /etc/haproxy plus a stub haproxy binary."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self.tmp = tempfile.mkdtemp(prefix='haproxy-rollback-test-')
|
||||||
|
self.addCleanup(shutil.rmtree, self.tmp, True)
|
||||||
|
|
||||||
|
bindir = os.path.join(self.tmp, 'bin')
|
||||||
|
os.makedirs(bindir)
|
||||||
|
stub = os.path.join(bindir, 'haproxy')
|
||||||
|
with open(stub, 'w') as fh:
|
||||||
|
fh.write(FAKE_HAPROXY)
|
||||||
|
os.chmod(stub, 0o755)
|
||||||
|
self._old_path = os.environ['PATH']
|
||||||
|
os.environ['PATH'] = bindir + os.pathsep + self._old_path
|
||||||
|
self.addCleanup(lambda: os.environ.__setitem__('PATH', self._old_path))
|
||||||
|
|
||||||
|
self.etc = os.path.join(self.tmp, 'etc')
|
||||||
|
os.makedirs(self.etc)
|
||||||
|
|
||||||
|
overrides = {
|
||||||
|
'DB_FILE': os.path.join(self.etc, 'haproxy_config.db'),
|
||||||
|
'HAPROXY_CONFIG_PATH': os.path.join(self.etc, 'haproxy.cfg'),
|
||||||
|
'HAPROXY_BACKUP_PATH': os.path.join(self.etc, 'haproxy.cfg.backup'),
|
||||||
|
'BLOCKED_IPS_MAP_PATH': os.path.join(self.etc, 'blocked_ips.map'),
|
||||||
|
'BLOCKED_IPS_MAP_BACKUP_PATH': os.path.join(self.etc, 'blocked_ips.map.backup'),
|
||||||
|
'CLUSTER_SECRET_PATH': os.path.join(self.etc, 'cluster-secret'),
|
||||||
|
'SSL_CERTS_DIR': os.path.join(self.etc, 'certs'),
|
||||||
|
'HAPROXY_SOCKET_PATH': os.path.join(self.etc, 'haproxy.sock'),
|
||||||
|
# Added by the rollback fix; older trees do not have it.
|
||||||
|
'CORAZA_SPOE_CONFIG_PATH': os.path.join(self.etc, 'coraza-spoe.cfg'),
|
||||||
|
'CORAZA_SPOE_BACKUP_PATH': os.path.join(self.etc, 'coraza-spoe.cfg.backup'),
|
||||||
|
}
|
||||||
|
self._saved = {}
|
||||||
|
for name, value in overrides.items():
|
||||||
|
self._saved[name] = getattr(hm, name, None)
|
||||||
|
setattr(hm, name, value)
|
||||||
|
self.addCleanup(self._restore_globals)
|
||||||
|
os.makedirs(hm.SSL_CERTS_DIR)
|
||||||
|
|
||||||
|
# log_operation() appends to a hardcoded /var/log path. Injecting `open`
|
||||||
|
# into the module namespace shadows the builtin for that module only
|
||||||
|
# (module globals are searched before builtins), so the real
|
||||||
|
# log_operation code still runs.
|
||||||
|
real_open = open
|
||||||
|
log_dir = self.tmp
|
||||||
|
|
||||||
|
def _redirecting_open(path, *args, **kwargs):
|
||||||
|
if isinstance(path, str) and path.startswith('/var/log/'):
|
||||||
|
path = os.path.join(log_dir, os.path.basename(path))
|
||||||
|
return real_open(path, *args, **kwargs)
|
||||||
|
|
||||||
|
hm.open = _redirecting_open
|
||||||
|
self.addCleanup(lambda: hm.__dict__.pop('open', None))
|
||||||
|
|
||||||
|
hm.init_db()
|
||||||
|
|
||||||
|
def _restore_globals(self):
|
||||||
|
for name, value in self._saved.items():
|
||||||
|
if value is None:
|
||||||
|
hm.__dict__.pop(name, None)
|
||||||
|
else:
|
||||||
|
setattr(hm, name, value)
|
||||||
|
|
||||||
|
# -- helpers ---------------------------------------------------------
|
||||||
|
def add_domain(self, domain, backend_name, address='10.0.0.1'):
|
||||||
|
with sqlite3.connect(hm.DB_FILE) as conn:
|
||||||
|
cur = conn.cursor()
|
||||||
|
cur.execute('INSERT INTO domains (domain, ssl_enabled) VALUES (?, 0)',
|
||||||
|
(domain,))
|
||||||
|
domain_id = cur.lastrowid
|
||||||
|
cur.execute('INSERT INTO backends (name, domain_id) VALUES (?, ?)',
|
||||||
|
(backend_name, domain_id))
|
||||||
|
backend_id = cur.lastrowid
|
||||||
|
cur.execute(
|
||||||
|
'INSERT INTO backend_servers '
|
||||||
|
'(backend_id, server_name, server_address, server_port) '
|
||||||
|
'VALUES (?, ?, ?, ?)',
|
||||||
|
(backend_id, 'srv1', address, 8080))
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
def block_ip(self, ip):
|
||||||
|
with sqlite3.connect(hm.DB_FILE) as conn:
|
||||||
|
conn.execute('INSERT INTO blocked_ips (ip_address, reason) VALUES (?, ?)',
|
||||||
|
(ip, 'test'))
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
def read(self, path):
|
||||||
|
with open(path) as fh:
|
||||||
|
return fh.read()
|
||||||
|
|
||||||
|
def config_is_loadable(self):
|
||||||
|
"""True if HAProxy would accept the config currently on disk."""
|
||||||
|
import subprocess
|
||||||
|
return subprocess.run(
|
||||||
|
['haproxy', '-c', '-f', hm.HAPROXY_CONFIG_PATH],
|
||||||
|
capture_output=True).returncode == 0
|
||||||
|
|
||||||
|
def generate_good_config(self):
|
||||||
|
self.add_domain('good.example.com', 'good_backend')
|
||||||
|
hm.generate_config()
|
||||||
|
self.assertTrue(self.config_is_loadable(),
|
||||||
|
'fixture precondition: first generated config must be valid')
|
||||||
|
return self.read(hm.HAPROXY_CONFIG_PATH)
|
||||||
|
|
||||||
|
def break_the_config(self):
|
||||||
|
"""Queue a domain whose rendered backend the validator rejects."""
|
||||||
|
self.add_domain('bad.example.com', BROKEN_TOKEN + '_backend', '10.0.0.2')
|
||||||
|
|
||||||
|
|
||||||
|
class TestBackupOrdering(RollbackTestCase):
|
||||||
|
|
||||||
|
def test_backup_is_taken_before_the_new_config_is_written(self):
|
||||||
|
"""The ordering invariant, asserted directly.
|
||||||
|
|
||||||
|
Whatever create_backup() sees on disk must be the OLD config; if the
|
||||||
|
write happens first the backup is a copy of the new config and rollback
|
||||||
|
is meaningless.
|
||||||
|
"""
|
||||||
|
good = self.generate_good_config()
|
||||||
|
|
||||||
|
seen = {}
|
||||||
|
real_create_backup = hm.create_backup
|
||||||
|
|
||||||
|
def spy(*args, **kwargs):
|
||||||
|
seen['config_on_disk'] = self.read(hm.HAPROXY_CONFIG_PATH)
|
||||||
|
return real_create_backup(*args, **kwargs)
|
||||||
|
|
||||||
|
hm.create_backup = spy
|
||||||
|
self.addCleanup(setattr, hm, 'create_backup', real_create_backup)
|
||||||
|
|
||||||
|
self.add_domain('second.example.com', 'second_backend', '10.0.0.3')
|
||||||
|
hm.generate_config()
|
||||||
|
|
||||||
|
self.assertIn('config_on_disk', seen,
|
||||||
|
'create_backup() was never called during generate_config()')
|
||||||
|
self.assertEqual(
|
||||||
|
seen['config_on_disk'], good,
|
||||||
|
'create_backup() ran AFTER the new config was written - the backup '
|
||||||
|
'is a copy of the new config, so rollback cannot undo anything')
|
||||||
|
|
||||||
|
def test_backup_tracks_the_last_known_good_config(self):
|
||||||
|
"""After a change that validated AND loaded, the backup is that config.
|
||||||
|
|
||||||
|
The rollback target is "the last configuration HAProxy actually ran",
|
||||||
|
not "the file that happened to be there last time".
|
||||||
|
"""
|
||||||
|
good = self.generate_good_config()
|
||||||
|
self.add_domain('second.example.com', 'second_backend', '10.0.0.3')
|
||||||
|
hm.generate_config()
|
||||||
|
|
||||||
|
live = self.read(hm.HAPROXY_CONFIG_PATH)
|
||||||
|
self.assertNotEqual(live, good, 'fixture sanity: the new config should differ')
|
||||||
|
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), live,
|
||||||
|
'the successful config was not recorded as known-good')
|
||||||
|
|
||||||
|
def test_backup_is_not_promoted_when_the_change_fails(self):
|
||||||
|
"""A config that never loaded must not become the rollback target."""
|
||||||
|
good = self.generate_good_config()
|
||||||
|
self.break_the_config()
|
||||||
|
with self.assertRaises(Exception):
|
||||||
|
hm.generate_config()
|
||||||
|
|
||||||
|
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good,
|
||||||
|
'a config that failed validation was promoted to backup')
|
||||||
|
|
||||||
|
|
||||||
|
class TestRollbackEndToEnd(RollbackTestCase):
|
||||||
|
|
||||||
|
def test_failed_validation_leaves_the_last_good_config_on_disk(self):
|
||||||
|
good = self.generate_good_config()
|
||||||
|
|
||||||
|
self.break_the_config()
|
||||||
|
with self.assertRaises(Exception):
|
||||||
|
hm.generate_config()
|
||||||
|
|
||||||
|
on_disk = self.read(hm.HAPROXY_CONFIG_PATH)
|
||||||
|
self.assertNotIn(BROKEN_TOKEN, on_disk,
|
||||||
|
'the rejected config is still on disk - rollback was a no-op')
|
||||||
|
self.assertEqual(on_disk, good,
|
||||||
|
'on-disk config is not byte-identical to the last good one')
|
||||||
|
|
||||||
|
def test_haproxy_would_still_start_after_a_failed_change(self):
|
||||||
|
"""The operational consequence: the edge can still come up."""
|
||||||
|
self.generate_good_config()
|
||||||
|
self.break_the_config()
|
||||||
|
with self.assertRaises(Exception):
|
||||||
|
hm.generate_config()
|
||||||
|
|
||||||
|
self.assertTrue(self.config_is_loadable(),
|
||||||
|
'HAProxy would refuse to start with the config left on disk')
|
||||||
|
with self.assertLogs('haproxy_manager', level='INFO') as captured:
|
||||||
|
hm.start_haproxy()
|
||||||
|
self.assertTrue(
|
||||||
|
any('HAProxy started successfully' in line for line in captured.output),
|
||||||
|
f'start_haproxy() did not succeed after rollback: {captured.output}')
|
||||||
|
|
||||||
|
def test_blocked_ips_map_is_rolled_back_too(self):
|
||||||
|
"""generate_config() rewrites the map file before writing haproxy.cfg."""
|
||||||
|
self.block_ip('192.0.2.10')
|
||||||
|
self.generate_good_config()
|
||||||
|
good_map = self.read(hm.BLOCKED_IPS_MAP_PATH)
|
||||||
|
|
||||||
|
self.block_ip('198.51.100.20')
|
||||||
|
self.break_the_config()
|
||||||
|
with self.assertRaises(Exception):
|
||||||
|
hm.generate_config()
|
||||||
|
|
||||||
|
self.assertEqual(self.read(hm.BLOCKED_IPS_MAP_PATH), good_map,
|
||||||
|
'blocked IPs map was not rolled back with the config')
|
||||||
|
|
||||||
|
def test_first_run_failure_reports_that_rollback_was_impossible(self):
|
||||||
|
"""No prior config: there is nothing to restore, and that must be said.
|
||||||
|
|
||||||
|
A missing backup must never be reported as a successful restore, and it
|
||||||
|
must never be turned into "restore an empty file".
|
||||||
|
"""
|
||||||
|
self.break_the_config()
|
||||||
|
with self.assertRaises(Exception) as ctx:
|
||||||
|
hm.generate_config()
|
||||||
|
|
||||||
|
self.assertIn('ROLLBACK FAILED', str(ctx.exception),
|
||||||
|
'a failed change with no backup was not reported as such')
|
||||||
|
self.assertFalse(os.path.exists(hm.HAPROXY_BACKUP_PATH),
|
||||||
|
'a backup was fabricated from the broken config')
|
||||||
|
# The broken config is deliberately left in place: start_haproxy() can
|
||||||
|
# then detect it and try to regenerate. It must not be blanked.
|
||||||
|
self.assertGreater(os.path.getsize(hm.HAPROXY_CONFIG_PATH), 0,
|
||||||
|
'config file was emptied instead of left for diagnosis')
|
||||||
|
|
||||||
|
|
||||||
|
class TestBackupPrimitives(RollbackTestCase):
|
||||||
|
|
||||||
|
def test_restore_backup_distinguishes_missing_backup_from_success(self):
|
||||||
|
restored, message = hm.restore_backup()
|
||||||
|
self.assertFalse(restored,
|
||||||
|
'restore_backup() reported success with no backup present')
|
||||||
|
self.assertIn('cannot roll back', message.lower())
|
||||||
|
|
||||||
|
good = self.generate_good_config()
|
||||||
|
with open(hm.HAPROXY_CONFIG_PATH, 'w') as fh:
|
||||||
|
fh.write('scribbled over\n')
|
||||||
|
|
||||||
|
restored, message = hm.restore_backup()
|
||||||
|
self.assertTrue(restored, message)
|
||||||
|
self.assertEqual(self.read(hm.HAPROXY_CONFIG_PATH), good)
|
||||||
|
|
||||||
|
def test_a_successful_generation_records_a_rollback_target(self):
|
||||||
|
"""Even the first-ever generation must leave something to roll back to."""
|
||||||
|
good = self.generate_good_config()
|
||||||
|
self.assertTrue(
|
||||||
|
os.path.exists(hm.HAPROXY_BACKUP_PATH),
|
||||||
|
'after a successful reload there is still no known-good backup')
|
||||||
|
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good)
|
||||||
|
|
||||||
|
def test_a_broken_current_config_does_not_replace_a_good_backup(self):
|
||||||
|
"""The known-good marker.
|
||||||
|
|
||||||
|
If the config already on disk is broken (previous failed write, manual
|
||||||
|
edit), snapshotting it would make "rollback" mean "restore a different
|
||||||
|
broken config". The older validated backup must survive.
|
||||||
|
"""
|
||||||
|
good = self.generate_good_config()
|
||||||
|
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good,
|
||||||
|
'fixture: a good backup should exist by now')
|
||||||
|
|
||||||
|
with open(hm.HAPROXY_CONFIG_PATH, 'w') as fh:
|
||||||
|
fh.write(f'garbage {BROKEN_TOKEN} config\n')
|
||||||
|
|
||||||
|
ok, status = hm.create_backup()
|
||||||
|
self.assertTrue(ok)
|
||||||
|
self.assertEqual(status, 'kept_previous')
|
||||||
|
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good,
|
||||||
|
'a broken config overwrote the known-good backup')
|
||||||
|
|
||||||
|
def test_reload_does_not_take_its_own_backup(self):
|
||||||
|
"""reload_haproxy_safely() runs after the write, so it must not back up."""
|
||||||
|
good = self.generate_good_config()
|
||||||
|
with open(hm.HAPROXY_CONFIG_PATH, 'w') as fh:
|
||||||
|
fh.write(f'broken {BROKEN_TOKEN}\n')
|
||||||
|
|
||||||
|
success, message = hm.reload_haproxy_safely(backup_status='created')
|
||||||
|
|
||||||
|
self.assertFalse(success)
|
||||||
|
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good,
|
||||||
|
'reload_haproxy_safely() overwrote the good backup')
|
||||||
|
self.assertEqual(self.read(hm.HAPROXY_CONFIG_PATH), good,
|
||||||
|
'reload_haproxy_safely() did not roll the config back')
|
||||||
|
|
||||||
|
def test_unchanged_config_is_not_revalidated(self):
|
||||||
|
"""Fast path: if the backup already is the live config, do no work.
|
||||||
|
|
||||||
|
generate_config() runs inside customer-facing API calls and
|
||||||
|
`haproxy -c` is expensive on an edge with hundreds of certificates.
|
||||||
|
"""
|
||||||
|
self.generate_good_config()
|
||||||
|
|
||||||
|
calls = []
|
||||||
|
real_validate = hm.validate_config_file
|
||||||
|
hm.validate_config_file = lambda path: (calls.append(path),
|
||||||
|
real_validate(path))[1]
|
||||||
|
self.addCleanup(setattr, hm, 'validate_config_file', real_validate)
|
||||||
|
|
||||||
|
ok, status = hm.create_backup()
|
||||||
|
self.assertTrue(ok)
|
||||||
|
self.assertEqual(status, 'created')
|
||||||
|
self.assertEqual(calls, [],
|
||||||
|
'the unchanged live config was re-validated needlessly')
|
||||||
|
|
||||||
|
def test_fast_path_does_not_hide_a_drifted_broken_config(self):
|
||||||
|
"""If the live config drifted from the backup, the gate must still run."""
|
||||||
|
good = self.generate_good_config()
|
||||||
|
with open(hm.HAPROXY_CONFIG_PATH, 'w') as fh:
|
||||||
|
fh.write(f'hand edited {BROKEN_TOKEN}\n')
|
||||||
|
|
||||||
|
ok, status = hm.create_backup()
|
||||||
|
self.assertTrue(ok)
|
||||||
|
self.assertEqual(status, 'kept_previous',
|
||||||
|
'a drifted broken config was silently accepted')
|
||||||
|
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good)
|
||||||
|
|
||||||
|
def test_backup_set_covers_every_file_generate_config_writes(self):
|
||||||
|
pairs = dict(hm._config_backup_pairs())
|
||||||
|
for path in (hm.HAPROXY_CONFIG_PATH, hm.BLOCKED_IPS_MAP_PATH,
|
||||||
|
hm.CORAZA_SPOE_CONFIG_PATH):
|
||||||
|
self.assertIn(path, pairs,
|
||||||
|
f'{path} is written by generate_config() but is not '
|
||||||
|
'part of the backed-up config set')
|
||||||
|
|
||||||
|
def test_coraza_spoe_config_round_trips(self):
|
||||||
|
self.generate_good_config()
|
||||||
|
with open(hm.CORAZA_SPOE_CONFIG_PATH, 'w') as fh:
|
||||||
|
fh.write('spoe-good\n')
|
||||||
|
hm.create_backup()
|
||||||
|
with open(hm.CORAZA_SPOE_CONFIG_PATH, 'w') as fh:
|
||||||
|
fh.write('spoe-broken\n')
|
||||||
|
restored, message = hm.restore_backup()
|
||||||
|
self.assertTrue(restored, message)
|
||||||
|
self.assertEqual(self.read(hm.CORAZA_SPOE_CONFIG_PATH), 'spoe-good\n')
|
||||||
|
|
||||||
|
|
||||||
|
class TestAtomicWrite(RollbackTestCase):
|
||||||
|
|
||||||
|
def test_write_is_atomic_and_preserves_mode(self):
|
||||||
|
path = os.path.join(self.etc, 'atomic.cfg')
|
||||||
|
with open(path, 'w') as fh:
|
||||||
|
fh.write('old')
|
||||||
|
os.chmod(path, 0o644)
|
||||||
|
|
||||||
|
hm.write_config_atomically(path, 'new content\n')
|
||||||
|
|
||||||
|
self.assertEqual(self.read(path), 'new content\n')
|
||||||
|
self.assertEqual(oct(os.stat(path).st_mode & 0o777), oct(0o644))
|
||||||
|
leftovers = [n for n in os.listdir(self.etc) if n.endswith('.tmp')]
|
||||||
|
self.assertEqual(leftovers, [], f'temp files left behind: {leftovers}')
|
||||||
|
|
||||||
|
def test_failed_write_leaves_the_previous_file_intact(self):
|
||||||
|
path = os.path.join(self.etc, 'atomic.cfg')
|
||||||
|
with open(path, 'w') as fh:
|
||||||
|
fh.write('old content\n')
|
||||||
|
|
||||||
|
# Anything that makes f.write() blow up mid-flight stands in for a full
|
||||||
|
# disk / killed container.
|
||||||
|
with self.assertRaises(Exception):
|
||||||
|
hm.write_config_atomically(path, object())
|
||||||
|
|
||||||
|
self.assertEqual(self.read(path), 'old content\n',
|
||||||
|
'a failed write clobbered the previous config')
|
||||||
|
leftovers = [n for n in os.listdir(self.etc) if n.endswith('.tmp')]
|
||||||
|
self.assertEqual(leftovers, [], f'temp files left behind: {leftovers}')
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
print(f"testing haproxy_manager from: {MODULE_DIR}")
|
||||||
|
unittest.main(verbosity=2)
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
# Long-lived backend for {{ name }} (template_override='hap_backend_longlived').
|
||||||
|
# Use for apps whose PRIMARY traffic holds connections open: media streaming,
|
||||||
|
# large up/downloads, or persistent viewer/streaming sessions. Both the primary
|
||||||
|
# and the SSE backend are tuned long-lived here (no http-server-close,
|
||||||
|
# http-no-delay, 6h server/tunnel/keep-alive timeouts).
|
||||||
|
#
|
||||||
|
# Compare hap_backend_websocket.tpl, which keeps the PRIMARY backend standard
|
||||||
|
# and only makes the -sse-backend long-lived. Pick this one when the main path
|
||||||
|
# itself needs long-lived connections, not just an SSE side-channel.
|
||||||
|
backend {{ name }}-backend
|
||||||
|
no option http-server-close
|
||||||
|
option http-no-delay
|
||||||
|
timeout server 6h
|
||||||
|
timeout tunnel 6h
|
||||||
|
timeout http-keep-alive 6h
|
||||||
|
option forwardfor
|
||||||
|
http-request add-header X-CLIENT-IP %[var(txn.real_ip)]
|
||||||
|
http-request set-header X-Real-IP %[var(txn.real_ip)]
|
||||||
|
http-request set-header X-Forwarded-For %[var(txn.real_ip)]
|
||||||
|
http-request set-header X-Forwarded-Proto https if { ssl_fc }
|
||||||
|
http-request set-header X-Forwarded-Proto http if !{ ssl_fc }
|
||||||
|
{% for server in servers %}
|
||||||
|
server {{ server.server_name }} {{ server.server_address }}:{{ server.server_port }} {{ server.server_options }} resolvers docker_dns init-addr last,libc,none
|
||||||
|
{% endfor %}
|
||||||
|
|
||||||
|
# SSE variant (Accept: text/event-stream / ?action=stream auto-routes here)
|
||||||
|
backend {{ name }}-sse-backend
|
||||||
|
no option http-server-close
|
||||||
|
option http-no-delay
|
||||||
|
timeout server 6h
|
||||||
|
timeout tunnel 6h
|
||||||
|
timeout http-keep-alive 6h
|
||||||
|
option forwardfor
|
||||||
|
http-request add-header X-CLIENT-IP %[var(txn.real_ip)]
|
||||||
|
http-request set-header X-Real-IP %[var(txn.real_ip)]
|
||||||
|
http-request set-header X-Forwarded-For %[var(txn.real_ip)]
|
||||||
|
http-request set-header X-Forwarded-Proto https if { ssl_fc }
|
||||||
|
http-request set-header X-Forwarded-Proto http if !{ ssl_fc }
|
||||||
|
{% for server in servers %}
|
||||||
|
server {{ server.server_name }} {{ server.server_address }}:{{ server.server_port }} {{ server.server_options }} resolvers docker_dns init-addr last,libc,none
|
||||||
|
{% endfor %}
|
||||||
@@ -27,6 +27,23 @@ global
|
|||||||
# SSL and Performance
|
# SSL and Performance
|
||||||
tune.ssl.default-dh-param 2048
|
tune.ssl.default-dh-param 2048
|
||||||
|
|
||||||
|
# HTTP/3 over QUIC. The Debian haproxy package is built against system
|
||||||
|
# OpenSSL via the compatibility shim (USE_QUIC_OPENSSL_COMPAT), which is
|
||||||
|
# not a native QUIC TLS stack. HAProxy therefore rejects `quic*@` binds
|
||||||
|
# unless this opt-in is set. `limited-quic` enables QUIC through the compat
|
||||||
|
# layer (no 0-RTT — that needs quictls/aws-lc or native OpenSSL 3.5 QUIC).
|
||||||
|
# Without this, the quic bind in the frontend fails to start: "this SSL
|
||||||
|
# library does not support the QUIC protocol".
|
||||||
|
limited-quic
|
||||||
|
{%- if cluster_secret %}
|
||||||
|
|
||||||
|
# Stable secret keying QUIC Retry/address-validation tokens. Self-healed
|
||||||
|
# to /etc/haproxy/cluster-secret (named volume) by the manager so it
|
||||||
|
# survives recreates; without it haproxy picks a random one per process
|
||||||
|
# and tokens don't survive reloads (benign, just a startup notice).
|
||||||
|
cluster-secret "{{ cluster_secret }}"
|
||||||
|
{%- endif %}
|
||||||
|
|
||||||
# HTTP/2 protection against Rapid Reset (CVE-2023-44487) and stream abuse
|
# HTTP/2 protection against Rapid Reset (CVE-2023-44487) and stream abuse
|
||||||
tune.h2.fe.max-total-streams 2000
|
tune.h2.fe.max-total-streams 2000
|
||||||
tune.h2.fe.glitches-threshold 50
|
tune.h2.fe.glitches-threshold 50
|
||||||
|
|||||||
@@ -4,6 +4,21 @@ frontend web
|
|||||||
# crt can now be a path, so it will load all .pem files in the path
|
# crt can now be a path, so it will load all .pem files in the path
|
||||||
bind 0.0.0.0:443 ssl crt {{ crt_path }} alpn h2,http/1.1
|
bind 0.0.0.0:443 ssl crt {{ crt_path }} alpn h2,http/1.1
|
||||||
|
|
||||||
|
# HTTP/3 over QUIC (UDP/443). Same cert path as the TCP listener above.
|
||||||
|
# The Debian haproxy package is built +QUIC (QUIC_OPENSSL_COMPAT), so this
|
||||||
|
# is config-only — no source build. Requires UDP/443 published on the
|
||||||
|
# container (`-p 443:443/udp`) and open at the host firewall. `h3` is the
|
||||||
|
# only ALPN QUIC negotiates; h2/http1 stay on the TCP bind above. Sharing
|
||||||
|
# the frontend means all the real-IP, rate-limit, IP-block and Coraza
|
||||||
|
# rules below apply identically to H3 traffic.
|
||||||
|
bind quic4@0.0.0.0:443 ssl crt {{ crt_path }} alpn h3
|
||||||
|
|
||||||
|
# Advertise H3 so browsers upgrade their existing TCP (h2) connection to
|
||||||
|
# QUIC on the next request. `ma` is how long (seconds) the client may
|
||||||
|
# cache the advertisement. http-after-response applies it to every
|
||||||
|
# response, including haproxy-generated ones (blocks, default page).
|
||||||
|
http-after-response set-header alt-svc "h3=\":443\"; ma=86400"
|
||||||
|
|
||||||
# Capture Host header so it appears in httplog output (in %hr field)
|
# Capture Host header so it appears in httplog output (in %hr field)
|
||||||
http-request capture req.hdr(Host) len 64
|
http-request capture req.hdr(Host) len 64
|
||||||
|
|
||||||
@@ -49,6 +64,67 @@ frontend web
|
|||||||
# High error rate: >100 errors in 30s (scanner/fuzzer behavior)
|
# High error rate: >100 errors in 30s (scanner/fuzzer behavior)
|
||||||
http-request tarpit deny_status 403 if { sc_http_err_rate(0) gt 100 } !is_local !is_trusted_ip !is_whitelisted !is_health_check
|
http-request tarpit deny_status 403 if { sc_http_err_rate(0) gt 100 } !is_local !is_trusted_ip !is_whitelisted !is_health_check
|
||||||
|
|
||||||
|
# --- WordPress wp-login.php brute-force protection ---
|
||||||
|
# The generic limits above are deliberately high (media-heavy sites), so a
|
||||||
|
# slow credential-stuffing run (dozens of login POSTs/min) slips under them.
|
||||||
|
# Track POSTs to wp-login.php per real client IP in a DEDICATED 60s table
|
||||||
|
# (sc1 / backend wp_bruteforce, defined in hap_security_tables.tpl) and
|
||||||
|
# tarpit once an IP exceeds 30/min. Only login POSTs are counted — GETs of
|
||||||
|
# the login form, normal browsing, and the handful of POSTs a legit user
|
||||||
|
# makes are unaffected; an offending IP can still browse, just not keep
|
||||||
|
# hammering login. path_end also covers subdirectory WP installs. Honors the
|
||||||
|
# same whitelist (RFC1918 / trusted_ips.list / trusted_ips.map).
|
||||||
|
acl wp_login_path path_end /wp-login.php
|
||||||
|
http-request track-sc1 var(txn.real_ip) table wp_bruteforce if METH_POST wp_login_path
|
||||||
|
http-request tarpit deny_status 429 if METH_POST wp_login_path { sc_http_req_rate(1) gt 30 } !is_local !is_trusted_ip !is_whitelisted
|
||||||
|
|
||||||
|
# --- WordPress wp-login.php "must-load-the-form-first" cookie challenge ---
|
||||||
|
# Defeats DISTRIBUTED credential-stuffing (hundreds of thousands of unique
|
||||||
|
# IPs, each low-and-slow, so the per-IP rule above can't see them). Such
|
||||||
|
# bots POST straight to /wp-login.php without ever GETting the form — on
|
||||||
|
# these sites the login POST:GET ratio is ~15:1. We hand out a cookie when
|
||||||
|
# the form is actually fetched (GET) and require it on POST; direct-POST
|
||||||
|
# bots lack it and are denied AT THE EDGE before reaching PHP. Real logins
|
||||||
|
# are unaffected — WordPress login already requires loading the page and
|
||||||
|
# accepting cookies. Immediate deny (NOT tarpit) — under a 300k-POST flood,
|
||||||
|
# holding tarpit connections would exhaust HAProxy. Honors the whitelist.
|
||||||
|
# Mark login-form GETs at REQUEST time (method/path are reliably evaluable
|
||||||
|
# here; in the response phase they are not) so the cookie is emitted on the
|
||||||
|
# form's own response.
|
||||||
|
http-request set-var(txn.wp_login_form) int(1) if METH_GET wp_login_path
|
||||||
|
http-after-response add-header set-cookie "whplc=1; Path=/; Max-Age=1800; HttpOnly; Secure; SameSite=Lax" if { var(txn.wp_login_form) -m found }
|
||||||
|
acl has_login_cookie req.cook(whplc) -m found
|
||||||
|
http-request deny deny_status 403 if METH_POST wp_login_path !has_login_cookie !is_local !is_trusted_ip !is_whitelisted
|
||||||
|
|
||||||
|
# WordPress REST batch endpoint lockdown ("wp2shell": CVE-2026-63030 +
|
||||||
|
# CVE-2026-60137). Chaining a core SQL injection with REST batch-route
|
||||||
|
# confusion gives unauthenticated RCE on WP 6.9.0-6.9.4 and 7.0.0-7.0.1
|
||||||
|
# (fixed in 6.9.5 / 7.0.2). Exploits are public and were used against this
|
||||||
|
# fleet on 2026-07-19/20; one site was compromised via this path before
|
||||||
|
# patching. This is a virtual patch: it does not repair the vulnerable
|
||||||
|
# application logic, it only removes reachability, so it stays until every
|
||||||
|
# site is confirmed on a fixed release.
|
||||||
|
#
|
||||||
|
# Both routing forms must be covered -- a rule matching only the pretty
|
||||||
|
# permalink path leaves the ?rest_route= fallback wide open, and urlp()
|
||||||
|
# does not URL-decode, hence the third ACL for the %2F spelling.
|
||||||
|
#
|
||||||
|
# Anonymous-only. batch/v1 is used legitimately by the block editor for
|
||||||
|
# multi-entity saves, so a blanket deny would break wp-admin for real
|
||||||
|
# users; requiring a wordpress_logged_in_* cookie costs them nothing.
|
||||||
|
# req.cook() needs an exact name and WordPress suffixes a per-site hash,
|
||||||
|
# so this substring-matches the raw Cookie header instead.
|
||||||
|
#
|
||||||
|
# Immediate deny, not tarpit -- holding connections open helps an attacker
|
||||||
|
# who is already scripting this. Honors the same whitelist as above.
|
||||||
|
acl wp_batch_path path_beg /wp-json/batch/v1
|
||||||
|
acl wp_batch_route urlp(rest_route) -i -m beg /batch/v1
|
||||||
|
acl wp_batch_route_enc query -i -m sub rest_route=%2Fbatch%2Fv1
|
||||||
|
acl has_wp_logged_in req.hdr(Cookie) -i -m sub wordpress_logged_in_
|
||||||
|
http-request deny deny_status 403 if wp_batch_path !has_wp_logged_in !is_local !is_trusted_ip !is_whitelisted
|
||||||
|
http-request deny deny_status 403 if wp_batch_route !has_wp_logged_in !is_local !is_trusted_ip !is_whitelisted
|
||||||
|
http-request deny deny_status 403 if wp_batch_route_enc !has_wp_logged_in !is_local !is_trusted_ip !is_whitelisted
|
||||||
|
|
||||||
# IP blocking using map file (manual blocks only)
|
# IP blocking using map file (manual blocks only)
|
||||||
# Map file format: /etc/haproxy/blocked_ips.map contains "<ip_or_cidr> 1" per line
|
# Map file format: /etc/haproxy/blocked_ips.map contains "<ip_or_cidr> 1" per line
|
||||||
# Runtime updates: echo "add map #0 IP_ADDRESS 1" | socat stdio /var/run/haproxy.sock
|
# Runtime updates: echo "add map #0 IP_ADDRESS 1" | socat stdio /var/run/haproxy.sock
|
||||||
|
|||||||
@@ -6,3 +6,11 @@ frontend stats
|
|||||||
stats refresh 30s
|
stats refresh 30s
|
||||||
stats show-legends
|
stats show-legends
|
||||||
stats show-node
|
stats show-node
|
||||||
|
|
||||||
|
# Dedicated stick-table for WordPress wp-login.php brute-force tracking.
|
||||||
|
# Tracked via track-sc1 from the `web` frontend (hap_listener.tpl); counts only
|
||||||
|
# login POSTs per real client IP over a 60s window. Separate from the generic
|
||||||
|
# sc0 connection/rate table so the login-attempt threshold is independent of
|
||||||
|
# the (much higher) flood thresholds.
|
||||||
|
backend wp_bruteforce
|
||||||
|
stick-table type ip size 100k expire 30m store http_req_rate(60s)
|
||||||
Reference in New Issue
Block a user