Compare commits
10
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9d16151120 | ||
|
|
b892438070 | ||
|
|
2a2b9739fc | ||
|
|
7732e2a2ff | ||
|
|
89c74c10cf | ||
|
|
1b557b9931 | ||
|
|
6ced2f8797 | ||
|
|
3917b6d1ae | ||
|
|
d9cc5311de | ||
|
|
f1c1954378 |
@@ -36,11 +36,26 @@ jobs:
|
||||
username: shadowdao
|
||||
password: ${{ secrets.GHCR_TOKEN }}
|
||||
|
||||
# Read the human-readable release version from the VERSION file so every
|
||||
# build is pinnable for rollback (alongside the immutable git SHA). Bump
|
||||
# VERSION (YYYY.MM.N) in the same commit as a release-worthy change.
|
||||
- name: Read version
|
||||
id: ver
|
||||
run: echo "version=$(cat VERSION)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Build Image
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
platforms: linux/amd64
|
||||
push: true
|
||||
build-args: |
|
||||
VERSION=${{ steps.ver.outputs.version }}
|
||||
# Three tags per registry: :latest (moving), :<version> (human-readable
|
||||
# release), :<sha> (immutable, guaranteed-unique rollback target).
|
||||
tags: |
|
||||
repo.anhonesthost.net/cloud-hosting-platform/haproxy-manager-base:latest
|
||||
repo.anhonesthost.net/cloud-hosting-platform/haproxy-manager-base:${{ steps.ver.outputs.version }}
|
||||
repo.anhonesthost.net/cloud-hosting-platform/haproxy-manager-base:${{ gitea.sha }}
|
||||
ghcr.io/shadowdao/haproxy-manager-base:latest
|
||||
ghcr.io/shadowdao/haproxy-manager-base:${{ steps.ver.outputs.version }}
|
||||
ghcr.io/shadowdao/haproxy-manager-base:${{ gitea.sha }}
|
||||
|
||||
+7
-1
@@ -14,9 +14,13 @@ FROM repo.anhonesthost.net/cloud-hosting-platform/python:3.12-slim
|
||||
# sidebar; pointing at the public GitHub mirror enables that linking. The
|
||||
# canonical source-of-truth git remote is still Gitea, but Gitea's registry
|
||||
# doesn't consume this label, so there's no contention.
|
||||
# Stamped from the VERSION file by CI (build-arg) so `docker inspect` reports
|
||||
# what's running on any host. Defaults to "dev" for local/manual builds.
|
||||
ARG VERSION=dev
|
||||
LABEL org.opencontainers.image.title="haproxy-manager-base" \
|
||||
org.opencontainers.image.description="HAProxy management API with Let's Encrypt automation, Coraza WAF integration, and template-driven config" \
|
||||
org.opencontainers.image.source="https://github.com/shadowdao/haproxy-manager-base" \
|
||||
org.opencontainers.image.version="${VERSION}" \
|
||||
org.opencontainers.image.licenses="MIT"
|
||||
|
||||
RUN apt update -y && apt dist-upgrade -y && apt install socat haproxy cron certbot curl jq net-tools -y && apt clean && rm -rf /var/lib/apt/lists/*
|
||||
@@ -43,7 +47,9 @@ RUN mkdir -p /var/spool/cron/crontabs && \
|
||||
echo '0 */12 * * * /haproxy/scripts/renew-certificates.sh >> /var/log/haproxy-manager.log 2>&1' >> /var/spool/cron/crontabs/root && \
|
||||
chmod 600 /var/spool/cron/crontabs/root && \
|
||||
chown root:crontab /var/spool/cron/crontabs/root
|
||||
EXPOSE 80 443 8000
|
||||
# 443/udp carries HTTP/3 (QUIC). EXPOSE is documentation only — the container
|
||||
# must still be run with `-p 443:443/udp` for the UDP listener to be reachable.
|
||||
EXPOSE 80 443 443/udp 8000
|
||||
# Add health check
|
||||
HEALTHCHECK --interval=30s --timeout=10s --start-period=60s --retries=3 \
|
||||
CMD curl -sf --max-time 5 http://localhost:8000/health && curl -s --max-time 5 -o /dev/null http://localhost/ || exit 1
|
||||
|
||||
@@ -6,10 +6,10 @@ A Flask-based API service for managing HAProxy configurations with dynamic SSL c
|
||||
To run the container:
|
||||
```bash
|
||||
# Without API key authentication (default)
|
||||
docker run -d -p 80:80 -p 443:443 -p 8000:8000 -v lets-encrypt:/etc/letsencrypt -v haproxy:/etc/haproxy --name haproxy-manager your-registry.example.com/cloud-hosting-platform/haproxy-manager-base:latest
|
||||
docker run -d -p 80:80 -p 443:443 -p 443:443/udp -p 8000:8000 -v lets-encrypt:/etc/letsencrypt -v haproxy:/etc/haproxy --name haproxy-manager your-registry.example.com/cloud-hosting-platform/haproxy-manager-base:latest
|
||||
|
||||
# With API key authentication (recommended for production)
|
||||
docker run -d -p 80:80 -p 443:443 -p 8000:8000 -v lets-encrypt:/etc/letsencrypt -v haproxy:/etc/haproxy -e HAPROXY_API_KEY=your-secure-api-key-here --name haproxy-manager your-registry.example.com/cloud-hosting-platform/haproxy-manager-base:latest
|
||||
docker run -d -p 80:80 -p 443:443 -p 443:443/udp -p 8000:8000 -v lets-encrypt:/etc/letsencrypt -v haproxy:/etc/haproxy -e HAPROXY_API_KEY=your-secure-api-key-here --name haproxy-manager your-registry.example.com/cloud-hosting-platform/haproxy-manager-base:latest
|
||||
```
|
||||
|
||||
## Features
|
||||
@@ -394,7 +394,7 @@ You can customize the default page by setting environment variables:
|
||||
|
||||
```bash
|
||||
docker run -d \
|
||||
-p 80:80 -p 443:443 -p 8000:8000 \
|
||||
-p 80:80 -p 443:443 -p 443:443/udp -p 8000:8000 \
|
||||
-v lets-encrypt:/etc/letsencrypt \
|
||||
-v haproxy:/etc/haproxy \
|
||||
-e HAPROXY_API_KEY=your-secure-api-key-here \
|
||||
@@ -411,7 +411,7 @@ docker run -d \
|
||||
```bash
|
||||
# Start container with API key
|
||||
docker run -d \
|
||||
-p 80:80 -p 443:443 -p 8000:8000 \
|
||||
-p 80:80 -p 443:443 -p 443:443/udp -p 8000:8000 \
|
||||
-v lets-encrypt:/etc/letsencrypt \
|
||||
-v haproxy:/etc/haproxy \
|
||||
-e HAPROXY_API_KEY=your-secure-api-key-here \
|
||||
|
||||
+422
-65
@@ -12,12 +12,45 @@ from datetime import datetime, timedelta
|
||||
import json
|
||||
import ipaddress
|
||||
import shutil
|
||||
import stat
|
||||
import tempfile
|
||||
import threading
|
||||
import time
|
||||
import re
|
||||
import fcntl
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Bounded subprocess execution (incident 2026-07-07)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Every external command this manager runs — certbot ACME issuance/renewal,
|
||||
# `socat` reloads over the haproxy admin socket, `haproxy -c` validation — is a
|
||||
# potential hang. The management API runs under gunicorn gthread workers, and a
|
||||
# subprocess.run() with NO timeout blocks its worker thread forever if the
|
||||
# command stalls (e.g. an ACME/upstream that stops responding mid-read).
|
||||
# gunicorn's --timeout does not rescue this: for gthread it only kills a worker
|
||||
# whose *main* thread stops heart-beating, but the main thread keeps polling
|
||||
# while pool threads are wedged. Enough stalled calls exhaust the 4-thread pool
|
||||
# and the whole API stops responding — "healthy" health-check, every request
|
||||
# 30s-timeouts — which is exactly what stalled WHP site updates on 2026-07-07.
|
||||
#
|
||||
# Fix: give EVERY subprocess.run() a default timeout unless the caller passes
|
||||
# one explicitly. On expiry Python kills the child and raises
|
||||
# subprocess.TimeoutExpired (a subclass of Exception); the existing per-endpoint
|
||||
# try/except turns that into a clean error AND releases the worker thread.
|
||||
# Bounding by default (instead of editing ~30 call sites) means no site can be
|
||||
# missed and any future call is protected automatically.
|
||||
DEFAULT_SUBPROCESS_TIMEOUT = int(os.environ.get('HAPROXY_MGR_SUBPROCESS_TIMEOUT', '180'))
|
||||
_unbounded_subprocess_run = subprocess.run
|
||||
|
||||
|
||||
def _bounded_subprocess_run(*args, **kwargs):
|
||||
if kwargs.get('timeout') is None:
|
||||
kwargs['timeout'] = DEFAULT_SUBPROCESS_TIMEOUT
|
||||
return _unbounded_subprocess_run(*args, **kwargs)
|
||||
|
||||
|
||||
subprocess.run = _bounded_subprocess_run
|
||||
|
||||
app = Flask(__name__)
|
||||
|
||||
# Default page server (port 8080) — served to HAProxy clients whose request hit
|
||||
@@ -73,8 +106,18 @@ HAPROXY_CONFIG_PATH = '/etc/haproxy/haproxy.cfg'
|
||||
HAPROXY_BACKUP_PATH = '/etc/haproxy/haproxy.cfg.backup'
|
||||
BLOCKED_IPS_MAP_PATH = '/etc/haproxy/blocked_ips.map'
|
||||
BLOCKED_IPS_MAP_BACKUP_PATH = '/etc/haproxy/blocked_ips.map.backup'
|
||||
# Coraza SPOE engine file. `haproxy -c` parses this too (the frontend's
|
||||
# `filter spoe engine coraza config <path>` line points at it), so it is part
|
||||
# of the same restorable config set as haproxy.cfg — rolling back haproxy.cfg
|
||||
# while leaving a broken coraza-spoe.cfg behind still fails validation.
|
||||
CORAZA_SPOE_CONFIG_PATH = '/etc/haproxy/coraza-spoe.cfg'
|
||||
CORAZA_SPOE_BACKUP_PATH = '/etc/haproxy/coraza-spoe.cfg.backup'
|
||||
HAPROXY_SOCKET_PATH = '/var/run/haproxy.sock'
|
||||
SSL_CERTS_DIR = '/etc/haproxy/certs'
|
||||
# Stable per-host secret for QUIC Retry/address-validation tokens. Lives in the
|
||||
# /etc/haproxy named volume so it survives container recreates; self-healed on
|
||||
# first config render. See get_or_create_cluster_secret().
|
||||
CLUSTER_SECRET_PATH = '/etc/haproxy/cluster-secret'
|
||||
API_KEY = os.environ.get('HAPROXY_API_KEY') # Optional API key for authentication
|
||||
|
||||
# Setup logging
|
||||
@@ -807,10 +850,12 @@ def renew_certificates():
|
||||
# Defensive: clear any stale lock left by a SIGKILLed prior run.
|
||||
clear_stale_certbot_locks()
|
||||
|
||||
# Run certbot renew
|
||||
# Run certbot renew. Explicit long timeout (overrides the module
|
||||
# default): `renew` walks every lineage and can legitimately make many
|
||||
# ACME round-trips when several certs are actually due.
|
||||
result = subprocess.run([
|
||||
'certbot', 'renew', '--quiet'
|
||||
], capture_output=True, text=True)
|
||||
], capture_output=True, text=True, timeout=900)
|
||||
|
||||
if result.returncode == 0:
|
||||
# Check if any certificates were renewed
|
||||
@@ -1687,6 +1732,45 @@ def dns_challenge_verify():
|
||||
log_operation('dns_challenge_verify', False, str(e))
|
||||
return jsonify({'success': False, 'error': str(e)}), 500
|
||||
|
||||
def get_or_create_cluster_secret():
|
||||
"""Return a stable secret for QUIC token derivation, generating it once.
|
||||
|
||||
HAProxy uses `cluster-secret` to key QUIC Retry/address-validation tokens.
|
||||
Without a stable value it picks a random one each (re)start and logs a
|
||||
notice; tokens then don't survive reloads. We persist one in the
|
||||
/etc/haproxy named volume so it's stable across container recreates.
|
||||
Exclusive-create avoids a race if two renders run concurrently. Failure to
|
||||
read/write is non-fatal: we fall back to an empty string and the template
|
||||
simply omits the directive (HAProxy reverts to its random-per-process
|
||||
behaviour), so QUIC still works.
|
||||
"""
|
||||
try:
|
||||
if os.path.exists(CLUSTER_SECRET_PATH):
|
||||
with open(CLUSTER_SECRET_PATH, 'r') as f:
|
||||
secret = f.read().strip()
|
||||
if secret:
|
||||
return secret
|
||||
# Generate and persist exclusively (0600). hex => config-safe charset.
|
||||
secret = os.urandom(32).hex()
|
||||
fd = os.open(CLUSTER_SECRET_PATH, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
||||
try:
|
||||
os.write(fd, secret.encode())
|
||||
finally:
|
||||
os.close(fd)
|
||||
logger.info("Generated new QUIC cluster-secret at %s", CLUSTER_SECRET_PATH)
|
||||
return secret
|
||||
except FileExistsError:
|
||||
# Lost the create race — another render just wrote it; read it back.
|
||||
try:
|
||||
with open(CLUSTER_SECRET_PATH, 'r') as f:
|
||||
return f.read().strip()
|
||||
except Exception as e:
|
||||
logger.error("Failed to read cluster-secret after race: %s", e)
|
||||
return ''
|
||||
except Exception as e:
|
||||
logger.error("Failed to get/create cluster-secret: %s", e)
|
||||
return ''
|
||||
|
||||
def generate_config():
|
||||
try:
|
||||
conn = sqlite3.connect(DB_FILE)
|
||||
@@ -1718,6 +1802,21 @@ def generate_config():
|
||||
|
||||
config_parts = []
|
||||
|
||||
# Snapshot the last-known-good config BEFORE anything below touches a
|
||||
# file in /etc/haproxy. Everything this function writes (haproxy.cfg,
|
||||
# blocked_ips.map, coraza-spoe.cfg) is validated as one set by
|
||||
# `haproxy -c`, so the rollback point has to predate the first of them.
|
||||
# Taking it here (rather than inside reload_haproxy_safely(), which runs
|
||||
# after the writes) is what makes rollback real - see create_backup().
|
||||
backup_ok, backup_status = create_backup()
|
||||
if not backup_ok:
|
||||
# Could not even attempt a snapshot (I/O error). Writing a new
|
||||
# config now would leave us with no way back, so refuse.
|
||||
raise Exception(
|
||||
"Refusing to regenerate config: failed to back up the current "
|
||||
"configuration, so a failed change could not be rolled back"
|
||||
)
|
||||
|
||||
# Optional Coraza WAF integration. When HAPROXY_CORAZA_SPOE_BACKEND is
|
||||
# set on the haproxy-manager container, we render an extra TCP backend
|
||||
# pointing at a coraza-spoa sidecar AND inject a `filter spoe ...` line
|
||||
@@ -1749,7 +1848,9 @@ def generate_config():
|
||||
logger.error(f"Failed to create {suspended_list_path}: {e}")
|
||||
|
||||
# Add Haproxy Default Headers
|
||||
default_headers = template_env.get_template('hap_header.tpl').render()
|
||||
default_headers = template_env.get_template('hap_header.tpl').render(
|
||||
cluster_secret = get_or_create_cluster_secret(),
|
||||
)
|
||||
config_parts.append(default_headers)
|
||||
|
||||
# Update blocked IPs map file first
|
||||
@@ -1810,7 +1911,11 @@ def generate_config():
|
||||
# First pass: exact domain ACLs (higher priority - evaluated first)
|
||||
for domain in exact_domains:
|
||||
if not domain['backend_name']:
|
||||
logger.warning(f"Skipping domain {domain['domain']} - no backend name")
|
||||
# Expected for domains registered without a proxy backend (e.g. the
|
||||
# panel's own hostname, present only for certificate management).
|
||||
# Log at INFO — not WARNING — so it doesn't trip log monitors as an
|
||||
# error; it recurs on every generate_config by design.
|
||||
logger.info(f"Skipping domain {domain['domain']} - no proxy backend (cert/management-only)")
|
||||
continue
|
||||
|
||||
try:
|
||||
@@ -1829,7 +1934,8 @@ def generate_config():
|
||||
# Second pass: wildcard domain ACLs (lower priority - evaluated after exact matches)
|
||||
for domain in wildcard_domains:
|
||||
if not domain['backend_name']:
|
||||
logger.warning(f"Skipping wildcard domain {domain['domain']} - no backend name")
|
||||
# See note above — INFO, not WARNING; expected for cert/management-only domains.
|
||||
logger.info(f"Skipping wildcard domain {domain['domain']} - no proxy backend (cert/management-only)")
|
||||
continue
|
||||
|
||||
try:
|
||||
@@ -1900,25 +2006,21 @@ backend default-backend
|
||||
# how the file was authored.
|
||||
if not coraza_spoe_cfg.endswith('\n'):
|
||||
coraza_spoe_cfg += '\n'
|
||||
coraza_spoe_path = '/etc/haproxy/coraza-spoe.cfg'
|
||||
with open(coraza_spoe_path, 'w') as f:
|
||||
f.write(coraza_spoe_cfg)
|
||||
logger.info(f"Coraza SPOE engine config written to {coraza_spoe_path} "
|
||||
write_config_atomically(CORAZA_SPOE_CONFIG_PATH, coraza_spoe_cfg)
|
||||
logger.info(f"Coraza SPOE engine config written to "
|
||||
f"{CORAZA_SPOE_CONFIG_PATH} "
|
||||
f"(SPOA target: {coraza_spoe_backend})")
|
||||
|
||||
# Write complete configuration to tmp
|
||||
temp_config_path = "/etc/haproxy/haproxy.cfg"
|
||||
|
||||
config_content = '\n'.join(config_parts)
|
||||
logger.debug("Generated HAProxy configuration")
|
||||
|
||||
# Write complete configuration to tmp
|
||||
# Write new configuration to file
|
||||
with open(HAPROXY_CONFIG_PATH, 'w') as f:
|
||||
f.write(config_content)
|
||||
|
||||
# Write new configuration to file (atomically - a truncated haproxy.cfg
|
||||
# is as fatal as an invalid one). The rollback point was taken above,
|
||||
# before this write.
|
||||
write_config_atomically(HAPROXY_CONFIG_PATH, config_content)
|
||||
|
||||
# Use safe reload with validation and rollback
|
||||
success, message = reload_haproxy_safely()
|
||||
success, message = reload_haproxy_safely(backup_status=backup_status)
|
||||
if success:
|
||||
logger.info("Configuration generated and HAProxy reloaded safely")
|
||||
log_operation('generate_config', True, 'Configuration generated and HAProxy reloaded safely')
|
||||
@@ -1935,63 +2037,306 @@ backend default-backend
|
||||
traceback.print_exc()
|
||||
raise
|
||||
|
||||
def create_backup():
|
||||
"""Create backup of current config and map files"""
|
||||
# ---------------------------------------------------------------------------
|
||||
# Config backup / rollback
|
||||
# ---------------------------------------------------------------------------
|
||||
# Rollback only works if the backup predates the write it is supposed to undo.
|
||||
# Until 2026-08 create_backup() ran from inside reload_haproxy_safely(), i.e.
|
||||
# AFTER generate_config() had already overwritten haproxy.cfg — so the "backup"
|
||||
# was a copy of the new (possibly broken) config and restore_backup() restored
|
||||
# the same broken bytes. The advertised rollback was a no-op and a fatal
|
||||
# haproxy.cfg persisted on disk, where start_haproxy() refuses to launch (the
|
||||
# June 2026 missing-template incident). create_backup() must now be called by
|
||||
# the writer, BEFORE the first byte is written.
|
||||
|
||||
# Statuses returned by create_backup() that mean a rollback target exists.
|
||||
_ROLLBACK_AVAILABLE_STATUSES = ('created', 'kept_previous')
|
||||
|
||||
|
||||
def _files_identical(path_a, path_b):
|
||||
"""Byte-compare two files.
|
||||
|
||||
Deliberately not filecmp.cmp(): it memoises on (size, mtime), and
|
||||
shutil.copy2() preserves mtime, so a stale cache entry could report a
|
||||
changed config as unchanged. These files are small; read them.
|
||||
"""
|
||||
try:
|
||||
if os.path.exists(HAPROXY_CONFIG_PATH):
|
||||
shutil.copy2(HAPROXY_CONFIG_PATH, HAPROXY_BACKUP_PATH)
|
||||
if os.path.exists(BLOCKED_IPS_MAP_PATH):
|
||||
shutil.copy2(BLOCKED_IPS_MAP_PATH, BLOCKED_IPS_MAP_BACKUP_PATH)
|
||||
logger.info("Backups created successfully")
|
||||
return True
|
||||
if os.path.getsize(path_a) != os.path.getsize(path_b):
|
||||
return False
|
||||
with open(path_a, 'rb') as fa, open(path_b, 'rb') as fb:
|
||||
while True:
|
||||
chunk_a = fa.read(65536)
|
||||
chunk_b = fb.read(65536)
|
||||
if chunk_a != chunk_b:
|
||||
return False
|
||||
if not chunk_a:
|
||||
return True
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
def _config_set_matches_backup():
|
||||
"""True if every live config file is byte-identical to its backup copy.
|
||||
|
||||
After a successful reload the live set has already been recorded as
|
||||
known-good (see promote_current_config_to_backup()), which is the common
|
||||
case at the start of the next generation. Recognising it lets create_backup()
|
||||
skip both the re-validation and the copy - worth doing because
|
||||
`haproxy -c` on an edge with hundreds of certificates is not free and
|
||||
generate_config() runs synchronously inside customer-facing API calls.
|
||||
"""
|
||||
for live_path, backup_path in _config_backup_pairs():
|
||||
if os.path.exists(live_path) != os.path.exists(backup_path):
|
||||
return False
|
||||
if (os.path.exists(live_path)
|
||||
and not _files_identical(live_path, backup_path)):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _config_backup_pairs():
|
||||
"""(live, backup) pairs forming one restorable config set.
|
||||
|
||||
Built at call time rather than at import so the module-level path constants
|
||||
stay patchable (tests, alternate deployments).
|
||||
"""
|
||||
return (
|
||||
(HAPROXY_CONFIG_PATH, HAPROXY_BACKUP_PATH),
|
||||
(BLOCKED_IPS_MAP_PATH, BLOCKED_IPS_MAP_BACKUP_PATH),
|
||||
(CORAZA_SPOE_CONFIG_PATH, CORAZA_SPOE_BACKUP_PATH),
|
||||
)
|
||||
|
||||
|
||||
def write_config_atomically(path, content):
|
||||
"""Write content to path via temp file + rename.
|
||||
|
||||
A half-written haproxy.cfg (disk full, container killed mid-write) is just
|
||||
as fatal as an invalid one and is invisible to the caller. os.replace() is
|
||||
atomic within a filesystem, so the file on disk is always either the whole
|
||||
old config or the whole new one — never a truncated hybrid. This also keeps
|
||||
the "existing config is already broken" case from being self-inflicted.
|
||||
"""
|
||||
directory = os.path.dirname(path) or '.'
|
||||
# Preserve the mode of the file we are replacing; mkstemp defaults to 0600
|
||||
# and HAProxy config files are conventionally 0644.
|
||||
try:
|
||||
mode = stat.S_IMODE(os.stat(path).st_mode)
|
||||
except OSError:
|
||||
mode = 0o644
|
||||
fd, tmp_path = tempfile.mkstemp(
|
||||
dir=directory, prefix=os.path.basename(path) + '.', suffix='.tmp'
|
||||
)
|
||||
try:
|
||||
with os.fdopen(fd, 'w') as f:
|
||||
f.write(content)
|
||||
f.flush()
|
||||
os.fsync(f.fileno())
|
||||
os.chmod(tmp_path, mode)
|
||||
os.replace(tmp_path, path)
|
||||
except Exception:
|
||||
try:
|
||||
os.unlink(tmp_path)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
|
||||
|
||||
def create_backup(require_valid=True):
|
||||
"""Snapshot the CURRENT on-disk config set as the rollback point.
|
||||
|
||||
MUST be called BEFORE the new configuration is written — see the module
|
||||
comment above. Calling it afterwards silently disarms rollback.
|
||||
|
||||
require_valid=True (default) refuses to promote a config that HAProxy
|
||||
already rejects. Backing up a broken config would make "rollback" mean
|
||||
"restore a different broken config"; keeping the older, validated backup
|
||||
instead means a rollback always lands on something HAProxy will actually
|
||||
start with. Cost is one `haproxy -c` run per config generation.
|
||||
|
||||
Returns (ok, status):
|
||||
ok=False, status='error' - the copy itself failed; caller decides.
|
||||
status='created' - backup now holds the current config.
|
||||
status='kept_previous' - current config missing or invalid; the
|
||||
existing (older, good) backup was kept.
|
||||
status='unavailable' - nothing to roll back to at all (first
|
||||
run, or broken config and no prior
|
||||
backup). Rollback is NOT possible.
|
||||
"""
|
||||
try:
|
||||
snapshot_ok = True
|
||||
reason = None
|
||||
|
||||
if not os.path.exists(HAPROXY_CONFIG_PATH):
|
||||
snapshot_ok = False
|
||||
reason = 'no existing HAProxy config on disk (first run?)'
|
||||
elif _config_set_matches_backup():
|
||||
# The backup already IS the current config, recorded when it last
|
||||
# loaded successfully. Nothing to copy and nothing to re-validate.
|
||||
logger.debug("Config backup already matches the live config")
|
||||
return True, 'created'
|
||||
elif require_valid:
|
||||
status, msg = validate_config_file(HAPROXY_CONFIG_PATH)
|
||||
if status == 'invalid':
|
||||
snapshot_ok = False
|
||||
reason = f'current config on disk does not validate: {msg}'
|
||||
elif status == 'unavailable':
|
||||
# The validator itself could not run (no haproxy binary, etc).
|
||||
# That is NOT evidence the config is bad, and refusing to back
|
||||
# up would leave us with no rollback target at all, so fall
|
||||
# back to last-written semantics and say so loudly.
|
||||
logger.warning(
|
||||
f"Could not verify current config before backup ({msg}); "
|
||||
"backing it up unverified"
|
||||
)
|
||||
|
||||
if not snapshot_ok:
|
||||
if os.path.exists(HAPROXY_BACKUP_PATH):
|
||||
logger.warning(
|
||||
f"Not refreshing config backup: {reason}. Keeping the "
|
||||
f"existing backup at {HAPROXY_BACKUP_PATH} as the rollback "
|
||||
"target."
|
||||
)
|
||||
return True, 'kept_previous'
|
||||
logger.error(
|
||||
f"No config backup could be taken: {reason}, and no previous "
|
||||
f"backup exists at {HAPROXY_BACKUP_PATH}. ROLLBACK IS NOT "
|
||||
"AVAILABLE for this configuration change."
|
||||
)
|
||||
return True, 'unavailable'
|
||||
|
||||
for live_path, backup_path in _config_backup_pairs():
|
||||
if os.path.exists(live_path):
|
||||
shutil.copy2(live_path, backup_path)
|
||||
logger.info("Backup of last-known-good config created successfully")
|
||||
return True, 'created'
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create backup: {e}")
|
||||
return False
|
||||
return False, 'error'
|
||||
|
||||
def restore_backup():
|
||||
"""Restore from backup files"""
|
||||
def promote_current_config_to_backup():
|
||||
"""Record the live config as the known-good rollback target.
|
||||
|
||||
Called ONLY after the config has both validated and been loaded by HAProxy,
|
||||
so "backup" really means "the last configuration this box was running".
|
||||
Must never be called before a reload attempt: doing so would make the
|
||||
backup a copy of the config we may still have to roll back from - the same
|
||||
class of bug as backing up after the write.
|
||||
|
||||
Without this, a box whose very first generation succeeded has no rollback
|
||||
target at all until its second successful generation, and any corruption of
|
||||
haproxy.cfg in between leaves nothing to recover to.
|
||||
"""
|
||||
try:
|
||||
if os.path.exists(HAPROXY_BACKUP_PATH):
|
||||
shutil.copy2(HAPROXY_BACKUP_PATH, HAPROXY_CONFIG_PATH)
|
||||
if os.path.exists(BLOCKED_IPS_MAP_BACKUP_PATH):
|
||||
shutil.copy2(BLOCKED_IPS_MAP_BACKUP_PATH, BLOCKED_IPS_MAP_PATH)
|
||||
logger.info("Backups restored successfully")
|
||||
for live_path, backup_path in _config_backup_pairs():
|
||||
if os.path.exists(live_path):
|
||||
shutil.copy2(live_path, backup_path)
|
||||
logger.debug("Known-good config backup updated after successful reload")
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to restore backup: {e}")
|
||||
# Non-fatal: the config is live and working, we just failed to record
|
||||
# it. Loud, because the next change now has a staler rollback target.
|
||||
logger.error(f"Failed to record known-good config backup: {e}")
|
||||
return False
|
||||
|
||||
def validate_haproxy_config():
|
||||
"""Validate HAProxy configuration file"""
|
||||
try:
|
||||
result = subprocess.run(['haproxy', '-c', '-f', HAPROXY_CONFIG_PATH],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode == 0:
|
||||
logger.info("HAProxy configuration validation passed")
|
||||
return True, None
|
||||
else:
|
||||
error_msg = f"HAProxy configuration validation failed: {result.stderr}"
|
||||
logger.error(error_msg)
|
||||
return False, error_msg
|
||||
except Exception as e:
|
||||
error_msg = f"Error validating HAProxy config: {e}"
|
||||
logger.error(error_msg)
|
||||
return False, error_msg
|
||||
|
||||
def reload_haproxy_safely():
|
||||
"""Safely reload HAProxy with validation and rollback"""
|
||||
def restore_backup():
|
||||
"""Restore the backed-up config set over the live files.
|
||||
|
||||
Returns (restored, message). restored=False means NOTHING was rolled back
|
||||
and the live config is still whatever the failed change left on disk —
|
||||
callers MUST surface that difference, it is the difference between "we
|
||||
recovered" and "this edge is sitting on a config HAProxy will not load".
|
||||
"""
|
||||
if not os.path.exists(HAPROXY_BACKUP_PATH):
|
||||
msg = (f"No config backup at {HAPROXY_BACKUP_PATH} - cannot roll back; "
|
||||
f"{HAPROXY_CONFIG_PATH} still holds the failed configuration")
|
||||
logger.critical(msg)
|
||||
return False, msg
|
||||
try:
|
||||
# Create backup before changes
|
||||
if not create_backup():
|
||||
return False, "Failed to create backup"
|
||||
|
||||
for live_path, backup_path in _config_backup_pairs():
|
||||
if os.path.exists(backup_path):
|
||||
shutil.copy2(backup_path, live_path)
|
||||
msg = f"Configuration restored from backup ({HAPROXY_BACKUP_PATH})"
|
||||
logger.info(msg)
|
||||
return True, msg
|
||||
except Exception as e:
|
||||
msg = (f"Failed to restore backup: {e} - {HAPROXY_CONFIG_PATH} may hold "
|
||||
"a broken configuration")
|
||||
logger.critical(msg)
|
||||
return False, msg
|
||||
|
||||
|
||||
def validate_config_file(config_path):
|
||||
"""Run `haproxy -c` against config_path.
|
||||
|
||||
Returns (status, message) with status one of:
|
||||
'valid' - HAProxy parsed the file successfully
|
||||
'invalid' - HAProxy rejected it (message carries stderr)
|
||||
'unavailable' - the validator could not be run at all (binary missing,
|
||||
timeout, ...). Deliberately distinct from 'invalid':
|
||||
it tells us nothing about the config.
|
||||
"""
|
||||
try:
|
||||
result = subprocess.run(['haproxy', '-c', '-f', config_path],
|
||||
capture_output=True, text=True)
|
||||
except Exception as e:
|
||||
return 'unavailable', f"Error validating HAProxy config: {e}"
|
||||
if result.returncode == 0:
|
||||
return 'valid', None
|
||||
return 'invalid', f"HAProxy configuration validation failed: {result.stderr}"
|
||||
|
||||
|
||||
def validate_haproxy_config():
|
||||
"""Validate the live HAProxy configuration file. Returns (is_valid, error)."""
|
||||
status, message = validate_config_file(HAPROXY_CONFIG_PATH)
|
||||
if status == 'valid':
|
||||
logger.info("HAProxy configuration validation passed")
|
||||
return True, None
|
||||
logger.error(message)
|
||||
return False, message
|
||||
|
||||
def reload_haproxy_safely(backup_status=None):
|
||||
"""Safely reload HAProxy with validation and rollback.
|
||||
|
||||
PRECONDITION: the caller must already have called create_backup() BEFORE
|
||||
writing the new config, and pass the status it returned. This function runs
|
||||
after the new config is on disk, so it cannot take a meaningful backup
|
||||
itself — doing so is exactly the bug this contract exists to prevent.
|
||||
|
||||
backup_status=None means the caller did not take a pre-write backup. We do
|
||||
NOT create one here (that would overwrite a genuinely good backup with the
|
||||
unverified new config); we log it and fall back to whatever backup already
|
||||
exists on disk.
|
||||
"""
|
||||
try:
|
||||
if backup_status is None:
|
||||
logger.error(
|
||||
"reload_haproxy_safely() called without a pre-write backup "
|
||||
"status - rollback will fall back to whatever backup already "
|
||||
"exists on disk. Callers must call create_backup() BEFORE "
|
||||
"writing the new configuration."
|
||||
)
|
||||
elif backup_status not in _ROLLBACK_AVAILABLE_STATUSES:
|
||||
logger.warning(
|
||||
f"Proceeding with reload without a rollback target "
|
||||
f"(backup status: {backup_status})"
|
||||
)
|
||||
|
||||
# Validate new configuration
|
||||
is_valid, error_msg = validate_haproxy_config()
|
||||
if not is_valid:
|
||||
# Restore backup on validation failure
|
||||
restore_backup()
|
||||
restored, restore_msg = restore_backup()
|
||||
if not restored:
|
||||
logger.critical(
|
||||
"Config validation failed AND rollback was not possible - "
|
||||
f"{HAPROXY_CONFIG_PATH} holds an invalid configuration that "
|
||||
"HAProxy will refuse to start with"
|
||||
)
|
||||
return False, (f"Config validation failed: {error_msg} | "
|
||||
f"ROLLBACK FAILED: {restore_msg}")
|
||||
return False, f"Config validation failed: {error_msg}"
|
||||
|
||||
|
||||
# Attempt reload
|
||||
if is_process_running('haproxy'):
|
||||
# Use HAProxy stats socket for graceful reload
|
||||
@@ -2010,20 +2355,28 @@ def reload_haproxy_safely():
|
||||
|
||||
if reload_result.returncode == 0:
|
||||
logger.info("HAProxy reloaded successfully")
|
||||
# Now - and only now - is this config known good.
|
||||
promote_current_config_to_backup()
|
||||
return True, "HAProxy reloaded successfully"
|
||||
else:
|
||||
# Reload failed, restore backup
|
||||
restore_backup()
|
||||
# Try to reload with backup config
|
||||
subprocess.run('echo "reload" | socat stdio /tmp/haproxy-cli',
|
||||
shell=True, capture_output=True)
|
||||
restored, restore_msg = restore_backup()
|
||||
if restored:
|
||||
# Try to reload with the restored (known-good) config
|
||||
subprocess.run(
|
||||
'echo "reload" | socat stdio /tmp/haproxy-cli',
|
||||
shell=True, capture_output=True)
|
||||
error_msg = f"HAProxy reload failed: {reload_result.stderr}"
|
||||
if not restored:
|
||||
error_msg += f" | ROLLBACK FAILED: {restore_msg}"
|
||||
logger.error(error_msg)
|
||||
return False, error_msg
|
||||
except Exception as e:
|
||||
# Critical error during reload, restore backup
|
||||
restore_backup()
|
||||
restored, restore_msg = restore_backup()
|
||||
error_msg = f"Critical error during reload: {e}"
|
||||
if not restored:
|
||||
error_msg += f" | ROLLBACK FAILED: {restore_msg}"
|
||||
logger.error(error_msg)
|
||||
return False, error_msg
|
||||
else:
|
||||
@@ -2034,11 +2387,15 @@ def reload_haproxy_safely():
|
||||
check=True, capture_output=True, text=True
|
||||
)
|
||||
logger.info("HAProxy started successfully")
|
||||
# Now - and only now - is this config known good.
|
||||
promote_current_config_to_backup()
|
||||
return True, "HAProxy started successfully"
|
||||
except subprocess.CalledProcessError as e:
|
||||
# Start failed, restore backup
|
||||
restore_backup()
|
||||
restored, restore_msg = restore_backup()
|
||||
error_msg = f"Failed to start HAProxy: {e.stderr}"
|
||||
if not restored:
|
||||
error_msg += f" | ROLLBACK FAILED: {restore_msg}"
|
||||
logger.error(error_msg)
|
||||
return False, error_msg
|
||||
except Exception as e:
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Idempotent haproxy liveness check — driven by the in-container supervisor loop.
|
||||
|
||||
Why this exists
|
||||
---------------
|
||||
haproxy runs as a *background child of PID 1* (gunicorn) — it is started once at
|
||||
container init (scripts/init.py -> do_initial_setup -> start_haproxy) and then
|
||||
left running. Nothing supervises it after that. If the haproxy master process
|
||||
dies mid-life (SIGABRT -> exit 134, segfault, or an OOM of the haproxy master),
|
||||
the container stays "up" because gunicorn is still PID 1, so Docker's
|
||||
`--restart` policy never fires. haproxy then stays down until the *external*
|
||||
host watchdog (haproxy-watchdog.sh) notices port 80 is dead for ~3 minutes and
|
||||
does a full `docker restart` — which drops every in-flight connection.
|
||||
|
||||
This script closes that gap: called on a short interval by the supervisor loop
|
||||
in start-up.sh, it re-launches haproxy *in place* within one interval.
|
||||
|
||||
Safety
|
||||
------
|
||||
start_haproxy() is guarded by `is_process_running('haproxy')` (psutil-based, so
|
||||
it works in this container which has no `ps`), so calling this while haproxy is
|
||||
healthy is a cheap no-op. It only ever acts when haproxy is genuinely gone.
|
||||
"""
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, '/haproxy')
|
||||
import haproxy_manager # noqa: E402 (sys.path manipulation must come first)
|
||||
|
||||
|
||||
def main():
|
||||
if haproxy_manager.is_process_running('haproxy'):
|
||||
return 0
|
||||
|
||||
haproxy_manager.logger.warning(
|
||||
"[haproxy-supervisor] haproxy process not found — attempting in-place restart"
|
||||
)
|
||||
# start_haproxy() validates the config (and regenerates it if invalid)
|
||||
# before launching, and swallows its own errors, so it will not raise here.
|
||||
haproxy_manager.start_haproxy()
|
||||
|
||||
if haproxy_manager.is_process_running('haproxy'):
|
||||
haproxy_manager.logger.info("[haproxy-supervisor] haproxy restarted in place")
|
||||
return 0
|
||||
|
||||
haproxy_manager.logger.error(
|
||||
"[haproxy-supervisor] haproxy restart FAILED — still not running after start_haproxy()"
|
||||
)
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -6,22 +6,40 @@
|
||||
SOCKET="/tmp/haproxy-cli"
|
||||
MAP_FILE="/etc/haproxy/blocked_ips.map"
|
||||
|
||||
# HAProxy runs in master-worker mode here, and /tmp/haproxy-cli is the MASTER
|
||||
# socket. Data-plane commands (map/table manipulation) are NOT accepted on the
|
||||
# master socket — they must be routed to a worker with the "@<n>" prefix. "@1"
|
||||
# targets the current active worker. (A bare "add map ..." on the master socket
|
||||
# fails with "Unknown command: 'add'".)
|
||||
cli() { printf '@1 %s\n' "$*" | socat stdio "$SOCKET"; }
|
||||
|
||||
# Map lookup in haproxy.cfg is `map_ip(...,0) -m int gt 0`, so each entry MUST be
|
||||
# "<ip_or_cidr> 1" — a bare IP yields an empty value (0) and is NOT blocked once
|
||||
# the map file is re-read on reload. The runtime map and the file must agree.
|
||||
MAP_VALUE=1
|
||||
|
||||
# Ensure map file exists
|
||||
if [ ! -f "$MAP_FILE" ]; then
|
||||
touch "$MAP_FILE"
|
||||
echo "# Blocked IPs - Format: IP_ADDRESS" > "$MAP_FILE"
|
||||
echo "# Blocked IPs - Format: <ip_or_cidr> 1 (one per line)" > "$MAP_FILE"
|
||||
fi
|
||||
|
||||
# Escape regex metacharacters (notably dots) in an IP/CIDR for anchored matching.
|
||||
esc_re() { printf '%s' "$1" | sed 's/[.[\*^$/]/\\&/g'; }
|
||||
|
||||
case "$1" in
|
||||
block)
|
||||
if [ -z "$2" ]; then
|
||||
echo "Usage: $0 block IP_ADDRESS"
|
||||
exit 1
|
||||
fi
|
||||
# Add IP to map file
|
||||
grep -q "^$2" "$MAP_FILE" || echo "$2" >> "$MAP_FILE"
|
||||
# Add to runtime map
|
||||
echo "add map /etc/haproxy/blocked_ips.map $2 1" | socat stdio "$SOCKET"
|
||||
re="$(esc_re "$2")"
|
||||
# Persist (idempotent, anchored so 1.2.3.4 doesn't match 1.2.3.45),
|
||||
# always with the trailing value so the block survives a reload.
|
||||
if ! grep -qE "^${re}([[:space:]]|$)" "$MAP_FILE"; then
|
||||
echo "$2 $MAP_VALUE" >> "$MAP_FILE"
|
||||
fi
|
||||
# Apply at runtime immediately (no reload).
|
||||
cli "add map $MAP_FILE $2 $MAP_VALUE"
|
||||
echo "Blocked IP: $2"
|
||||
;;
|
||||
|
||||
@@ -30,31 +48,33 @@ case "$1" in
|
||||
echo "Usage: $0 unblock IP_ADDRESS"
|
||||
exit 1
|
||||
fi
|
||||
# Remove from map file
|
||||
sed -i "/^$2$/d" "$MAP_FILE"
|
||||
# Remove from runtime map
|
||||
echo "del map /etc/haproxy/blocked_ips.map $2" | socat stdio "$SOCKET"
|
||||
re="$(esc_re "$2")"
|
||||
# Remove from map file (match "<ip>" optionally followed by a value).
|
||||
sed -i -E "/^${re}([[:space:]]|$)/d" "$MAP_FILE"
|
||||
# Remove from runtime map.
|
||||
cli "del map $MAP_FILE $2"
|
||||
echo "Unblocked IP: $2"
|
||||
;;
|
||||
|
||||
list)
|
||||
echo "Currently blocked IPs:"
|
||||
echo "show map /etc/haproxy/blocked_ips.map" | socat stdio "$SOCKET" | awk '{print $1}'
|
||||
# `show map` output is "<ptr> <key> <value>" — the IP is field 2.
|
||||
cli "show map $MAP_FILE" | awk 'NF>=2 {print $2}'
|
||||
;;
|
||||
|
||||
clear)
|
||||
echo "Clearing all blocked IPs..."
|
||||
echo "clear map /etc/haproxy/blocked_ips.map" | socat stdio "$SOCKET"
|
||||
echo "# Blocked IPs - Format: IP_ADDRESS" > "$MAP_FILE"
|
||||
cli "clear map $MAP_FILE"
|
||||
echo "# Blocked IPs - Format: <ip_or_cidr> 1 (one per line)" > "$MAP_FILE"
|
||||
echo "All IPs unblocked"
|
||||
;;
|
||||
|
||||
stats)
|
||||
echo "=== HAProxy 3.0.11 Threat Intelligence Dashboard ==="
|
||||
echo "show table web" | socat stdio "$SOCKET" | awk 'NR<=21'
|
||||
cli "show table web" | awk 'NR<=21'
|
||||
echo ""
|
||||
echo "=== Top Threat Scores ==="
|
||||
echo "show table web" | socat stdio "$SOCKET" | awk '
|
||||
cli "show table web" | awk '
|
||||
NR>1 {
|
||||
ip = $1
|
||||
auth_fail = 0
|
||||
@@ -84,7 +104,7 @@ case "$1" in
|
||||
exit 1
|
||||
fi
|
||||
# Add to manual blacklist using GPC(13)
|
||||
echo "set table web key $2 data.gpc(13) 1" | socat stdio "$SOCKET"
|
||||
cli "set table web key $2 data.gpc(13) 1"
|
||||
echo "Manually blacklisted IP: $2 (GPC(13) = 1)"
|
||||
;;
|
||||
|
||||
@@ -94,7 +114,7 @@ case "$1" in
|
||||
exit 1
|
||||
fi
|
||||
# Clear manual blacklist flag
|
||||
echo "set table web key $2 data.gpc(13) 0" | socat stdio "$SOCKET"
|
||||
cli "set table web key $2 data.gpc(13) 0"
|
||||
echo "Removed manual blacklist for IP: $2"
|
||||
;;
|
||||
|
||||
@@ -104,7 +124,7 @@ case "$1" in
|
||||
exit 1
|
||||
fi
|
||||
# Add to auto-blacklist using GPC(14)
|
||||
echo "set table web key $2 data.gpc(14) 1" | socat stdio "$SOCKET"
|
||||
cli "set table web key $2 data.gpc(14) 1"
|
||||
echo "Auto-blacklisted IP: $2 (GPC(14) = 1)"
|
||||
;;
|
||||
|
||||
@@ -115,14 +135,14 @@ case "$1" in
|
||||
fi
|
||||
# Show detailed threat breakdown for specific IP
|
||||
echo "Threat analysis for $2:"
|
||||
echo "show table web key $2" | socat stdio "$SOCKET"
|
||||
cli "show table web key $2"
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "Usage: $0 {block|unblock|list|clear|blacklist|unblacklist|auto-blacklist|threat-score|stats} [IP_ADDRESS]"
|
||||
echo ""
|
||||
echo "HAProxy 3.0.11 Enhanced Security Commands:"
|
||||
echo " block IP - Block IP via map file (immediate)"
|
||||
echo " block IP - Block IP via map file (immediate + persisted)"
|
||||
echo " unblock IP - Unblock IP from map file"
|
||||
echo " blacklist IP - Manual blacklist via GPC(13) array"
|
||||
echo " unblacklist IP - Remove manual blacklist flag"
|
||||
@@ -141,4 +161,4 @@ case "$1" in
|
||||
echo " gpc(14): Auto-blacklist candidate × 50"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
esac
|
||||
|
||||
+23
-1
@@ -27,11 +27,33 @@ cron &
|
||||
# Phase 1: container init
|
||||
python /haproxy/scripts/init.py
|
||||
|
||||
# Phase 1.5: in-container haproxy supervisor.
|
||||
# haproxy runs as a background child of PID 1 (gunicorn) with NOTHING watching
|
||||
# it after init. If the haproxy master dies mid-life (e.g. SIGABRT -> exit 134,
|
||||
# segfault), the container stays "up" (gunicorn is PID 1), Docker's --restart
|
||||
# policy never fires, and haproxy is down until the external host watchdog
|
||||
# full-restarts the whole container minutes later (dropping every connection).
|
||||
# This loop revives haproxy in place within one interval. ensure_haproxy.py is
|
||||
# idempotent — a cheap no-op whenever haproxy is already running.
|
||||
HAPROXY_SUPERVISOR_INTERVAL="${HAPROXY_SUPERVISOR_INTERVAL:-15}"
|
||||
(
|
||||
while true; do
|
||||
sleep "${HAPROXY_SUPERVISOR_INTERVAL}"
|
||||
python /haproxy/scripts/ensure_haproxy.py 2>&1 || true
|
||||
done
|
||||
) &
|
||||
|
||||
# Phase 2: WSGI servers
|
||||
# Tunable via env: HAPROXY_MGR_API_WORKERS (default 1), HAPROXY_MGR_API_TIMEOUT
|
||||
# (default 120 — API can do slow ACME calls), HAPROXY_MGR_MAX_REQUESTS (default
|
||||
# 1000 — worker recycle frequency).
|
||||
API_WORKERS="${HAPROXY_MGR_API_WORKERS:-1}"
|
||||
#
|
||||
# API_WORKERS default is 2 (was 1). A single worker is a single point of
|
||||
# failure: if its gthread pool ever wedges (see the 2026-07-07 subprocess-hang
|
||||
# incident — now bounded by DEFAULT_SUBPROCESS_TIMEOUT in haproxy_manager.py),
|
||||
# the entire management API goes dark. A second worker keeps the API answering
|
||||
# (config regenerate, health, SSL) while the other recycles via --max-requests.
|
||||
API_WORKERS="${HAPROXY_MGR_API_WORKERS:-2}"
|
||||
API_TIMEOUT="${HAPROXY_MGR_API_TIMEOUT:-120}"
|
||||
MAX_REQ="${HAPROXY_MGR_MAX_REQUESTS:-1000}"
|
||||
MAX_REQ_JITTER="${HAPROXY_MGR_MAX_REQUESTS_JITTER:-100}"
|
||||
|
||||
Executable
+467
@@ -0,0 +1,467 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Regression tests for HAProxy config backup / rollback ordering.
|
||||
|
||||
Why this file exists
|
||||
--------------------
|
||||
generate_config() used to write the new haproxy.cfg and only THEN call
|
||||
reload_haproxy_safely() -> create_backup(), so the "backup" was a copy of the
|
||||
config that had just been written. On a validation failure restore_backup()
|
||||
restored the identical broken bytes: the advertised rollback was a no-op and a
|
||||
fatal haproxy.cfg stayed on disk, where start_haproxy() refuses to launch.
|
||||
|
||||
These tests pin the ordering invariant (backup predates the write) and the
|
||||
observable end-to-end behaviour (after a failed validation the file on disk is
|
||||
the previous working config and HAProxy will start with it).
|
||||
|
||||
Running
|
||||
-------
|
||||
python3 scripts/test-config-rollback.py # tests the repo checkout
|
||||
HAPROXY_MANAGER_DIR=/some/other/tree \
|
||||
python3 scripts/test-config-rollback.py # tests another tree
|
||||
|
||||
The repo has no Python test framework (scripts/test-*.sh are curl-based
|
||||
integration scripts against a running API), so this is a self-contained
|
||||
stdlib-unittest script - no pytest, no venv, no new dependencies beyond the
|
||||
application's own requirements.txt (Flask/Jinja2/psutil), which are already
|
||||
present in the container image.
|
||||
|
||||
No HAProxy binary is required: a stub `haproxy` is put on PATH that mimics
|
||||
`haproxy -c -f <file>` by rejecting any config containing the token
|
||||
__BROKEN__, which is how the tests inject an invalid configuration.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import shutil
|
||||
import sqlite3
|
||||
import logging
|
||||
import tempfile
|
||||
import textwrap
|
||||
import unittest
|
||||
|
||||
BROKEN_TOKEN = '__BROKEN__'
|
||||
|
||||
MODULE_DIR = os.path.abspath(
|
||||
os.environ.get('HAPROXY_MANAGER_DIR',
|
||||
os.path.join(os.path.dirname(os.path.abspath(__file__)), '..'))
|
||||
)
|
||||
|
||||
# haproxy_manager builds its Jinja2 environment from the relative path
|
||||
# Path('templates'), so it has to be imported with the module dir as cwd.
|
||||
os.chdir(MODULE_DIR)
|
||||
sys.path.insert(0, MODULE_DIR)
|
||||
|
||||
# The module opens /var/log/haproxy-manager.log at import time via
|
||||
# logging.FileHandler. Redirect that one call so the suite runs unprivileged.
|
||||
_LOG_DIR = tempfile.mkdtemp(prefix='haproxy-mgr-test-logs-')
|
||||
_real_file_handler = logging.FileHandler
|
||||
logging.FileHandler = (
|
||||
lambda fn, *a, **kw: _real_file_handler(
|
||||
os.path.join(_LOG_DIR, os.path.basename(fn)), *a, **kw)
|
||||
)
|
||||
try:
|
||||
import haproxy_manager as hm
|
||||
except ImportError as exc: # pragma: no cover - environment problem, not a failure
|
||||
sys.stderr.write(
|
||||
f"SKIP: cannot import haproxy_manager ({exc}).\n"
|
||||
"Install the application requirements first: pip install -r requirements.txt\n"
|
||||
)
|
||||
raise SystemExit(77)
|
||||
finally:
|
||||
logging.FileHandler = _real_file_handler
|
||||
|
||||
logging.getLogger('haproxy_manager').setLevel(logging.CRITICAL)
|
||||
|
||||
FAKE_HAPROXY = textwrap.dedent(f"""\
|
||||
#!/bin/sh
|
||||
# Test stub for the haproxy binary.
|
||||
# haproxy -c -f FILE -> exit 1 if FILE contains {BROKEN_TOKEN}, else 0
|
||||
# haproxy -W -S ... -f FILE (start) -> same validation, then exit 0
|
||||
cfg=""
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in -f) cfg="$2"; shift ;; esac
|
||||
shift
|
||||
done
|
||||
if [ -n "$cfg" ] && grep -q '{BROKEN_TOKEN}' "$cfg" 2>/dev/null; then
|
||||
echo "[ALERT] parsing [$cfg:1] : unknown keyword '{BROKEN_TOKEN}'" >&2
|
||||
exit 1
|
||||
fi
|
||||
exit 0
|
||||
""")
|
||||
|
||||
|
||||
class RollbackTestCase(unittest.TestCase):
|
||||
"""Base fixture: an isolated fake /etc/haproxy plus a stub haproxy binary."""
|
||||
|
||||
def setUp(self):
|
||||
self.tmp = tempfile.mkdtemp(prefix='haproxy-rollback-test-')
|
||||
self.addCleanup(shutil.rmtree, self.tmp, True)
|
||||
|
||||
bindir = os.path.join(self.tmp, 'bin')
|
||||
os.makedirs(bindir)
|
||||
stub = os.path.join(bindir, 'haproxy')
|
||||
with open(stub, 'w') as fh:
|
||||
fh.write(FAKE_HAPROXY)
|
||||
os.chmod(stub, 0o755)
|
||||
self._old_path = os.environ['PATH']
|
||||
os.environ['PATH'] = bindir + os.pathsep + self._old_path
|
||||
self.addCleanup(lambda: os.environ.__setitem__('PATH', self._old_path))
|
||||
|
||||
self.etc = os.path.join(self.tmp, 'etc')
|
||||
os.makedirs(self.etc)
|
||||
|
||||
overrides = {
|
||||
'DB_FILE': os.path.join(self.etc, 'haproxy_config.db'),
|
||||
'HAPROXY_CONFIG_PATH': os.path.join(self.etc, 'haproxy.cfg'),
|
||||
'HAPROXY_BACKUP_PATH': os.path.join(self.etc, 'haproxy.cfg.backup'),
|
||||
'BLOCKED_IPS_MAP_PATH': os.path.join(self.etc, 'blocked_ips.map'),
|
||||
'BLOCKED_IPS_MAP_BACKUP_PATH': os.path.join(self.etc, 'blocked_ips.map.backup'),
|
||||
'CLUSTER_SECRET_PATH': os.path.join(self.etc, 'cluster-secret'),
|
||||
'SSL_CERTS_DIR': os.path.join(self.etc, 'certs'),
|
||||
'HAPROXY_SOCKET_PATH': os.path.join(self.etc, 'haproxy.sock'),
|
||||
# Added by the rollback fix; older trees do not have it.
|
||||
'CORAZA_SPOE_CONFIG_PATH': os.path.join(self.etc, 'coraza-spoe.cfg'),
|
||||
'CORAZA_SPOE_BACKUP_PATH': os.path.join(self.etc, 'coraza-spoe.cfg.backup'),
|
||||
}
|
||||
self._saved = {}
|
||||
for name, value in overrides.items():
|
||||
self._saved[name] = getattr(hm, name, None)
|
||||
setattr(hm, name, value)
|
||||
self.addCleanup(self._restore_globals)
|
||||
os.makedirs(hm.SSL_CERTS_DIR)
|
||||
|
||||
# log_operation() appends to a hardcoded /var/log path. Injecting `open`
|
||||
# into the module namespace shadows the builtin for that module only
|
||||
# (module globals are searched before builtins), so the real
|
||||
# log_operation code still runs.
|
||||
real_open = open
|
||||
log_dir = self.tmp
|
||||
|
||||
def _redirecting_open(path, *args, **kwargs):
|
||||
if isinstance(path, str) and path.startswith('/var/log/'):
|
||||
path = os.path.join(log_dir, os.path.basename(path))
|
||||
return real_open(path, *args, **kwargs)
|
||||
|
||||
hm.open = _redirecting_open
|
||||
self.addCleanup(lambda: hm.__dict__.pop('open', None))
|
||||
|
||||
hm.init_db()
|
||||
|
||||
def _restore_globals(self):
|
||||
for name, value in self._saved.items():
|
||||
if value is None:
|
||||
hm.__dict__.pop(name, None)
|
||||
else:
|
||||
setattr(hm, name, value)
|
||||
|
||||
# -- helpers ---------------------------------------------------------
|
||||
def add_domain(self, domain, backend_name, address='10.0.0.1'):
|
||||
with sqlite3.connect(hm.DB_FILE) as conn:
|
||||
cur = conn.cursor()
|
||||
cur.execute('INSERT INTO domains (domain, ssl_enabled) VALUES (?, 0)',
|
||||
(domain,))
|
||||
domain_id = cur.lastrowid
|
||||
cur.execute('INSERT INTO backends (name, domain_id) VALUES (?, ?)',
|
||||
(backend_name, domain_id))
|
||||
backend_id = cur.lastrowid
|
||||
cur.execute(
|
||||
'INSERT INTO backend_servers '
|
||||
'(backend_id, server_name, server_address, server_port) '
|
||||
'VALUES (?, ?, ?, ?)',
|
||||
(backend_id, 'srv1', address, 8080))
|
||||
conn.commit()
|
||||
|
||||
def block_ip(self, ip):
|
||||
with sqlite3.connect(hm.DB_FILE) as conn:
|
||||
conn.execute('INSERT INTO blocked_ips (ip_address, reason) VALUES (?, ?)',
|
||||
(ip, 'test'))
|
||||
conn.commit()
|
||||
|
||||
def read(self, path):
|
||||
with open(path) as fh:
|
||||
return fh.read()
|
||||
|
||||
def config_is_loadable(self):
|
||||
"""True if HAProxy would accept the config currently on disk."""
|
||||
import subprocess
|
||||
return subprocess.run(
|
||||
['haproxy', '-c', '-f', hm.HAPROXY_CONFIG_PATH],
|
||||
capture_output=True).returncode == 0
|
||||
|
||||
def generate_good_config(self):
|
||||
self.add_domain('good.example.com', 'good_backend')
|
||||
hm.generate_config()
|
||||
self.assertTrue(self.config_is_loadable(),
|
||||
'fixture precondition: first generated config must be valid')
|
||||
return self.read(hm.HAPROXY_CONFIG_PATH)
|
||||
|
||||
def break_the_config(self):
|
||||
"""Queue a domain whose rendered backend the validator rejects."""
|
||||
self.add_domain('bad.example.com', BROKEN_TOKEN + '_backend', '10.0.0.2')
|
||||
|
||||
|
||||
class TestBackupOrdering(RollbackTestCase):
|
||||
|
||||
def test_backup_is_taken_before_the_new_config_is_written(self):
|
||||
"""The ordering invariant, asserted directly.
|
||||
|
||||
Whatever create_backup() sees on disk must be the OLD config; if the
|
||||
write happens first the backup is a copy of the new config and rollback
|
||||
is meaningless.
|
||||
"""
|
||||
good = self.generate_good_config()
|
||||
|
||||
seen = {}
|
||||
real_create_backup = hm.create_backup
|
||||
|
||||
def spy(*args, **kwargs):
|
||||
seen['config_on_disk'] = self.read(hm.HAPROXY_CONFIG_PATH)
|
||||
return real_create_backup(*args, **kwargs)
|
||||
|
||||
hm.create_backup = spy
|
||||
self.addCleanup(setattr, hm, 'create_backup', real_create_backup)
|
||||
|
||||
self.add_domain('second.example.com', 'second_backend', '10.0.0.3')
|
||||
hm.generate_config()
|
||||
|
||||
self.assertIn('config_on_disk', seen,
|
||||
'create_backup() was never called during generate_config()')
|
||||
self.assertEqual(
|
||||
seen['config_on_disk'], good,
|
||||
'create_backup() ran AFTER the new config was written - the backup '
|
||||
'is a copy of the new config, so rollback cannot undo anything')
|
||||
|
||||
def test_backup_tracks_the_last_known_good_config(self):
|
||||
"""After a change that validated AND loaded, the backup is that config.
|
||||
|
||||
The rollback target is "the last configuration HAProxy actually ran",
|
||||
not "the file that happened to be there last time".
|
||||
"""
|
||||
good = self.generate_good_config()
|
||||
self.add_domain('second.example.com', 'second_backend', '10.0.0.3')
|
||||
hm.generate_config()
|
||||
|
||||
live = self.read(hm.HAPROXY_CONFIG_PATH)
|
||||
self.assertNotEqual(live, good, 'fixture sanity: the new config should differ')
|
||||
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), live,
|
||||
'the successful config was not recorded as known-good')
|
||||
|
||||
def test_backup_is_not_promoted_when_the_change_fails(self):
|
||||
"""A config that never loaded must not become the rollback target."""
|
||||
good = self.generate_good_config()
|
||||
self.break_the_config()
|
||||
with self.assertRaises(Exception):
|
||||
hm.generate_config()
|
||||
|
||||
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good,
|
||||
'a config that failed validation was promoted to backup')
|
||||
|
||||
|
||||
class TestRollbackEndToEnd(RollbackTestCase):
|
||||
|
||||
def test_failed_validation_leaves_the_last_good_config_on_disk(self):
|
||||
good = self.generate_good_config()
|
||||
|
||||
self.break_the_config()
|
||||
with self.assertRaises(Exception):
|
||||
hm.generate_config()
|
||||
|
||||
on_disk = self.read(hm.HAPROXY_CONFIG_PATH)
|
||||
self.assertNotIn(BROKEN_TOKEN, on_disk,
|
||||
'the rejected config is still on disk - rollback was a no-op')
|
||||
self.assertEqual(on_disk, good,
|
||||
'on-disk config is not byte-identical to the last good one')
|
||||
|
||||
def test_haproxy_would_still_start_after_a_failed_change(self):
|
||||
"""The operational consequence: the edge can still come up."""
|
||||
self.generate_good_config()
|
||||
self.break_the_config()
|
||||
with self.assertRaises(Exception):
|
||||
hm.generate_config()
|
||||
|
||||
self.assertTrue(self.config_is_loadable(),
|
||||
'HAProxy would refuse to start with the config left on disk')
|
||||
with self.assertLogs('haproxy_manager', level='INFO') as captured:
|
||||
hm.start_haproxy()
|
||||
self.assertTrue(
|
||||
any('HAProxy started successfully' in line for line in captured.output),
|
||||
f'start_haproxy() did not succeed after rollback: {captured.output}')
|
||||
|
||||
def test_blocked_ips_map_is_rolled_back_too(self):
|
||||
"""generate_config() rewrites the map file before writing haproxy.cfg."""
|
||||
self.block_ip('192.0.2.10')
|
||||
self.generate_good_config()
|
||||
good_map = self.read(hm.BLOCKED_IPS_MAP_PATH)
|
||||
|
||||
self.block_ip('198.51.100.20')
|
||||
self.break_the_config()
|
||||
with self.assertRaises(Exception):
|
||||
hm.generate_config()
|
||||
|
||||
self.assertEqual(self.read(hm.BLOCKED_IPS_MAP_PATH), good_map,
|
||||
'blocked IPs map was not rolled back with the config')
|
||||
|
||||
def test_first_run_failure_reports_that_rollback_was_impossible(self):
|
||||
"""No prior config: there is nothing to restore, and that must be said.
|
||||
|
||||
A missing backup must never be reported as a successful restore, and it
|
||||
must never be turned into "restore an empty file".
|
||||
"""
|
||||
self.break_the_config()
|
||||
with self.assertRaises(Exception) as ctx:
|
||||
hm.generate_config()
|
||||
|
||||
self.assertIn('ROLLBACK FAILED', str(ctx.exception),
|
||||
'a failed change with no backup was not reported as such')
|
||||
self.assertFalse(os.path.exists(hm.HAPROXY_BACKUP_PATH),
|
||||
'a backup was fabricated from the broken config')
|
||||
# The broken config is deliberately left in place: start_haproxy() can
|
||||
# then detect it and try to regenerate. It must not be blanked.
|
||||
self.assertGreater(os.path.getsize(hm.HAPROXY_CONFIG_PATH), 0,
|
||||
'config file was emptied instead of left for diagnosis')
|
||||
|
||||
|
||||
class TestBackupPrimitives(RollbackTestCase):
|
||||
|
||||
def test_restore_backup_distinguishes_missing_backup_from_success(self):
|
||||
restored, message = hm.restore_backup()
|
||||
self.assertFalse(restored,
|
||||
'restore_backup() reported success with no backup present')
|
||||
self.assertIn('cannot roll back', message.lower())
|
||||
|
||||
good = self.generate_good_config()
|
||||
with open(hm.HAPROXY_CONFIG_PATH, 'w') as fh:
|
||||
fh.write('scribbled over\n')
|
||||
|
||||
restored, message = hm.restore_backup()
|
||||
self.assertTrue(restored, message)
|
||||
self.assertEqual(self.read(hm.HAPROXY_CONFIG_PATH), good)
|
||||
|
||||
def test_a_successful_generation_records_a_rollback_target(self):
|
||||
"""Even the first-ever generation must leave something to roll back to."""
|
||||
good = self.generate_good_config()
|
||||
self.assertTrue(
|
||||
os.path.exists(hm.HAPROXY_BACKUP_PATH),
|
||||
'after a successful reload there is still no known-good backup')
|
||||
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good)
|
||||
|
||||
def test_a_broken_current_config_does_not_replace_a_good_backup(self):
|
||||
"""The known-good marker.
|
||||
|
||||
If the config already on disk is broken (previous failed write, manual
|
||||
edit), snapshotting it would make "rollback" mean "restore a different
|
||||
broken config". The older validated backup must survive.
|
||||
"""
|
||||
good = self.generate_good_config()
|
||||
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good,
|
||||
'fixture: a good backup should exist by now')
|
||||
|
||||
with open(hm.HAPROXY_CONFIG_PATH, 'w') as fh:
|
||||
fh.write(f'garbage {BROKEN_TOKEN} config\n')
|
||||
|
||||
ok, status = hm.create_backup()
|
||||
self.assertTrue(ok)
|
||||
self.assertEqual(status, 'kept_previous')
|
||||
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good,
|
||||
'a broken config overwrote the known-good backup')
|
||||
|
||||
def test_reload_does_not_take_its_own_backup(self):
|
||||
"""reload_haproxy_safely() runs after the write, so it must not back up."""
|
||||
good = self.generate_good_config()
|
||||
with open(hm.HAPROXY_CONFIG_PATH, 'w') as fh:
|
||||
fh.write(f'broken {BROKEN_TOKEN}\n')
|
||||
|
||||
success, message = hm.reload_haproxy_safely(backup_status='created')
|
||||
|
||||
self.assertFalse(success)
|
||||
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good,
|
||||
'reload_haproxy_safely() overwrote the good backup')
|
||||
self.assertEqual(self.read(hm.HAPROXY_CONFIG_PATH), good,
|
||||
'reload_haproxy_safely() did not roll the config back')
|
||||
|
||||
def test_unchanged_config_is_not_revalidated(self):
|
||||
"""Fast path: if the backup already is the live config, do no work.
|
||||
|
||||
generate_config() runs inside customer-facing API calls and
|
||||
`haproxy -c` is expensive on an edge with hundreds of certificates.
|
||||
"""
|
||||
self.generate_good_config()
|
||||
|
||||
calls = []
|
||||
real_validate = hm.validate_config_file
|
||||
hm.validate_config_file = lambda path: (calls.append(path),
|
||||
real_validate(path))[1]
|
||||
self.addCleanup(setattr, hm, 'validate_config_file', real_validate)
|
||||
|
||||
ok, status = hm.create_backup()
|
||||
self.assertTrue(ok)
|
||||
self.assertEqual(status, 'created')
|
||||
self.assertEqual(calls, [],
|
||||
'the unchanged live config was re-validated needlessly')
|
||||
|
||||
def test_fast_path_does_not_hide_a_drifted_broken_config(self):
|
||||
"""If the live config drifted from the backup, the gate must still run."""
|
||||
good = self.generate_good_config()
|
||||
with open(hm.HAPROXY_CONFIG_PATH, 'w') as fh:
|
||||
fh.write(f'hand edited {BROKEN_TOKEN}\n')
|
||||
|
||||
ok, status = hm.create_backup()
|
||||
self.assertTrue(ok)
|
||||
self.assertEqual(status, 'kept_previous',
|
||||
'a drifted broken config was silently accepted')
|
||||
self.assertEqual(self.read(hm.HAPROXY_BACKUP_PATH), good)
|
||||
|
||||
def test_backup_set_covers_every_file_generate_config_writes(self):
|
||||
pairs = dict(hm._config_backup_pairs())
|
||||
for path in (hm.HAPROXY_CONFIG_PATH, hm.BLOCKED_IPS_MAP_PATH,
|
||||
hm.CORAZA_SPOE_CONFIG_PATH):
|
||||
self.assertIn(path, pairs,
|
||||
f'{path} is written by generate_config() but is not '
|
||||
'part of the backed-up config set')
|
||||
|
||||
def test_coraza_spoe_config_round_trips(self):
|
||||
self.generate_good_config()
|
||||
with open(hm.CORAZA_SPOE_CONFIG_PATH, 'w') as fh:
|
||||
fh.write('spoe-good\n')
|
||||
hm.create_backup()
|
||||
with open(hm.CORAZA_SPOE_CONFIG_PATH, 'w') as fh:
|
||||
fh.write('spoe-broken\n')
|
||||
restored, message = hm.restore_backup()
|
||||
self.assertTrue(restored, message)
|
||||
self.assertEqual(self.read(hm.CORAZA_SPOE_CONFIG_PATH), 'spoe-good\n')
|
||||
|
||||
|
||||
class TestAtomicWrite(RollbackTestCase):
|
||||
|
||||
def test_write_is_atomic_and_preserves_mode(self):
|
||||
path = os.path.join(self.etc, 'atomic.cfg')
|
||||
with open(path, 'w') as fh:
|
||||
fh.write('old')
|
||||
os.chmod(path, 0o644)
|
||||
|
||||
hm.write_config_atomically(path, 'new content\n')
|
||||
|
||||
self.assertEqual(self.read(path), 'new content\n')
|
||||
self.assertEqual(oct(os.stat(path).st_mode & 0o777), oct(0o644))
|
||||
leftovers = [n for n in os.listdir(self.etc) if n.endswith('.tmp')]
|
||||
self.assertEqual(leftovers, [], f'temp files left behind: {leftovers}')
|
||||
|
||||
def test_failed_write_leaves_the_previous_file_intact(self):
|
||||
path = os.path.join(self.etc, 'atomic.cfg')
|
||||
with open(path, 'w') as fh:
|
||||
fh.write('old content\n')
|
||||
|
||||
# Anything that makes f.write() blow up mid-flight stands in for a full
|
||||
# disk / killed container.
|
||||
with self.assertRaises(Exception):
|
||||
hm.write_config_atomically(path, object())
|
||||
|
||||
self.assertEqual(self.read(path), 'old content\n',
|
||||
'a failed write clobbered the previous config')
|
||||
leftovers = [n for n in os.listdir(self.etc) if n.endswith('.tmp')]
|
||||
self.assertEqual(leftovers, [], f'temp files left behind: {leftovers}')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
print(f"testing haproxy_manager from: {MODULE_DIR}")
|
||||
unittest.main(verbosity=2)
|
||||
@@ -0,0 +1,41 @@
|
||||
# Long-lived backend for {{ name }} (template_override='hap_backend_longlived').
|
||||
# Use for apps whose PRIMARY traffic holds connections open: media streaming,
|
||||
# large up/downloads, or persistent viewer/streaming sessions. Both the primary
|
||||
# and the SSE backend are tuned long-lived here (no http-server-close,
|
||||
# http-no-delay, 6h server/tunnel/keep-alive timeouts).
|
||||
#
|
||||
# Compare hap_backend_websocket.tpl, which keeps the PRIMARY backend standard
|
||||
# and only makes the -sse-backend long-lived. Pick this one when the main path
|
||||
# itself needs long-lived connections, not just an SSE side-channel.
|
||||
backend {{ name }}-backend
|
||||
no option http-server-close
|
||||
option http-no-delay
|
||||
timeout server 6h
|
||||
timeout tunnel 6h
|
||||
timeout http-keep-alive 6h
|
||||
option forwardfor
|
||||
http-request add-header X-CLIENT-IP %[var(txn.real_ip)]
|
||||
http-request set-header X-Real-IP %[var(txn.real_ip)]
|
||||
http-request set-header X-Forwarded-For %[var(txn.real_ip)]
|
||||
http-request set-header X-Forwarded-Proto https if { ssl_fc }
|
||||
http-request set-header X-Forwarded-Proto http if !{ ssl_fc }
|
||||
{% for server in servers %}
|
||||
server {{ server.server_name }} {{ server.server_address }}:{{ server.server_port }} {{ server.server_options }} resolvers docker_dns init-addr last,libc,none
|
||||
{% endfor %}
|
||||
|
||||
# SSE variant (Accept: text/event-stream / ?action=stream auto-routes here)
|
||||
backend {{ name }}-sse-backend
|
||||
no option http-server-close
|
||||
option http-no-delay
|
||||
timeout server 6h
|
||||
timeout tunnel 6h
|
||||
timeout http-keep-alive 6h
|
||||
option forwardfor
|
||||
http-request add-header X-CLIENT-IP %[var(txn.real_ip)]
|
||||
http-request set-header X-Real-IP %[var(txn.real_ip)]
|
||||
http-request set-header X-Forwarded-For %[var(txn.real_ip)]
|
||||
http-request set-header X-Forwarded-Proto https if { ssl_fc }
|
||||
http-request set-header X-Forwarded-Proto http if !{ ssl_fc }
|
||||
{% for server in servers %}
|
||||
server {{ server.server_name }} {{ server.server_address }}:{{ server.server_port }} {{ server.server_options }} resolvers docker_dns init-addr last,libc,none
|
||||
{% endfor %}
|
||||
@@ -27,6 +27,23 @@ global
|
||||
# SSL and Performance
|
||||
tune.ssl.default-dh-param 2048
|
||||
|
||||
# HTTP/3 over QUIC. The Debian haproxy package is built against system
|
||||
# OpenSSL via the compatibility shim (USE_QUIC_OPENSSL_COMPAT), which is
|
||||
# not a native QUIC TLS stack. HAProxy therefore rejects `quic*@` binds
|
||||
# unless this opt-in is set. `limited-quic` enables QUIC through the compat
|
||||
# layer (no 0-RTT — that needs quictls/aws-lc or native OpenSSL 3.5 QUIC).
|
||||
# Without this, the quic bind in the frontend fails to start: "this SSL
|
||||
# library does not support the QUIC protocol".
|
||||
limited-quic
|
||||
{%- if cluster_secret %}
|
||||
|
||||
# Stable secret keying QUIC Retry/address-validation tokens. Self-healed
|
||||
# to /etc/haproxy/cluster-secret (named volume) by the manager so it
|
||||
# survives recreates; without it haproxy picks a random one per process
|
||||
# and tokens don't survive reloads (benign, just a startup notice).
|
||||
cluster-secret "{{ cluster_secret }}"
|
||||
{%- endif %}
|
||||
|
||||
# HTTP/2 protection against Rapid Reset (CVE-2023-44487) and stream abuse
|
||||
tune.h2.fe.max-total-streams 2000
|
||||
tune.h2.fe.glitches-threshold 50
|
||||
|
||||
@@ -4,6 +4,21 @@ frontend web
|
||||
# crt can now be a path, so it will load all .pem files in the path
|
||||
bind 0.0.0.0:443 ssl crt {{ crt_path }} alpn h2,http/1.1
|
||||
|
||||
# HTTP/3 over QUIC (UDP/443). Same cert path as the TCP listener above.
|
||||
# The Debian haproxy package is built +QUIC (QUIC_OPENSSL_COMPAT), so this
|
||||
# is config-only — no source build. Requires UDP/443 published on the
|
||||
# container (`-p 443:443/udp`) and open at the host firewall. `h3` is the
|
||||
# only ALPN QUIC negotiates; h2/http1 stay on the TCP bind above. Sharing
|
||||
# the frontend means all the real-IP, rate-limit, IP-block and Coraza
|
||||
# rules below apply identically to H3 traffic.
|
||||
bind quic4@0.0.0.0:443 ssl crt {{ crt_path }} alpn h3
|
||||
|
||||
# Advertise H3 so browsers upgrade their existing TCP (h2) connection to
|
||||
# QUIC on the next request. `ma` is how long (seconds) the client may
|
||||
# cache the advertisement. http-after-response applies it to every
|
||||
# response, including haproxy-generated ones (blocks, default page).
|
||||
http-after-response set-header alt-svc "h3=\":443\"; ma=86400"
|
||||
|
||||
# Capture Host header so it appears in httplog output (in %hr field)
|
||||
http-request capture req.hdr(Host) len 64
|
||||
|
||||
@@ -49,6 +64,67 @@ frontend web
|
||||
# High error rate: >100 errors in 30s (scanner/fuzzer behavior)
|
||||
http-request tarpit deny_status 403 if { sc_http_err_rate(0) gt 100 } !is_local !is_trusted_ip !is_whitelisted !is_health_check
|
||||
|
||||
# --- WordPress wp-login.php brute-force protection ---
|
||||
# The generic limits above are deliberately high (media-heavy sites), so a
|
||||
# slow credential-stuffing run (dozens of login POSTs/min) slips under them.
|
||||
# Track POSTs to wp-login.php per real client IP in a DEDICATED 60s table
|
||||
# (sc1 / backend wp_bruteforce, defined in hap_security_tables.tpl) and
|
||||
# tarpit once an IP exceeds 30/min. Only login POSTs are counted — GETs of
|
||||
# the login form, normal browsing, and the handful of POSTs a legit user
|
||||
# makes are unaffected; an offending IP can still browse, just not keep
|
||||
# hammering login. path_end also covers subdirectory WP installs. Honors the
|
||||
# same whitelist (RFC1918 / trusted_ips.list / trusted_ips.map).
|
||||
acl wp_login_path path_end /wp-login.php
|
||||
http-request track-sc1 var(txn.real_ip) table wp_bruteforce if METH_POST wp_login_path
|
||||
http-request tarpit deny_status 429 if METH_POST wp_login_path { sc_http_req_rate(1) gt 30 } !is_local !is_trusted_ip !is_whitelisted
|
||||
|
||||
# --- WordPress wp-login.php "must-load-the-form-first" cookie challenge ---
|
||||
# Defeats DISTRIBUTED credential-stuffing (hundreds of thousands of unique
|
||||
# IPs, each low-and-slow, so the per-IP rule above can't see them). Such
|
||||
# bots POST straight to /wp-login.php without ever GETting the form — on
|
||||
# these sites the login POST:GET ratio is ~15:1. We hand out a cookie when
|
||||
# the form is actually fetched (GET) and require it on POST; direct-POST
|
||||
# bots lack it and are denied AT THE EDGE before reaching PHP. Real logins
|
||||
# are unaffected — WordPress login already requires loading the page and
|
||||
# accepting cookies. Immediate deny (NOT tarpit) — under a 300k-POST flood,
|
||||
# holding tarpit connections would exhaust HAProxy. Honors the whitelist.
|
||||
# Mark login-form GETs at REQUEST time (method/path are reliably evaluable
|
||||
# here; in the response phase they are not) so the cookie is emitted on the
|
||||
# form's own response.
|
||||
http-request set-var(txn.wp_login_form) int(1) if METH_GET wp_login_path
|
||||
http-after-response add-header set-cookie "whplc=1; Path=/; Max-Age=1800; HttpOnly; Secure; SameSite=Lax" if { var(txn.wp_login_form) -m found }
|
||||
acl has_login_cookie req.cook(whplc) -m found
|
||||
http-request deny deny_status 403 if METH_POST wp_login_path !has_login_cookie !is_local !is_trusted_ip !is_whitelisted
|
||||
|
||||
# WordPress REST batch endpoint lockdown ("wp2shell": CVE-2026-63030 +
|
||||
# CVE-2026-60137). Chaining a core SQL injection with REST batch-route
|
||||
# confusion gives unauthenticated RCE on WP 6.9.0-6.9.4 and 7.0.0-7.0.1
|
||||
# (fixed in 6.9.5 / 7.0.2). Exploits are public and were used against this
|
||||
# fleet on 2026-07-19/20; one site was compromised via this path before
|
||||
# patching. This is a virtual patch: it does not repair the vulnerable
|
||||
# application logic, it only removes reachability, so it stays until every
|
||||
# site is confirmed on a fixed release.
|
||||
#
|
||||
# Both routing forms must be covered -- a rule matching only the pretty
|
||||
# permalink path leaves the ?rest_route= fallback wide open, and urlp()
|
||||
# does not URL-decode, hence the third ACL for the %2F spelling.
|
||||
#
|
||||
# Anonymous-only. batch/v1 is used legitimately by the block editor for
|
||||
# multi-entity saves, so a blanket deny would break wp-admin for real
|
||||
# users; requiring a wordpress_logged_in_* cookie costs them nothing.
|
||||
# req.cook() needs an exact name and WordPress suffixes a per-site hash,
|
||||
# so this substring-matches the raw Cookie header instead.
|
||||
#
|
||||
# Immediate deny, not tarpit -- holding connections open helps an attacker
|
||||
# who is already scripting this. Honors the same whitelist as above.
|
||||
acl wp_batch_path path_beg /wp-json/batch/v1
|
||||
acl wp_batch_route urlp(rest_route) -i -m beg /batch/v1
|
||||
acl wp_batch_route_enc query -i -m sub rest_route=%2Fbatch%2Fv1
|
||||
acl has_wp_logged_in req.hdr(Cookie) -i -m sub wordpress_logged_in_
|
||||
http-request deny deny_status 403 if wp_batch_path !has_wp_logged_in !is_local !is_trusted_ip !is_whitelisted
|
||||
http-request deny deny_status 403 if wp_batch_route !has_wp_logged_in !is_local !is_trusted_ip !is_whitelisted
|
||||
http-request deny deny_status 403 if wp_batch_route_enc !has_wp_logged_in !is_local !is_trusted_ip !is_whitelisted
|
||||
|
||||
# IP blocking using map file (manual blocks only)
|
||||
# Map file format: /etc/haproxy/blocked_ips.map contains "<ip_or_cidr> 1" per line
|
||||
# Runtime updates: echo "add map #0 IP_ADDRESS 1" | socat stdio /var/run/haproxy.sock
|
||||
|
||||
@@ -5,4 +5,12 @@ frontend stats
|
||||
stats uri /stats
|
||||
stats refresh 30s
|
||||
stats show-legends
|
||||
stats show-node
|
||||
stats show-node
|
||||
|
||||
# Dedicated stick-table for WordPress wp-login.php brute-force tracking.
|
||||
# Tracked via track-sc1 from the `web` frontend (hap_listener.tpl); counts only
|
||||
# login POSTs per real client IP over a 60s window. Separate from the generic
|
||||
# sc0 connection/rate table so the login-attempt threshold is independent of
|
||||
# the (much higher) flood thresholds.
|
||||
backend wp_bruteforce
|
||||
stick-table type ip size 100k expire 30m store http_req_rate(60s)
|
||||
Reference in New Issue
Block a user