Compare commits
31
Commits
main
...
489290ed33
@@ -0,0 +1,44 @@
|
||||
name: Build and push coraza-spoa
|
||||
run-name: ${{ gitea.actor }} pushed a change to coraza-spoa/
|
||||
|
||||
# Triggers only on changes to the coraza-spoa subdirectory or this workflow
|
||||
# file itself — keeps the main haproxy-manager-base build and the coraza-spoa
|
||||
# build independent. workflow_dispatch lets us trigger manually after bumping
|
||||
# the upstream coraza-spoa version pin.
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- 'coraza-spoa/**'
|
||||
- '.gitea/workflows/build-push-coraza.yaml'
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
Build-and-Push:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: https://github.com/docker/setup-buildx-action@v3
|
||||
|
||||
- name: Login to Gitea
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: repo.anhonesthost.net
|
||||
username: ${{ secrets.CI_USER }}
|
||||
password: ${{ secrets.CI_TOKEN }}
|
||||
|
||||
- name: Build Image
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: ./coraza-spoa
|
||||
platforms: linux/amd64
|
||||
push: true
|
||||
tags: |
|
||||
repo.anhonesthost.net/cloud-hosting-platform/coraza-spoa:latest
|
||||
@@ -0,0 +1,65 @@
|
||||
name: Mirror base images
|
||||
run-name: weekly base-image mirror
|
||||
|
||||
# Pulls each declared base image from upstream and re-pushes to the in-house
|
||||
# registry, so any of our images that FROM these don't depend on docker.io's
|
||||
# Cloudflare R2 blob storage being reachable. The 2026-05-12 Cloudflare
|
||||
# incident motivated this for python:3.12-slim and again for golang:1.25
|
||||
# when the coraza-spoa build hit the same blob-fetch failure.
|
||||
#
|
||||
# Adding a new mirror = add one entry to the matrix below. The destination
|
||||
# tag is always cloud-hosting-platform/<image>:<tag>, matching upstream.
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# Mondays 06:00 UTC — outside customer peak hours and well before the
|
||||
# typical Tuesday/Thursday push cycles. workflow_dispatch lets us trigger
|
||||
# manually from the Gitea UI when upstream publishes patches.
|
||||
- cron: '0 6 * * 1'
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
Mirror-Base:
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
# fail-fast=false so one image's upstream being down doesn't block the
|
||||
# others from refreshing.
|
||||
fail-fast: false
|
||||
matrix:
|
||||
image:
|
||||
- { src: 'docker.io/library/python:3.12-slim', dst_path: 'cloud-hosting-platform/python', tag: '3.12-slim' }
|
||||
- { src: 'docker.io/library/golang:1.25', dst_path: 'cloud-hosting-platform/golang', tag: '1.25' }
|
||||
|
||||
steps:
|
||||
- name: Login to in-house registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: repo.anhonesthost.net
|
||||
username: ${{ secrets.CI_USER }}
|
||||
password: ${{ secrets.CI_TOKEN }}
|
||||
|
||||
- name: Pull, retag, push ${{ matrix.image.src }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
SRC="${{ matrix.image.src }}"
|
||||
DST="repo.anhonesthost.net/${{ matrix.image.dst_path }}:${{ matrix.image.tag }}"
|
||||
|
||||
echo "::group::Pulling ${SRC}"
|
||||
docker pull "${SRC}"
|
||||
echo "::endgroup::"
|
||||
|
||||
# Capture the upstream digest so the workflow log shows what we
|
||||
# actually pushed. Helps diagnose "did the mirror really update"
|
||||
# questions later.
|
||||
SRC_DIGEST=$(docker image inspect "${SRC}" -f '{{index .RepoDigests 0}}')
|
||||
echo "upstream digest: ${SRC_DIGEST}"
|
||||
|
||||
docker tag "${SRC}" "${DST}"
|
||||
|
||||
echo "::group::Pushing ${DST}"
|
||||
docker push "${DST}"
|
||||
echo "::endgroup::"
|
||||
|
||||
# Sanity: the in-house tag should now resolve to the same content.
|
||||
DST_DIGEST=$(docker image inspect "${DST}" -f '{{index .RepoDigests 0}}')
|
||||
echo "mirror digest: ${DST_DIGEST}"
|
||||
@@ -39,10 +39,11 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
||||
- `backend_servers` - Individual servers within backend groups
|
||||
|
||||
3. **Template System** - Jinja2 templates for HAProxy configuration generation:
|
||||
- `hap_header.tpl` - Global HAProxy settings and defaults
|
||||
- `hap_header.tpl` - Global HAProxy settings, defaults, and HTTP/2 tuning
|
||||
- `hap_backend.tpl` - Backend server definitions
|
||||
- `hap_listener.tpl` - Frontend listener configurations
|
||||
- `hap_listener.tpl` - Frontend listener configurations with rate limiting
|
||||
- `hap_letsencrypt.tpl` - SSL certificate configurations
|
||||
- `hap_security_tables.tpl` - Stats frontend and security stick tables
|
||||
- Template override support for custom backend configurations
|
||||
|
||||
4. **Certificate Management** - Automated SSL certificate handling:
|
||||
@@ -73,10 +74,51 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
||||
- Certificate private keys combined with certificates in HAProxy-compatible format
|
||||
- Default backend page for unmatched domains instead of exposing HAProxy errors
|
||||
|
||||
### Rate Limiting & Connection Limits (hap_listener.tpl)
|
||||
|
||||
- **Stick table**: `type ip size 200k expire 10m` tracking `conn_cur`, `conn_rate(10s)`, `http_req_rate(10s)`, `http_err_rate(30s)`
|
||||
- Tracks real client IP via `var(txn.real_ip)` to work correctly behind Cloudflare/proxies
|
||||
- **Rate limit thresholds**:
|
||||
- Tarpit at 3000 req/10s (300 req/s)
|
||||
- Hard block (deny) at 5000 req/10s (500 req/s)
|
||||
- Connection rate limit: 500/10s
|
||||
- Concurrent connection limit: 500
|
||||
- Error rate limit: 100/30s
|
||||
- **Whitelist bypasses** (exempt from rate limits):
|
||||
- `is_local` — RFC1918 private address ranges
|
||||
- `is_trusted_ip` — source IPs listed in `trusted_ips.list`
|
||||
- `is_whitelisted` — real IPs (from proxy headers) matched in `trusted_ips.map`
|
||||
|
||||
### Trusted IP Whitelist Files
|
||||
|
||||
- `trusted_ips.list` — Source IP whitelist for rate limit bypass (one CIDR/IP per line)
|
||||
- `trusted_ips.map` — Real IP whitelist for proxy-header matching (format: `<IP> 1`)
|
||||
- Both files are baked into the Docker image via `COPY` in the Dockerfile
|
||||
- Currently contains phone system IP `172.116.197.166`
|
||||
|
||||
### Timeout Hardening (hap_header.tpl)
|
||||
|
||||
- `timeout http-request`: 300s -> 30s (slowloris protection)
|
||||
- `timeout connect`: 120s -> 10s
|
||||
- `timeout client`: 10m -> 5m
|
||||
- `timeout http-keep-alive`: 120s -> 30s
|
||||
|
||||
### HTTP/2 Protection (hap_header.tpl)
|
||||
|
||||
- `tune.h2.fe.max-total-streams 2000` — limits total streams per HTTP/2 connection
|
||||
- `tune.h2.fe.glitches-threshold 50` — CVE-2023-44487 Rapid Reset protection
|
||||
|
||||
### Stats Frontend (hap_security_tables.tpl)
|
||||
|
||||
- HAProxy stats page bound to `127.0.0.1:8404` (localhost only, accessible inside container)
|
||||
- Template: `templates/hap_security_tables.tpl`
|
||||
|
||||
### Deployment Context
|
||||
|
||||
- Designed to run as Docker container with persistent volumes for certificates and configurations
|
||||
- Exposes ports 80 (HTTP), 443 (HTTPS), and 8000 (management API/UI)
|
||||
- Stats page on port 8404 (localhost only inside container)
|
||||
- Management interface on port 8000 should be firewall-protected in production
|
||||
- Dockerfile HEALTHCHECK verifies both port 8000 (Flask API) and port 80 (HAProxy), with `start-period=60s` and `timeout=10s`
|
||||
- Supports deployment on servers with git directory at `/root/whp` and web file sync via rsync to `/docker/whp/web/`
|
||||
- HAProxy is version 3.0.11
|
||||
+13
-1
@@ -1,10 +1,22 @@
|
||||
FROM python:3.12-slim
|
||||
# Base image mirrored into the in-house registry to remove docker.io
|
||||
# (Cloudflare R2) as a single point of failure for CI builds. The 2026-05-12
|
||||
# Cloudflare incident took down docker.io blob pulls and broke this image's CI.
|
||||
# Refresh procedure (run on a workstation that can reach docker.io, e.g.
|
||||
# monthly or when Python patches drop):
|
||||
# docker pull docker.io/library/python:3.12-slim
|
||||
# docker tag docker.io/library/python:3.12-slim \
|
||||
# repo.anhonesthost.net/cloud-hosting-platform/python:3.12-slim
|
||||
# docker push repo.anhonesthost.net/cloud-hosting-platform/python:3.12-slim
|
||||
# Future improvement: a scheduled Gitea Action that does the above automatically.
|
||||
FROM repo.anhonesthost.net/cloud-hosting-platform/python:3.12-slim
|
||||
RUN apt update -y && apt dist-upgrade -y && apt install socat haproxy cron certbot curl jq net-tools -y && apt clean && rm -rf /var/lib/apt/lists/*
|
||||
WORKDIR /haproxy
|
||||
COPY ./templates /haproxy/templates
|
||||
COPY requirements.txt /haproxy/
|
||||
COPY haproxy_manager.py /haproxy/
|
||||
COPY scripts /haproxy/scripts
|
||||
COPY trusted_ips.list /etc/haproxy/trusted_ips.list
|
||||
COPY trusted_ips.map /etc/haproxy/trusted_ips.map
|
||||
RUN chmod +x /haproxy/scripts/*
|
||||
RUN pip install -r requirements.txt
|
||||
# Create log directories
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
# Coraza-SPOA sidecar for haproxy-manager.
|
||||
#
|
||||
# Layout: built from upstream source. main.go is at the repo root; CRS rules
|
||||
# are bundled into the binary at build time (referenced as @owasp_crs/), so
|
||||
# the CRS version is whatever ships with the pinned coraza-spoa tag.
|
||||
#
|
||||
# Pin: review the upstream CHANGELOG (https://github.com/corazawaf/coraza-spoa/releases)
|
||||
# before bumping. New tags can ship newer CRS, which can introduce new rules
|
||||
# whose IDs fall into the "enforce day-one" ranges in overrides.conf — verify
|
||||
# those are still high-confidence before promoting a new tag to prod.
|
||||
|
||||
ARG CORAZA_SPOA_VERSION=v0.7.1
|
||||
|
||||
# golang:1.25 from the in-house mirror. The 2026-05-12 Cloudflare incident
|
||||
# took out docker.io blob pulls TWICE in one day (first for python:3.12-slim,
|
||||
# then for this image's golang:1.25), so both are mirrored at
|
||||
# repo.anhonesthost.net via the .gitea/workflows/mirror-base-image.yaml
|
||||
# weekly job.
|
||||
FROM repo.anhonesthost.net/cloud-hosting-platform/golang:1.25 AS build
|
||||
ARG CORAZA_SPOA_VERSION
|
||||
WORKDIR /src
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends git \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
RUN git clone --depth 1 --branch "${CORAZA_SPOA_VERSION}" \
|
||||
https://github.com/corazawaf/coraza-spoa.git . \
|
||||
&& go mod download \
|
||||
&& CGO_ENABLED=0 go build -trimpath -ldflags='-s -w' -o /out/coraza-spoa .
|
||||
|
||||
# Catalog extractor: walks the bundled CRS at build time and emits
|
||||
# rules-catalog.json so WHP's UI can render rule metadata without parsing
|
||||
# .conf files at runtime. Uses the SAME coraza-coreruleset version pin as
|
||||
# the coraza-spoa binary above (drift between the two would mislabel rules).
|
||||
FROM repo.anhonesthost.net/cloud-hosting-platform/golang:1.25 AS catalog
|
||||
WORKDIR /src
|
||||
COPY catalog-extractor/ .
|
||||
RUN go build -trimpath -o /out/catalog-extractor . \
|
||||
&& /out/catalog-extractor > /out/rules-catalog.json
|
||||
|
||||
# Distroless runtime: no shell, no package manager, no /tmp by default —
|
||||
# smallest attack surface for an exposed service. Audit log directory is
|
||||
# bind-mounted; coraza-spoa writes to it via direct file I/O (no shell needed).
|
||||
FROM gcr.io/distroless/static-debian12:nonroot
|
||||
|
||||
LABEL org.opencontainers.image.title="coraza-spoa-whp" \
|
||||
org.opencontainers.image.description="Coraza WAF SPOA agent configured for WHP haproxy-manager integration" \
|
||||
org.opencontainers.image.source="https://repo.anhonesthost.net/cloud-hosting-platform/haproxy-manager-base"
|
||||
|
||||
COPY --from=build /out/coraza-spoa /coraza-spoa
|
||||
COPY config.yaml /etc/coraza-spoa/config.yaml
|
||||
COPY overrides.conf /etc/coraza/overrides.conf
|
||||
COPY local-overrides.conf /etc/coraza/local-overrides.conf
|
||||
COPY host-exceptions/ /etc/coraza/host-exceptions/
|
||||
COPY --from=catalog /out/rules-catalog.json /etc/coraza/rules-catalog.json
|
||||
|
||||
# Audit log directory — bind-mount /var/log/coraza:/var/log/coraza from host
|
||||
# so logs persist across container restarts and AI Monitor can tail them.
|
||||
# Distroless nonroot user has UID 65532; the host directory must be writable
|
||||
# by that UID (install script will chown it appropriately).
|
||||
VOLUME ["/var/log/coraza"]
|
||||
|
||||
# SPOE TCP port — bound on 0.0.0.0:9000 inside the container. The host-side
|
||||
# port mapping is controlled by `docker run -p` (typically not exposed beyond
|
||||
# the internal docker network, since haproxy-manager reaches it by container
|
||||
# name on client-net).
|
||||
EXPOSE 9000
|
||||
|
||||
ENTRYPOINT ["/coraza-spoa", "--config", "/etc/coraza-spoa/config.yaml"]
|
||||
@@ -0,0 +1,78 @@
|
||||
# coraza-spoa sidecar
|
||||
|
||||
A sidecar container that runs [Coraza-SPOA](https://github.com/corazawaf/coraza-spoa) as a WAF engine for `haproxy-manager`. HAProxy consults it per-request via the SPOE/SPOP protocol; Coraza evaluates the request against OWASP CRS rules and tells HAProxy whether to allow or block.
|
||||
|
||||
## Design constraints
|
||||
|
||||
- **`haproxy-manager` does NOT depend on this sidecar.** The base image works standalone (used in other projects and home networks) without WAF. SPOE config in the generated `haproxy.cfg` is opt-in via an env var on `haproxy-manager`.
|
||||
- **Fail-open when the sidecar is unhealthy.** `option set-on-error continue` in the HAProxy SPOE config means request flow continues uninspected if coraza-spoa is unreachable, rather than 503-ing customer traffic.
|
||||
- **Detect-only globally; enforce explicitly.** See `overrides.conf` for the day-one enforce list. Most CRS rules log without blocking until we've tuned per-customer false positives.
|
||||
|
||||
## Deployment shape
|
||||
|
||||
Two containers per host, both on the `client-net` docker network:
|
||||
|
||||
```
|
||||
haproxy-manager (existing) — ports 80, 443, 8000
|
||||
│ SPOE TCP/9000 → reach coraza-spoa by container DNS
|
||||
▼
|
||||
coraza-spoa (this image)
|
||||
port 9000 (SPOE) — NOT exposed on host; internal network only
|
||||
/var/log/coraza — bind-mounted to host for AI Monitor consumption
|
||||
```
|
||||
|
||||
Typical `docker run`:
|
||||
|
||||
```bash
|
||||
mkdir -p /var/log/coraza
|
||||
chown 65532:65532 /var/log/coraza # distroless nonroot UID
|
||||
|
||||
docker run -d \
|
||||
--name coraza-spoa \
|
||||
--network client-net \
|
||||
--restart unless-stopped \
|
||||
-v /var/log/coraza:/var/log/coraza \
|
||||
repo.anhonesthost.net/cloud-hosting-platform/coraza-spoa:latest
|
||||
```
|
||||
|
||||
Then on the `haproxy-manager` container, add the env var:
|
||||
|
||||
```
|
||||
-e HAPROXY_CORAZA_SPOE_BACKEND=coraza-spoa:9000
|
||||
```
|
||||
|
||||
The haproxy-manager template engine sees the env var and renders the SPOE config block pointing at this sidecar. Without the env var, no SPOE blocks render — the haproxy-manager image's behavior is unchanged.
|
||||
|
||||
## Files
|
||||
|
||||
| File | Purpose |
|
||||
|---|---|
|
||||
| `Dockerfile` | Multi-stage build (golang:1.25 → distroless), pinned to upstream coraza-spoa tag |
|
||||
| `config.yaml` | SPOA listener config + one named application `haproxy` |
|
||||
| `overrides.conf` | Day-one enforce list (`ctl:ruleEngine=On` for high-confidence rule IDs) |
|
||||
| `README.md` | This file |
|
||||
|
||||
## Audit log
|
||||
|
||||
`/var/log/coraza/audit.log` — JSON, one event per line, RelevantOnly (only requests that triggered ≥1 rule are logged). AI Monitor should be configured to tail this on each host.
|
||||
|
||||
Entries include rule IDs, matched patterns, request metadata, and action taken (`log` for detect-only, `deny` for enforced). Use the JSON `action` field to filter blocked vs. observed.
|
||||
|
||||
## Upgrading the pin
|
||||
|
||||
CRS rules are bundled into the coraza-spoa binary at build time, so the CRS version is whatever ships with the pinned coraza-spoa tag. To upgrade:
|
||||
|
||||
1. Check upstream releases: <https://github.com/corazawaf/coraza-spoa/releases>
|
||||
2. Skim the CHANGELOG for new/changed rules in the `overrides.conf` ID ranges.
|
||||
3. Bump `ARG CORAZA_SPOA_VERSION` in the Dockerfile.
|
||||
4. Push to `main` — the Gitea workflow at `.gitea/workflows/build-push-coraza.yaml` rebuilds + pushes `:latest`.
|
||||
5. On each host, run `container-manager.sh recreate coraza-spoa` to pull the new image.
|
||||
|
||||
## Tuning false positives
|
||||
|
||||
When a legitimate request triggers a blocked rule, the audit log shows the rule ID. Two ways to silence it:
|
||||
|
||||
1. **Per-rule exception** in `overrides.conf`: `SecRuleRemoveById <id>` (full disable) or `SecRuleRemoveTargetById <id> "<target>"` (targeted exception).
|
||||
2. **Drop from the enforce list**: remove the rule's ID range from the `ctl:ruleEngine=On` overrides; it falls back to detect-only.
|
||||
|
||||
After tuning, push the change — CI rebuilds, then `recreate coraza-spoa` on each host to apply.
|
||||
@@ -0,0 +1,7 @@
|
||||
module catalog-extractor
|
||||
|
||||
go 1.25
|
||||
|
||||
require github.com/corazawaf/coraza-coreruleset/v4 v4.25.0
|
||||
|
||||
require github.com/magefile/mage v1.17.0 // indirect
|
||||
@@ -0,0 +1,12 @@
|
||||
github.com/corazawaf/coraza-coreruleset/v4 v4.25.0 h1:tqFO1lfVpTiyWtlN618OXpZMfw+nnN0Q4///W5W+/HM=
|
||||
github.com/corazawaf/coraza-coreruleset/v4 v4.25.0/go.mod h1:nRuGXITxOPvsLF2VxaTB7pYok8QB8BitX3ZenXcUryY=
|
||||
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
|
||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/magefile/mage v1.17.0 h1:dS4tkq997Ism03akafC8509iqDjeE7TNTexI25Y7sXM=
|
||||
github.com/magefile/mage v1.17.0/go.mod h1:Yj51kqllmsgFpvvSzgrZPK9WtluG3kUhFaBUVLo4feA=
|
||||
github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM=
|
||||
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
|
||||
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
||||
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
@@ -0,0 +1,80 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"os"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
crs "github.com/corazawaf/coraza-coreruleset/v4"
|
||||
)
|
||||
|
||||
type Rule struct {
|
||||
ID int `json:"id"`
|
||||
Msg string `json:"msg"`
|
||||
Severity string `json:"severity"`
|
||||
Tags []string `json:"tags"`
|
||||
File string `json:"file"`
|
||||
}
|
||||
|
||||
var (
|
||||
idRe = regexp.MustCompile(`(?i)\bid:'?(\d+)'?`)
|
||||
msgRe = regexp.MustCompile(`(?i)\bmsg:'([^']+)'`)
|
||||
severityRe = regexp.MustCompile(`(?i)\bseverity:'?([A-Z]+)'?`)
|
||||
tagRe = regexp.MustCompile(`(?i)\btag:'([^']+)'`)
|
||||
)
|
||||
|
||||
func main() {
|
||||
out := []Rule{}
|
||||
err := fs.WalkDir(crs.FS, ".", func(path string, d fs.DirEntry, err error) error {
|
||||
if err != nil || d.IsDir() {
|
||||
return err
|
||||
}
|
||||
if !strings.HasSuffix(path, ".conf") {
|
||||
return nil
|
||||
}
|
||||
b, err := fs.ReadFile(crs.FS, path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// Coalesce backslash-continuation lines so id/msg/etc on the same
|
||||
// logical rule are visible to the per-line scanner.
|
||||
text := regexp.MustCompile(`\\\s*\n\s*`).ReplaceAllString(string(b), " ")
|
||||
for _, line := range strings.Split(text, "\n") {
|
||||
line = strings.TrimSpace(line)
|
||||
if !strings.HasPrefix(line, "SecRule") && !strings.HasPrefix(line, "SecAction") {
|
||||
continue
|
||||
}
|
||||
m := idRe.FindStringSubmatch(line)
|
||||
if m == nil {
|
||||
continue
|
||||
}
|
||||
id, _ := strconv.Atoi(m[1])
|
||||
r := Rule{ID: id, File: path}
|
||||
if mm := msgRe.FindStringSubmatch(line); mm != nil {
|
||||
r.Msg = mm[1]
|
||||
}
|
||||
if mm := severityRe.FindStringSubmatch(line); mm != nil {
|
||||
r.Severity = strings.ToLower(mm[1])
|
||||
}
|
||||
for _, mm := range tagRe.FindAllStringSubmatch(line, -1) {
|
||||
r.Tags = append(r.Tags, mm[1])
|
||||
}
|
||||
out = append(out, r)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
os.Exit(1)
|
||||
}
|
||||
enc := json.NewEncoder(os.Stdout)
|
||||
enc.SetIndent("", " ")
|
||||
if err := enc.Encode(out); err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
# Coraza-SPOA configuration for WHP haproxy-manager integration.
|
||||
#
|
||||
# One named application "haproxy" — the haproxy-manager spoe template
|
||||
# references this same name in its spoe-agent block, so the SPOA knows
|
||||
# which rules to apply when HAProxy dispatches a request.
|
||||
#
|
||||
# Mode: SecRuleEngine DetectionOnly globally; overrides.conf promotes
|
||||
# specific high-confidence rule ID ranges to enforcement individually.
|
||||
# This is the safest posture for v1 — every rule logs, but only the
|
||||
# unambiguous ones (scanner UAs, RCE, LFI, webshells, Log4Shell) block.
|
||||
|
||||
bind: 0.0.0.0:9000
|
||||
|
||||
# Process-level logging (separate from per-request audit logging below)
|
||||
log_level: info
|
||||
log_file: /dev/stdout
|
||||
log_format: json
|
||||
|
||||
# Fallback when the request doesn't match a named application — we only
|
||||
# have one, so it's also the default.
|
||||
default_application: haproxy
|
||||
|
||||
applications:
|
||||
- name: haproxy
|
||||
directives: |
|
||||
# CRS-bundled defaults: recommended Coraza settings + CRS setup +
|
||||
# the rule pack itself (~16 MB of rules embedded in the binary).
|
||||
Include @coraza.conf-recommended
|
||||
Include @crs-setup.conf.example
|
||||
Include @owasp_crs/*.conf
|
||||
|
||||
# WHP-specific overrides — day-one enforce list, plus tuning for
|
||||
# the customer mix (WordPress, WooCommerce, Divi). Read this file
|
||||
# to see exactly what blocks vs what's detect-only.
|
||||
Include /etc/coraza/overrides.conf
|
||||
|
||||
# Runtime-managed overrides written by WHP UI. Empty by default.
|
||||
Include /etc/coraza/local-overrides.conf
|
||||
|
||||
# Global mode: log all alerts, block only what overrides.conf
|
||||
# explicitly promotes via ctl:ruleEngine=On.
|
||||
SecRuleEngine DetectionOnly
|
||||
|
||||
# Audit log: JSON to a bind-mounted file so AI Monitor + log
|
||||
# rotation can pick it up. RelevantOnly means we don't log every
|
||||
# passing request, only ones that triggered at least one rule.
|
||||
SecAuditEngine RelevantOnly
|
||||
SecAuditLog /var/log/coraza/audit.log
|
||||
SecAuditLogFormat JSON
|
||||
SecAuditLogParts ABIJDEFHKZ
|
||||
|
||||
# HAProxy sends request-only events for v1. Response inspection adds
|
||||
# latency on every page render with marginal additional protection
|
||||
# for our customer mix; can be turned on later if we want it.
|
||||
response_check: false
|
||||
|
||||
# Transactions cache for 60s. SPOE protocol is fire-and-forget per
|
||||
# request, so this is just how long Coraza holds context for any
|
||||
# multi-stage processing.
|
||||
transaction_ttl_ms: 60000
|
||||
|
||||
log_level: info
|
||||
log_file: /var/log/coraza/spoa.log
|
||||
log_format: json
|
||||
@@ -0,0 +1,3 @@
|
||||
# AUTOGENERATED by WHP — do not hand-edit.
|
||||
# Source of truth: whp.security_db coraza_rule_overrides table.
|
||||
# Empty file = no runtime overrides; baked-in overrides.conf governs.
|
||||
@@ -0,0 +1,105 @@
|
||||
# WHP day-one enforce overrides for coraza-spoa.
|
||||
#
|
||||
# Global mode in config.yaml is SecRuleEngine DetectionOnly. The rule ID
|
||||
# ranges below are promoted to enforcement individually, chosen for very
|
||||
# low false-positive rate on the kinds of customer traffic seen on WHP
|
||||
# (WordPress, WooCommerce, Divi page builders).
|
||||
#
|
||||
# When bumping the upstream coraza-spoa pin (and thus the bundled CRS):
|
||||
# 1. Skim the CRS CHANGELOG for new/changed rules in these ID ranges.
|
||||
# 2. Verify they're still high-confidence before promoting the new image.
|
||||
# 3. Smoke-test in staging detect-only mode for 24h before flipping enforce.
|
||||
#
|
||||
# Per-customer false-positive tuning lives in a future per-customer
|
||||
# override mechanism; v1 is server-wide.
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 930120 — LFI: explicit traversal to sensitive system files
|
||||
# (/etc/passwd, /proc/self/, /.ssh/, /etc/shadow, /etc/group, etc.)
|
||||
# Unambiguous probe pattern; no legitimate site path leads here.
|
||||
# Note: 930xxx as a whole includes broader traversal patterns that can FP
|
||||
# on legitimate relative-path file browsers — keep those detect-only.
|
||||
# ---------------------------------------------------------------------------
|
||||
SecRuleUpdateActionById 930120 "ctl:ruleEngine=On"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 932100-932160 — RCE: Unix shell command injection
|
||||
# Patterns like `; cat /etc/passwd`, `|whoami`, backtick `\`uname\``,
|
||||
# $(...) substitution, &&/|| chaining with shell builtins.
|
||||
# Don't appear in normal POST bodies, URL params, or headers. Targeting
|
||||
# these is unambiguous attempted command execution.
|
||||
# ---------------------------------------------------------------------------
|
||||
SecRuleUpdateActionById 932100-932160 "ctl:ruleEngine=On"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 933170-933200 — PHP Webshell access patterns
|
||||
# Direct requests to known webshell paths: c99.php, r57.php, b374k.php,
|
||||
# wso.php, alfa.php, mini.php, etc. Almost universally reconnaissance
|
||||
# scanning for post-exploitation. Even legitimate WordPress installs
|
||||
# never serve these paths.
|
||||
# ---------------------------------------------------------------------------
|
||||
SecRuleUpdateActionById 933170-933200 "ctl:ruleEngine=On"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 944100-944300 — Log4Shell / JNDI injection
|
||||
# `${jndi:ldap://}`, `${jndi:rmi://}`, and obfuscated variants thereof
|
||||
# in headers, query strings, or bodies. Even our PHP/Node stack isn't
|
||||
# vulnerable, but blocking at the edge keeps logs clean and protects
|
||||
# any future Java workloads.
|
||||
# ---------------------------------------------------------------------------
|
||||
SecRuleUpdateActionById 944100-944300 "ctl:ruleEngine=On"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 920440 — URL file extension restricted by policy
|
||||
# Catches probes for backup / config / dump files: .bak, .old, .save,
|
||||
# .swp, .sql, .dist, .backup. Promoted to enforce after empirical
|
||||
# observation on whp01 (2026-05-12, first ~30 min of detect-only):
|
||||
# 124 events, all backup-file recon — `/wp-config.php.old`,
|
||||
# `/db_backup.sql`, `/.env.save`, `/releases.sql`, etc. — from a
|
||||
# single GCP-hosted scanner. Zero false positives observed; standard
|
||||
# WP/WooCommerce/Divi/HPR URLs do not end in these extensions.
|
||||
# ---------------------------------------------------------------------------
|
||||
SecRuleUpdateActionById 920440 "ctl:ruleEngine=On"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 930130 — Restricted File Access Attempt
|
||||
# Catches dotfile / VCS / config-disclosure probes: .env (and .env.local /
|
||||
# .env.bak / .env.save variants), .git/config, config.php at root or under
|
||||
# /admin /backend, etc. Distinct from 930120 (system file paths like
|
||||
# /etc/passwd); this targets application secret files.
|
||||
#
|
||||
# Promoted to enforce on the same observation pass that justified 920440:
|
||||
# 117 events split across joshuaknapp.net (136), cgdannyb.com (51),
|
||||
# onlinesupplements.net (23) — all `.env`-class disclosure probes.
|
||||
# Zero false positives observed. Notably, HPR's `/ccdn.php?filename=...`
|
||||
# audio delivery path does NOT trigger this rule — verified empirically.
|
||||
# ---------------------------------------------------------------------------
|
||||
SecRuleUpdateActionById 930130 "ctl:ruleEngine=On"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Rule families intentionally kept at DETECT-ONLY for v1 — high FP rate
|
||||
# on customer mix. Promote individually after observation:
|
||||
#
|
||||
# 913xxx (Scanner UAs)— matches legitimate ActivityPub federation
|
||||
# (Mastodon's "...Bot" UA) and SiteLockSpider (a
|
||||
# paid customer-security service some sites use).
|
||||
# Observed on whp01 burn-in 2026-05-13:
|
||||
# 20/185 hits = ~11% FP rate on HPR + greggfranklin
|
||||
# + suchascream. Detection adds anomaly score
|
||||
# either way; enforce upside is low.
|
||||
# 941xxx (XSS) — Divi rich-text editor saves, TinyMCE submissions
|
||||
# 942xxx (SQLi) — WP admin queries reflected in params
|
||||
# 920xxx (other) — most 920xxx rules; 920440 specifically promoted above
|
||||
# 933150 — PHP injection FP on WooCommerce checkout
|
||||
# (`session_start` literal appearing in billing form data)
|
||||
# 950xxx-953xxx — Data leakage / backup-file disclosure (mixed FP)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# RESERVED RULE-ID RANGE: 990000000 – 990999999
|
||||
# WHP's coraza_rule_manager generates per-host-exception rules in this range
|
||||
# (rule ID = 990000000 + target_rule_id). Do NOT add new rules in this range
|
||||
# from any other source. When bumping the coraza-spoa pin, check the CRS
|
||||
# changelog for new rules with 9-digit IDs (rare but possible) and re-namespace
|
||||
# if collision risk emerges.
|
||||
# ---------------------------------------------------------------------------
|
||||
+436
-33
@@ -16,9 +16,56 @@ import tempfile
|
||||
import threading
|
||||
import time
|
||||
import re
|
||||
import fcntl
|
||||
|
||||
app = Flask(__name__)
|
||||
|
||||
# Default page server (port 8080) — served to HAProxy clients whose request hit
|
||||
# an unconfigured domain OR whose IP is blocked. Defined at module level so
|
||||
# gunicorn can import it from start-up.sh; previously this was created inside
|
||||
# the __main__ block, which prevented out-of-process WSGI servers from reaching
|
||||
# it. Routes accept ALL HTTP methods because HAProxy proxies the original
|
||||
# request verb unchanged — a POST to a blocked domain would otherwise 405,
|
||||
# which is just log noise.
|
||||
default_app = Flask('haproxy_default')
|
||||
default_app.template_folder = 'templates'
|
||||
|
||||
_ANY_METHOD = ['GET', 'POST', 'PUT', 'DELETE', 'PATCH', 'HEAD', 'OPTIONS']
|
||||
|
||||
|
||||
@default_app.route('/', methods=_ANY_METHOD)
|
||||
def default_page():
|
||||
"""Serve the default page for unmatched domains."""
|
||||
return render_template(
|
||||
'default_page.html',
|
||||
page_title=os.environ.get('HAPROXY_DEFAULT_PAGE_TITLE', 'Site Not Configured'),
|
||||
main_message=os.environ.get(
|
||||
'HAPROXY_DEFAULT_MAIN_MESSAGE',
|
||||
'This domain has not been configured yet. Please contact your '
|
||||
'system administrator to set up this website.'
|
||||
),
|
||||
secondary_message=os.environ.get(
|
||||
'HAPROXY_DEFAULT_SECONDARY_MESSAGE',
|
||||
'If you believe this is an error, please check the domain name '
|
||||
'and try again.'
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@default_app.route('/blocked-ip', methods=_ANY_METHOD)
|
||||
def blocked_ip_page():
|
||||
"""Serve the blocked IP page for blocked clients (HTTP 403)."""
|
||||
return render_template('blocked_ip_page.html'), 403
|
||||
|
||||
|
||||
@default_app.route('/suspended', methods=_ANY_METHOD)
|
||||
def suspended_page():
|
||||
"""Serve the suspended-site page (HTTP 503) for hosts listed in
|
||||
/etc/haproxy/suspended_domains.list. Routed here via the frontend
|
||||
path-rewrite ACL when HAPROXY_SUSPENSION_ENABLED=true."""
|
||||
return render_template('suspended_page.html'), 503
|
||||
|
||||
|
||||
# Configuration
|
||||
DB_FILE = '/etc/haproxy/haproxy_config.db'
|
||||
TEMPLATE_DIR = Path('templates')
|
||||
@@ -137,6 +184,69 @@ def validate_ip_address(ip_string):
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
# Certbot uses fasteners (fcntl-based) to serialize concurrent invocations.
|
||||
# When a previous certbot run is SIGKILLed mid-execution (container restart,
|
||||
# OOM, manual kill), the kernel releases the fcntl lock automatically — but
|
||||
# the LOCK FILE on disk persists. Subsequent runs sometimes report
|
||||
# "Another instance of Certbot is already running" anyway, blocking SSL
|
||||
# issuance until someone manually clears the files.
|
||||
#
|
||||
# Our hung-process scenario (observed 2026-05-09 during the bundling rollout):
|
||||
# certbot from a previous attempt sat in defunct state holding the lock fd.
|
||||
# Once the process eventually exited, the locks were physically removable but
|
||||
# the symptoms persisted across multiple subsequent attempts.
|
||||
#
|
||||
# This helper probes each known lock path with fcntl.LOCK_NB. If we get the
|
||||
# lock, no real process holds it and the file is stale — we delete it. If we
|
||||
# DON'T get the lock, a real certbot is running and we leave it alone (so we
|
||||
# never accidentally trigger concurrent certbot runs).
|
||||
CERTBOT_LOCK_PATHS = (
|
||||
'/etc/letsencrypt/.certbot.lock',
|
||||
'/var/lib/letsencrypt/.certbot.lock',
|
||||
'/var/log/letsencrypt/.certbot.lock',
|
||||
)
|
||||
|
||||
def clear_stale_certbot_locks():
|
||||
"""Remove stale certbot lock files. Safe to call before any ACME run.
|
||||
Returns {'cleared': [paths...], 'held': [paths...]} for logging.
|
||||
"""
|
||||
cleared, held = [], []
|
||||
for path in CERTBOT_LOCK_PATHS:
|
||||
if not os.path.exists(path):
|
||||
continue
|
||||
try:
|
||||
fd = os.open(path, os.O_RDWR)
|
||||
except FileNotFoundError:
|
||||
continue
|
||||
except Exception as e:
|
||||
held.append(f'{path} (open: {e})')
|
||||
continue
|
||||
try:
|
||||
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
except BlockingIOError:
|
||||
# A real process holds it; do not touch.
|
||||
os.close(fd)
|
||||
held.append(path)
|
||||
continue
|
||||
try:
|
||||
# We hold the lock now. Release before unlinking so the lock
|
||||
# state is clean if someone races us.
|
||||
fcntl.flock(fd, fcntl.LOCK_UN)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
os.close(fd)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
os.remove(path)
|
||||
cleared.append(path)
|
||||
except FileNotFoundError:
|
||||
cleared.append(path)
|
||||
except Exception as e:
|
||||
held.append(f'{path} (unlink: {e})')
|
||||
return {'cleared': cleared, 'held': held}
|
||||
|
||||
def find_certbot_live_dir(base_domain):
|
||||
"""Find the most recent certbot live directory for a domain.
|
||||
Certbot creates -NNNN suffixed dirs for repeated requests."""
|
||||
@@ -403,6 +513,9 @@ def request_ssl():
|
||||
return jsonify({'status': 'error', 'message': 'Domain is required'}), 400
|
||||
|
||||
try:
|
||||
# Defensive: clear any stale lock left by a SIGKILLed prior run.
|
||||
clear_stale_certbot_locks()
|
||||
|
||||
# Request Let's Encrypt certificate
|
||||
result = subprocess.run([
|
||||
'certbot', 'certonly', '-n', '--standalone',
|
||||
@@ -455,11 +568,245 @@ def request_ssl():
|
||||
log_operation('request_ssl', False, str(e))
|
||||
return jsonify({'status': 'error', 'message': str(e)}), 500
|
||||
|
||||
def _cleanup_superseded_lineages(keep_path, keep_lineage, bundle_names):
|
||||
"""Remove cert files + certbot lineages that the just-issued bundle supersedes.
|
||||
|
||||
A `.pem` in /etc/haproxy/certs/ is "superseded" iff its certificate's CN
|
||||
is one of the bundle's names AND the file isn't the bundle's own combined
|
||||
file. We don't look at SANs of the OLD certs — being the CN is enough,
|
||||
since that's what HAProxy SNI-matches against and what the file
|
||||
convention names it after.
|
||||
|
||||
Also drops the corresponding certbot renewal config so `certbot renew`
|
||||
stops trying to renew the dead lineage on its next 12h cron tick.
|
||||
|
||||
Returns a small summary dict for logging / API response.
|
||||
"""
|
||||
summary = {'removed': [], 'errors': [], 'skipped': []}
|
||||
|
||||
if not os.path.isdir(SSL_CERTS_DIR):
|
||||
return summary
|
||||
|
||||
keep_basename = os.path.basename(keep_path)
|
||||
|
||||
for fname in sorted(os.listdir(SSL_CERTS_DIR)):
|
||||
if not fname.endswith('.pem'):
|
||||
continue
|
||||
if fname == keep_basename:
|
||||
continue
|
||||
fpath = os.path.join(SSL_CERTS_DIR, fname)
|
||||
try:
|
||||
cn_proc = subprocess.run(
|
||||
['openssl', 'x509', '-in', fpath, '-noout', '-subject', '-nameopt', 'multiline'],
|
||||
capture_output=True, text=True
|
||||
)
|
||||
if cn_proc.returncode != 0:
|
||||
summary['skipped'].append({'file': fname, 'reason': 'openssl read failed'})
|
||||
continue
|
||||
# `-nameopt multiline` lays out the subject one RDN per line; CN is
|
||||
# the row matching `commonName`. Robust against unusual subject orderings.
|
||||
cn = None
|
||||
for line in cn_proc.stdout.splitlines():
|
||||
line = line.strip()
|
||||
if line.startswith('commonName'):
|
||||
# format: "commonName = example.com"
|
||||
parts = line.split('=', 1)
|
||||
if len(parts) == 2:
|
||||
cn = parts[1].strip()
|
||||
break
|
||||
if not cn:
|
||||
summary['skipped'].append({'file': fname, 'reason': 'no CN found'})
|
||||
continue
|
||||
except Exception as e:
|
||||
summary['skipped'].append({'file': fname, 'reason': f'inspect failed: {e}'})
|
||||
continue
|
||||
|
||||
if cn not in bundle_names:
|
||||
continue # not superseded — different domain group
|
||||
|
||||
# This file's CN is now part of our new bundle — supersede it.
|
||||
lineage_name = fname[:-len('.pem')]
|
||||
if lineage_name == keep_lineage:
|
||||
# Defensive: shouldn't happen because of keep_basename check, but
|
||||
# don't accidentally drop the lineage we just wrote.
|
||||
continue
|
||||
|
||||
try:
|
||||
os.remove(fpath)
|
||||
removed_entry = {'file': fname, 'cn': cn, 'lineage_deleted': False}
|
||||
# Best-effort certbot lineage delete. Some files may not have a
|
||||
# corresponding lineage (e.g. self-signed dev certs); ignore those.
|
||||
try:
|
||||
cb_proc = subprocess.run(
|
||||
['certbot', 'delete', '--cert-name', lineage_name, '-n'],
|
||||
capture_output=True, text=True
|
||||
)
|
||||
removed_entry['lineage_deleted'] = (cb_proc.returncode == 0)
|
||||
if cb_proc.returncode != 0:
|
||||
removed_entry['certbot_stderr'] = (cb_proc.stderr or '').strip()[:200]
|
||||
except Exception as e:
|
||||
removed_entry['certbot_error'] = str(e)
|
||||
summary['removed'].append(removed_entry)
|
||||
except Exception as e:
|
||||
summary['errors'].append({'file': fname, 'error': str(e)})
|
||||
|
||||
return summary
|
||||
|
||||
@app.route('/api/ssl/bundle', methods=['POST'])
|
||||
@require_api_key
|
||||
def request_ssl_bundle():
|
||||
"""Issue a single Let's Encrypt cert covering multiple SANs.
|
||||
|
||||
Used by WHP's per-site bundling: one ACME order, one combined .pem,
|
||||
one DB row update per included name. Replaces N separate single-domain
|
||||
/api/ssl calls when a site has multiple domains.
|
||||
|
||||
Body:
|
||||
{"primary": "example.com", "sans": ["www.example.com", ...]}
|
||||
|
||||
The cert lineage uses --cert-name <primary>, so renewal under the same
|
||||
name doesn't proliferate -0001/-0002 dirs (the issue we hit with the
|
||||
legacy single-domain flow). The combined PEM is written to
|
||||
/etc/haproxy/certs/<primary>.pem; HAProxy matches SNI against the cert's
|
||||
SAN list, so this single file serves all included names.
|
||||
"""
|
||||
data = request.get_json() or {}
|
||||
primary = (data.get('primary') or '').strip()
|
||||
sans = data.get('sans') or []
|
||||
|
||||
if not primary:
|
||||
log_operation('request_ssl_bundle', False, 'primary not provided')
|
||||
return jsonify({'status': 'error', 'message': '"primary" is required'}), 400
|
||||
|
||||
# Basic shape validation. certbot will hard-validate the rest.
|
||||
domain_re = re.compile(
|
||||
r'^(?:\*\.)?(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}$',
|
||||
re.IGNORECASE,
|
||||
)
|
||||
if not domain_re.match(primary):
|
||||
return jsonify({'status': 'error', 'message': f'invalid primary: {primary!r}'}), 400
|
||||
|
||||
# Build the unique ordered name list — primary first, then de-duped SANs.
|
||||
if not isinstance(sans, list):
|
||||
return jsonify({'status': 'error', 'message': '"sans" must be a list'}), 400
|
||||
cleaned_sans = []
|
||||
for s in sans:
|
||||
if not isinstance(s, str):
|
||||
return jsonify({'status': 'error', 'message': f'invalid SAN entry: {s!r}'}), 400
|
||||
s = s.strip()
|
||||
if not s:
|
||||
continue
|
||||
if not domain_re.match(s):
|
||||
return jsonify({'status': 'error', 'message': f'invalid SAN: {s!r}'}), 400
|
||||
cleaned_sans.append(s)
|
||||
|
||||
seen = {primary}
|
||||
names = [primary]
|
||||
for s in cleaned_sans:
|
||||
if s not in seen:
|
||||
names.append(s)
|
||||
seen.add(s)
|
||||
|
||||
# Let's Encrypt allows up to 100 names per cert.
|
||||
if len(names) > 100:
|
||||
return jsonify({
|
||||
'status': 'error',
|
||||
'message': f'Too many SANs ({len(names)}); Let\'s Encrypt limit is 100',
|
||||
}), 400
|
||||
|
||||
cmd = [
|
||||
'certbot', 'certonly', '-n', '--standalone',
|
||||
'--preferred-challenges', 'http', '--http-01-port=8688',
|
||||
'--cert-name', primary,
|
||||
]
|
||||
for n in names:
|
||||
cmd.extend(['-d', n])
|
||||
|
||||
try:
|
||||
# Defensive: clear any stale lock left by a SIGKILLed prior run.
|
||||
clear_stale_certbot_locks()
|
||||
|
||||
result = subprocess.run(cmd, capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
stderr_excerpt = (result.stderr or '').strip()[:800]
|
||||
error_msg = f'Failed to obtain SSL bundle for {primary}: {stderr_excerpt}'
|
||||
log_operation('request_ssl_bundle', False, error_msg)
|
||||
return jsonify({
|
||||
'status': 'error',
|
||||
'message': error_msg,
|
||||
'primary': primary,
|
||||
'attempted_names': names,
|
||||
}), 500
|
||||
|
||||
# Locate the lineage. With --cert-name primary, this should be a
|
||||
# stable directory name (no -NNNN suffix on the first issuance).
|
||||
live_dir = find_certbot_live_dir(primary)
|
||||
if not live_dir:
|
||||
error_msg = f'Bundle issued but live dir not found for {primary}'
|
||||
log_operation('request_ssl_bundle', False, error_msg)
|
||||
return jsonify({'status': 'error', 'message': error_msg}), 500
|
||||
|
||||
cert_path = os.path.join(live_dir, 'fullchain.pem')
|
||||
key_path = os.path.join(live_dir, 'privkey.pem')
|
||||
combined_path = f'{SSL_CERTS_DIR}/{primary}.pem'
|
||||
|
||||
os.makedirs(SSL_CERTS_DIR, exist_ok=True)
|
||||
with open(combined_path, 'w') as combined:
|
||||
subprocess.run(['cat', cert_path, key_path], stdout=combined)
|
||||
|
||||
# Mark every name in the bundle as ssl_enabled, all pointing at the
|
||||
# same combined .pem. HAProxy serves one file for many SNI hostnames.
|
||||
with sqlite3.connect(DB_FILE) as conn:
|
||||
cursor = conn.cursor()
|
||||
for n in names:
|
||||
cursor.execute('''
|
||||
UPDATE domains
|
||||
SET ssl_enabled = 1, ssl_cert_path = ?
|
||||
WHERE domain = ?
|
||||
''', (combined_path, n))
|
||||
conn.commit()
|
||||
cursor.close()
|
||||
|
||||
# Clean up superseded lineages. When the bundle covers names that were
|
||||
# previously each in their own single-SAN -0001/-0002 lineage, those
|
||||
# older .pem files coexist in /etc/haproxy/certs/ and get loaded by the
|
||||
# `bind ... ssl crt /etc/haproxy/certs` directive. HAProxy then picks
|
||||
# one of them by alphabetical/load order — frequently the older
|
||||
# single-SAN file — and the new bundle has no effect on what's served.
|
||||
# This block deletes those superseded files (and their certbot lineage)
|
||||
# before the generate_config() reload so HAProxy picks up the bundle.
|
||||
cleanup_summary = _cleanup_superseded_lineages(
|
||||
keep_path=combined_path,
|
||||
keep_lineage=primary,
|
||||
bundle_names=set(names),
|
||||
)
|
||||
|
||||
generate_config()
|
||||
log_operation(
|
||||
'request_ssl_bundle', True,
|
||||
f'SSL bundle issued for {primary} covering {len(names)} names; '
|
||||
f'cleaned up {len(cleanup_summary["removed"])} superseded lineage(s)'
|
||||
)
|
||||
return jsonify({
|
||||
'status': 'success',
|
||||
'primary': primary,
|
||||
'names': names,
|
||||
'cert_path': combined_path,
|
||||
'cleanup': cleanup_summary,
|
||||
'message': f'Bundled certificate obtained for {len(names)} names',
|
||||
})
|
||||
except Exception as e:
|
||||
log_operation('request_ssl_bundle', False, str(e))
|
||||
return jsonify({'status': 'error', 'message': str(e)}), 500
|
||||
|
||||
@app.route('/api/certificates/renew', methods=['POST'])
|
||||
@require_api_key
|
||||
def renew_certificates():
|
||||
"""Renew all certificates and reload HAProxy"""
|
||||
try:
|
||||
# Defensive: clear any stale lock left by a SIGKILLed prior run.
|
||||
clear_stale_certbot_locks()
|
||||
|
||||
# Run certbot renew
|
||||
result = subprocess.run([
|
||||
'certbot', 'renew', '--quiet'
|
||||
@@ -1371,6 +1718,36 @@ def generate_config():
|
||||
|
||||
config_parts = []
|
||||
|
||||
# Optional Coraza WAF integration. When HAPROXY_CORAZA_SPOE_BACKEND is
|
||||
# set on the haproxy-manager container, we render an extra TCP backend
|
||||
# pointing at a coraza-spoa sidecar AND inject a `filter spoe ...` line
|
||||
# into the frontend via hap_listener.tpl. Unset (the default for
|
||||
# standalone deployments, home networks, and any non-WHP use of this
|
||||
# image) -> the generated haproxy.cfg is byte-identical to today's.
|
||||
coraza_spoe_backend = os.environ.get('HAPROXY_CORAZA_SPOE_BACKEND')
|
||||
|
||||
# Optional site-suspension routing. When HAPROXY_SUSPENSION_ENABLED is
|
||||
# set (any truthy value), the frontend gets an ACL that rewrites the
|
||||
# path to /suspended and routes through default-backend for any host
|
||||
# listed in /etc/haproxy/suspended_domains.list. The /suspended Flask
|
||||
# route in this same process returns HTTP 503 + a static page — no
|
||||
# separate container needed (mirrors the existing /blocked-ip pattern).
|
||||
# Same opt-in shape as Coraza: unset -> config byte-identical to today.
|
||||
# We just ensure the list file exists (haproxy refuses to start with
|
||||
# `-f` pointing at a missing file).
|
||||
suspension_raw = os.environ.get('HAPROXY_SUSPENSION_ENABLED', '').strip().lower()
|
||||
suspension_enabled = suspension_raw in ('1', 'true', 'yes', 'on')
|
||||
if suspension_enabled:
|
||||
suspended_list_path = '/etc/haproxy/suspended_domains.list'
|
||||
if not os.path.exists(suspended_list_path):
|
||||
try:
|
||||
with open(suspended_list_path, 'w') as f:
|
||||
f.write('')
|
||||
os.chmod(suspended_list_path, 0o644)
|
||||
logger.info(f"Created empty {suspended_list_path}")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create {suspended_list_path}: {e}")
|
||||
|
||||
# Add Haproxy Default Headers
|
||||
default_headers = template_env.get_template('hap_header.tpl').render()
|
||||
config_parts.append(default_headers)
|
||||
@@ -1380,7 +1757,9 @@ def generate_config():
|
||||
|
||||
# Add Listener Block
|
||||
listener_block = template_env.get_template('hap_listener.tpl').render(
|
||||
crt_path = SSL_CERTS_DIR
|
||||
crt_path = SSL_CERTS_DIR,
|
||||
coraza_spoe_backend = coraza_spoe_backend,
|
||||
suspension_enabled = suspension_enabled,
|
||||
)
|
||||
config_parts.append(listener_block)
|
||||
|
||||
@@ -1500,6 +1879,33 @@ backend default-backend
|
||||
config_parts.append(fallback_backend)
|
||||
# Add Backends
|
||||
config_parts.append('\n' .join(config_backends) + '\n')
|
||||
|
||||
# Coraza WAF backend + SPOE engine config file (only when env var set).
|
||||
# Writing /etc/haproxy/coraza-spoe.cfg here keeps it in sync with the
|
||||
# filter line that hap_listener.tpl just rendered into the frontend.
|
||||
# Explicit trailing '\n' because this is now the LAST config_part —
|
||||
# HAProxy fails parse with "Missing LF on last line" otherwise.
|
||||
if coraza_spoe_backend:
|
||||
coraza_backend_block = template_env.get_template(
|
||||
'hap_coraza_spoa_backend.tpl'
|
||||
).render(agent_target=coraza_spoe_backend)
|
||||
config_parts.append(coraza_backend_block + '\n')
|
||||
|
||||
coraza_spoe_cfg = template_env.get_template(
|
||||
'hap_coraza_spoe_engine.tpl'
|
||||
).render()
|
||||
# HAProxy also rejects this file without a trailing LF
|
||||
# ("Missing LF on last line"). Belt-and-suspenders — even if the
|
||||
# template ends with a newline, Jinja2 can trim it depending on
|
||||
# how the file was authored.
|
||||
if not coraza_spoe_cfg.endswith('\n'):
|
||||
coraza_spoe_cfg += '\n'
|
||||
coraza_spoe_path = '/etc/haproxy/coraza-spoe.cfg'
|
||||
with open(coraza_spoe_path, 'w') as f:
|
||||
f.write(coraza_spoe_cfg)
|
||||
logger.info(f"Coraza SPOE engine config written to {coraza_spoe_path} "
|
||||
f"(SPOA target: {coraza_spoe_backend})")
|
||||
|
||||
# Write complete configuration to tmp
|
||||
temp_config_path = "/etc/haproxy/haproxy.cfg"
|
||||
|
||||
@@ -1776,8 +2182,22 @@ def start_haproxy():
|
||||
log_operation('start_haproxy', False, error_msg)
|
||||
logger.warning("Container will continue without HAProxy running")
|
||||
|
||||
if __name__ == '__main__':
|
||||
def do_initial_setup():
|
||||
"""One-time container-startup setup: DB schema, certbot account, fresh
|
||||
self-signed cert, config generation, and HAProxy launch. Idempotent;
|
||||
safe to re-run, but in prod it should run exactly once per container
|
||||
instance (via scripts/init.py before gunicorn workers spawn) so that
|
||||
start_haproxy() doesn't race with itself across forks.
|
||||
"""
|
||||
init_db()
|
||||
# Clear any stale certbot locks left from a previous container instance
|
||||
# that didn't shut down cleanly. Safe — only removes locks that no live
|
||||
# process holds (verified via fcntl probe).
|
||||
_stale = clear_stale_certbot_locks()
|
||||
if _stale['cleared']:
|
||||
logger.info(f"Cleared stale certbot lock(s) at startup: {_stale['cleared']}")
|
||||
if _stale['held']:
|
||||
logger.warning(f"certbot lock(s) actively held at startup: {_stale['held']}")
|
||||
certbot_register()
|
||||
generate_self_signed_cert(SSL_CERTS_DIR)
|
||||
|
||||
@@ -1792,36 +2212,19 @@ if __name__ == '__main__':
|
||||
start_haproxy()
|
||||
certbot_register()
|
||||
|
||||
# Run Flask app on port 8000 for API and port 8080 for default page
|
||||
|
||||
if __name__ == '__main__':
|
||||
# Direct-invocation path: `python haproxy_manager.py`. Used for local dev
|
||||
# and as a fallback. In the container this runs only when scripts/start-up.sh
|
||||
# is bypassed; production uses gunicorn after scripts/init.py.
|
||||
do_initial_setup()
|
||||
|
||||
# Run both Flask apps on the werkzeug dev server. Acceptable for local
|
||||
# development but NOT production — gunicorn is the prod server, invoked
|
||||
# from scripts/start-up.sh.
|
||||
from threading import Thread
|
||||
|
||||
def run_default_page_server():
|
||||
"""Run a separate Flask app on port 8080 for the default page"""
|
||||
from flask import Flask, render_template
|
||||
default_app = Flask(__name__)
|
||||
default_app.template_folder = 'templates'
|
||||
|
||||
@default_app.route('/')
|
||||
def default_page():
|
||||
"""Serve the default page for unmatched domains"""
|
||||
admin_email = os.environ.get('HAPROXY_ADMIN_EMAIL', 'admin@example.com')
|
||||
|
||||
return render_template('default_page.html',
|
||||
page_title=os.environ.get('HAPROXY_DEFAULT_PAGE_TITLE', 'Site Not Configured'),
|
||||
main_message=os.environ.get('HAPROXY_DEFAULT_MAIN_MESSAGE', 'This domain has not been configured yet. Please contact your system administrator to set up this website.'),
|
||||
secondary_message=os.environ.get('HAPROXY_DEFAULT_SECONDARY_MESSAGE', 'If you believe this is an error, please check the domain name and try again.')
|
||||
)
|
||||
|
||||
@default_app.route('/blocked-ip')
|
||||
def blocked_ip_page():
|
||||
"""Serve the blocked IP page for blocked clients"""
|
||||
return render_template('blocked_ip_page.html'), 403
|
||||
|
||||
default_app.run(host='0.0.0.0', port=8080)
|
||||
|
||||
# Start the default page server in a separate thread
|
||||
default_server_thread = Thread(target=run_default_page_server, daemon=True)
|
||||
default_server_thread.start()
|
||||
|
||||
# Run the main API server
|
||||
Thread(
|
||||
target=lambda: default_app.run(host='0.0.0.0', port=8080),
|
||||
daemon=True,
|
||||
).start()
|
||||
app.run(host='0.0.0.0', port=8000)
|
||||
|
||||
@@ -1,3 +1,7 @@
|
||||
Flask==2.3.3
|
||||
Jinja2==3.1.2
|
||||
psutil
|
||||
# Production WSGI server. Replaces Flask's built-in werkzeug dev server, which
|
||||
# is single-threaded and leaks workers over long uptimes (root cause of the
|
||||
# 2026-05 haproxy-manager "healthy but stalled" incidents).
|
||||
gunicorn==23.0.0
|
||||
|
||||
Executable
+13
@@ -0,0 +1,13 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Container init: DB schema, certbot account, config generation, HAProxy start.
|
||||
|
||||
Runs once per container start, BEFORE gunicorn workers spawn. Keeping init out
|
||||
of the WSGI app's module-load path avoids fork-time races (multiple workers
|
||||
attempting to start_haproxy() simultaneously, certbot lock contention, etc.).
|
||||
"""
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, '/haproxy')
|
||||
import haproxy_manager # noqa: E402 (sys.path manipulation must come first)
|
||||
|
||||
haproxy_manager.do_initial_setup()
|
||||
@@ -20,12 +20,21 @@ log_error() {
|
||||
|
||||
log_info "Starting certificate renewal process"
|
||||
|
||||
# Run certbot renewal
|
||||
if certbot renew --quiet --no-random-sleep-on-renew; then
|
||||
log_info "Certbot renewal completed"
|
||||
# Run certbot renewal — don't exit on failure, some certs may have
|
||||
# renewed successfully even if others failed (e.g., domain no longer
|
||||
# pointed here). Continue to copy/combine whatever succeeded.
|
||||
CERTBOT_OUTPUT=$(certbot renew --no-random-sleep-on-renew 2>&1)
|
||||
CERTBOT_EXIT=$?
|
||||
|
||||
if [ $CERTBOT_EXIT -eq 0 ]; then
|
||||
log_info "Certbot renewal completed successfully"
|
||||
else
|
||||
log_error "Certbot renewal failed with exit code $?"
|
||||
exit 1
|
||||
log_error "Certbot renewal had failures (exit code $CERTBOT_EXIT):"
|
||||
# Log the specific failures
|
||||
echo "$CERTBOT_OUTPUT" | grep -E "Failed to renew|failure" | while read -r line; do
|
||||
log_error " $line"
|
||||
done
|
||||
log_info "Continuing to process successfully renewed certificates..."
|
||||
fi
|
||||
|
||||
# Copy all certificates to HAProxy format
|
||||
|
||||
Regular → Executable
+57
-2
@@ -1,6 +1,61 @@
|
||||
#!/usr/bin/env bash
|
||||
# Container entrypoint. Two-phase startup:
|
||||
# 1. One-shot init (init.py): DB schema, certbot register, config gen, start HAProxy.
|
||||
# Runs synchronously and to completion so haproxy is up before the API binds.
|
||||
# 2. WSGI serving via gunicorn (replacing the Flask dev server). Two gunicorn
|
||||
# instances:
|
||||
# - port 8080 -> default_app (default page + blocked-ip page; HAProxy
|
||||
# proxies unmatched / blocked traffic here)
|
||||
# - port 8000 -> app (management API)
|
||||
#
|
||||
# Why gunicorn:
|
||||
# Flask's built-in werkzeug "development server" is single-threaded and leaks
|
||||
# workers under sustained load. It carried haproxy-manager for a long time but
|
||||
# stalled out around 24-48h uptime ("healthy" health-check, but every request
|
||||
# queued behind a stuck worker). Gunicorn with --max-requests cycles workers
|
||||
# periodically, which prevents the slow-leak failure mode entirely.
|
||||
|
||||
# Exit on error
|
||||
set -eo pipefail
|
||||
|
||||
# Ensure trusted IP whitelist files exist (volume-mounted /etc/haproxy may shadow image defaults)
|
||||
mkdir -p /etc/haproxy
|
||||
[ -f /etc/haproxy/trusted_ips.list ] || : > /etc/haproxy/trusted_ips.list
|
||||
[ -f /etc/haproxy/trusted_ips.map ] || : > /etc/haproxy/trusted_ips.map
|
||||
|
||||
cron &
|
||||
python /haproxy/haproxy_manager.py
|
||||
|
||||
# Phase 1: container init
|
||||
python /haproxy/scripts/init.py
|
||||
|
||||
# Phase 2: WSGI servers
|
||||
# Tunable via env: HAPROXY_MGR_API_WORKERS (default 1), HAPROXY_MGR_API_TIMEOUT
|
||||
# (default 120 — API can do slow ACME calls), HAPROXY_MGR_MAX_REQUESTS (default
|
||||
# 1000 — worker recycle frequency).
|
||||
API_WORKERS="${HAPROXY_MGR_API_WORKERS:-1}"
|
||||
API_TIMEOUT="${HAPROXY_MGR_API_TIMEOUT:-120}"
|
||||
MAX_REQ="${HAPROXY_MGR_MAX_REQUESTS:-1000}"
|
||||
MAX_REQ_JITTER="${HAPROXY_MGR_MAX_REQUESTS_JITTER:-100}"
|
||||
|
||||
# Default page server on :8080. Stays in the background.
|
||||
# --threads 4 lets one worker handle bursts of blocked-IP/default-page hits
|
||||
# without forking. --max-requests recycles the worker to bound memory drift.
|
||||
gunicorn \
|
||||
--bind 0.0.0.0:8080 \
|
||||
--workers 1 --threads 4 --worker-class gthread \
|
||||
--max-requests "${MAX_REQ}" --max-requests-jitter "${MAX_REQ_JITTER}" \
|
||||
--timeout 30 \
|
||||
--access-logfile - --error-logfile - --log-level info \
|
||||
--pythonpath /haproxy \
|
||||
'haproxy_manager:default_app' &
|
||||
|
||||
# Main API server on :8000 in the foreground. exec so signals propagate
|
||||
# correctly and the container exits if the API dies (docker --restart picks it
|
||||
# up). Longer --timeout because cert issuance hits ACME and can take a while.
|
||||
exec gunicorn \
|
||||
--bind 0.0.0.0:8000 \
|
||||
--workers "${API_WORKERS}" --threads 4 --worker-class gthread \
|
||||
--max-requests "${MAX_REQ}" --max-requests-jitter "${MAX_REQ_JITTER}" \
|
||||
--timeout "${API_TIMEOUT}" \
|
||||
--access-logfile - --error-logfile - --log-level info \
|
||||
--pythonpath /haproxy \
|
||||
'haproxy_manager:app'
|
||||
|
||||
@@ -11,7 +11,7 @@ backend {{ name }}-backend
|
||||
http-request set-header X-Forwarded-Proto http if !{ ssl_fc }
|
||||
|
||||
{% for server in servers %}
|
||||
server {{ server.server_name }} {{ server.server_address }}:{{ server.server_port }} {{ server.server_options }}
|
||||
server {{ server.server_name }} {{ server.server_address }}:{{ server.server_port }} {{ server.server_options }} resolvers docker_dns init-addr last,libc,none
|
||||
{% endfor %}
|
||||
|
||||
# SSE-specific backend - optimized for Server-Sent Events long-lived connections
|
||||
@@ -36,5 +36,5 @@ backend {{ name }}-sse-backend
|
||||
http-request set-header X-Forwarded-Proto http if !{ ssl_fc }
|
||||
|
||||
{% for server in servers %}
|
||||
server {{ server.server_name }} {{ server.server_address }}:{{ server.server_port }} {{ server.server_options }}
|
||||
server {{ server.server_name }} {{ server.server_address }}:{{ server.server_port }} {{ server.server_options }} resolvers docker_dns init-addr last,libc,none
|
||||
{% endfor %}
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
# Coraza-SPOA backend.
|
||||
# Only rendered into haproxy.cfg when HAPROXY_CORAZA_SPOE_BACKEND env var is
|
||||
# set on the haproxy-manager container. SPOE traffic to this backend is TCP,
|
||||
# not HTTP. The agent target comes from the env var so a single image can be
|
||||
# deployed against different sidecar host:port pairs (typically the sidecar
|
||||
# container's name + 9000 inside the shared docker network).
|
||||
backend coraza-spoa-backend
|
||||
mode tcp
|
||||
# spop-check actually speaks the SPOE protocol against the agent —
|
||||
# confirms the agent can negotiate a session, not just that the TCP
|
||||
# port is open. Required to detect a half-broken SPOA that's listening
|
||||
# but not actually processing.
|
||||
option spop-check
|
||||
timeout connect 5s
|
||||
timeout server 30s
|
||||
server coraza-spoa {{ agent_target }} check
|
||||
@@ -0,0 +1,54 @@
|
||||
# Coraza SPOE engine configuration.
|
||||
#
|
||||
# Written to /etc/haproxy/coraza-spoe.cfg by haproxy_manager.generate_config()
|
||||
# when HAPROXY_CORAZA_SPOE_BACKEND env var is set. Referenced from haproxy.cfg
|
||||
# via `filter spoe engine coraza config /etc/haproxy/coraza-spoe.cfg`.
|
||||
#
|
||||
# Engine name "coraza" must match the engine name in the filter line in the
|
||||
# main config; group name "coraza-req" must match the send-spoe-group action.
|
||||
# Application name "haproxy" must match the application block in coraza-spoa's
|
||||
# config.yaml.
|
||||
#
|
||||
# Reference: this config follows the shape from coraza-spoa's upstream
|
||||
# example/haproxy/coraza.cfg (v0.7.1). Arg names + ordering are required by
|
||||
# Coraza-SPOA exactly as specified — DO NOT reorder or rename without
|
||||
# coordinating with the agent.
|
||||
|
||||
[coraza]
|
||||
|
||||
spoe-agent coraza
|
||||
# `groups` (not `messages`) lists the spoe-group names this engine offers
|
||||
# via `send-spoe-group` actions. The same group name appears below in a
|
||||
# spoe-group block, which in turn references the actual message.
|
||||
groups coraza-req
|
||||
|
||||
# Prefix for variables the agent sets back on the request transaction —
|
||||
# e.g. var(txn.coraza.error) when set-on-error triggers.
|
||||
option var-prefix coraza
|
||||
|
||||
# On agent error/timeout, set var(txn.coraza.error). We DON'T add a
|
||||
# corresponding `http-request deny if { var(txn.coraza.error) -m bool }`
|
||||
# in the frontend, so the request continues uninspected. This is the
|
||||
# fail-open posture: WAF outage shouldn't 503 customer traffic.
|
||||
option set-on-error error
|
||||
|
||||
timeout hello 2s
|
||||
timeout idle 2m
|
||||
timeout processing 100ms
|
||||
|
||||
use-backend coraza-spoa-backend
|
||||
log global
|
||||
|
||||
# Per-request inspection message. No `event` directive — fires only when
|
||||
# explicitly invoked from haproxy.cfg via `http-request send-spoe-group`.
|
||||
# Arg order/names are mandatory: Coraza-SPOA parses positionally and renames
|
||||
# break the agent. `app=str(haproxy)` is the literal application name from
|
||||
# coraza-spoa's config.yaml `applications:` block.
|
||||
spoe-message coraza-req
|
||||
args app=str(haproxy) src-ip=src src-port=src_port dst-ip=dst dst-port=dst_port method=method path=path query=query version=req.ver headers=req.hdrs body=req.body
|
||||
|
||||
# Group binding for send-spoe-group invocation in the frontend. One group,
|
||||
# one message; could add more in the future (e.g. coraza-res for response
|
||||
# inspection — currently disabled in coraza-spoa's config.yaml).
|
||||
spoe-group coraza-req
|
||||
messages coraza-req
|
||||
@@ -33,6 +33,24 @@ global
|
||||
|
||||
# Stats persistence for zero-downtime reloads
|
||||
stats-file /var/lib/haproxy/stats.dat
|
||||
|
||||
#---------------------------------------------------------------------
|
||||
# DNS resolver for Docker container name resolution
|
||||
# Re-resolves backend server addresses so container IP changes
|
||||
# (from restarts, recreations, scaling) are picked up automatically
|
||||
#---------------------------------------------------------------------
|
||||
resolvers docker_dns
|
||||
nameserver dns1 127.0.0.11:53
|
||||
resolve_retries 3
|
||||
timeout resolve 1s
|
||||
timeout retry 1s
|
||||
hold valid 10s
|
||||
hold other 10s
|
||||
hold refused 10s
|
||||
hold nx 10s
|
||||
hold timeout 10s
|
||||
hold obsolete 10s
|
||||
|
||||
#---------------------------------------------------------------------
|
||||
# common defaults that all the 'listen' and 'backend' sections will
|
||||
# use if not designated in their block
|
||||
|
||||
+57
-11
@@ -23,21 +23,27 @@ frontend web
|
||||
stick-table type ip size 200k expire 10m store conn_cur,conn_rate(10s),http_req_rate(10s),http_err_rate(30s)
|
||||
http-request track-sc0 var(txn.real_ip)
|
||||
|
||||
# Whitelist: let health checks and local traffic bypass rate limits
|
||||
# Whitelist: let health checks, local, and trusted traffic bypass rate limits
|
||||
acl is_local src 127.0.0.0/8 10.0.0.0/8 172.16.0.0/12 192.168.0.0/16
|
||||
acl is_trusted_ip src -f /etc/haproxy/trusted_ips.list
|
||||
acl is_health_check path_beg /.well-known/acme-challenge
|
||||
acl is_whitelisted var(txn.real_ip),map_ip(/etc/haproxy/trusted_ips.map,0) -m int gt 0
|
||||
|
||||
# --- Rate limit rules (applied in order, first match wins) ---
|
||||
# Hard block: >500 req/10s per IP (sustained flood)
|
||||
http-request deny deny_status 429 if { sc_http_req_rate(0) gt 500 } !is_local !is_health_check
|
||||
# Tarpit: >200 req/10s per IP (aggressive scraping / light flood)
|
||||
http-request tarpit deny_status 429 if { sc_http_req_rate(0) gt 200 } !is_local !is_health_check
|
||||
# Connection rate limit: >150 new connections per 10s per IP
|
||||
http-request deny deny_status 429 if { sc_conn_rate(0) gt 150 } !is_local !is_health_check
|
||||
# Concurrent connection limit: >100 simultaneous connections per IP
|
||||
http-request deny deny_status 429 if { sc_conn_cur(0) gt 100 } !is_local !is_health_check
|
||||
# High error rate: >20 errors in 30s (scanner/fuzzer behavior)
|
||||
http-request tarpit deny_status 403 if { sc_http_err_rate(0) gt 20 } !is_local !is_health_check
|
||||
# Thresholds are generous to accommodate media-heavy sites where a
|
||||
# single page can load 100+ images/assets. These only trigger on
|
||||
# obvious automated abuse, not real users.
|
||||
#
|
||||
# Hard block: >5000 req/10s per IP (500 req/s — sustained flood)
|
||||
http-request deny deny_status 429 if { sc_http_req_rate(0) gt 5000 } !is_local !is_trusted_ip !is_whitelisted !is_health_check
|
||||
# Tarpit: >3000 req/10s per IP (300 req/s — aggressive bot/scraper)
|
||||
http-request tarpit deny_status 429 if { sc_http_req_rate(0) gt 3000 } !is_local !is_trusted_ip !is_whitelisted !is_health_check
|
||||
# Connection rate limit: >500 new connections per 10s per IP
|
||||
http-request deny deny_status 429 if { sc_conn_rate(0) gt 500 } !is_local !is_trusted_ip !is_whitelisted !is_health_check
|
||||
# Concurrent connection limit: >500 simultaneous connections per IP
|
||||
http-request deny deny_status 429 if { sc_conn_cur(0) gt 500 } !is_local !is_trusted_ip !is_whitelisted !is_health_check
|
||||
# High error rate: >100 errors in 30s (scanner/fuzzer behavior)
|
||||
http-request tarpit deny_status 403 if { sc_http_err_rate(0) gt 100 } !is_local !is_trusted_ip !is_whitelisted !is_health_check
|
||||
|
||||
# IP blocking using map file (manual blocks only)
|
||||
# Map file format: /etc/haproxy/blocked_ips.map contains "<ip_or_cidr> 1" per line
|
||||
@@ -47,3 +53,43 @@ frontend web
|
||||
acl is_blocked_ip var(txn.real_ip),map_ip(/etc/haproxy/blocked_ips.map,0) -m int gt 0
|
||||
http-request set-path /blocked-ip if is_blocked_ip
|
||||
use_backend default-backend if is_blocked_ip
|
||||
{%- if suspension_enabled %}
|
||||
|
||||
# Site suspension routing. Any Host header listed in
|
||||
# /etc/haproxy/suspended_domains.list is rewritten to /suspended and
|
||||
# routed through default-backend, which is the same Flask app that
|
||||
# serves the default page and blocked-ip page (port 8080 inside this
|
||||
# container). The `/suspended` route returns HTTP 503 with a static
|
||||
# suspension page. External tooling (e.g. WHP's site_disable.php)
|
||||
# maintains the list file via `docker cp`. An empty list is safe —
|
||||
# the ACL simply doesn't match. Sits after IP-blocking so 429/403
|
||||
# still trigger first.
|
||||
acl is_suspended_domain hdr(host),lower -f /etc/haproxy/suspended_domains.list
|
||||
http-request set-path /suspended if is_suspended_domain
|
||||
use_backend default-backend if is_suspended_domain
|
||||
{%- endif %}
|
||||
{%- if coraza_spoe_backend %}
|
||||
|
||||
# Coraza WAF inspection via SPOE. Runs AFTER rate-limit and IP-block
|
||||
# guards (no point asking the WAF about requests we're already dropping)
|
||||
# and AFTER the real-client-IP resolution (so Coraza sees the right src).
|
||||
filter spoe engine coraza config /etc/haproxy/coraza-spoe.cfg
|
||||
http-request send-spoe-group coraza coraza-req
|
||||
|
||||
# Enforce Coraza's verdict. The SPOA sets var(txn.coraza.action) to
|
||||
# "deny" / "drop" / "redirect" when a rule with the corresponding
|
||||
# disruptive action fires (depends on SecRuleEngine mode + per-rule
|
||||
# ctl:ruleEngine overrides). Without these rules, Coraza would inspect
|
||||
# but never block.
|
||||
http-request deny deny_status 403 hdr waf-block "request" if { var(txn.coraza.action) -m str deny }
|
||||
http-response deny deny_status 403 hdr waf-block "response" if { var(txn.coraza.action) -m str deny }
|
||||
http-request silent-drop if { var(txn.coraza.action) -m str drop }
|
||||
http-response silent-drop if { var(txn.coraza.action) -m str drop }
|
||||
http-request redirect code 302 location %[var(txn.coraza.data)] if { var(txn.coraza.action) -m str redirect }
|
||||
http-response redirect code 302 location %[var(txn.coraza.data)] if { var(txn.coraza.action) -m str redirect }
|
||||
|
||||
# FAIL-OPEN on SPOA error. Upstream's example does the opposite — denies
|
||||
# 500 if var(txn.coraza.error) is set — but for a hosting platform we'd
|
||||
# rather lose WAF coverage briefly than 503 customer sites. The error
|
||||
# variable still gets set, so monitoring can observe it.
|
||||
{%- endif %}
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<meta name="robots" content="noindex,nofollow">
|
||||
<title>Site temporarily unavailable</title>
|
||||
<style>
|
||||
body {
|
||||
font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, Oxygen, Ubuntu, Cantarell, sans-serif;
|
||||
text-align: center;
|
||||
padding: 50px 20px;
|
||||
background: linear-gradient(135deg, #1e293b 0%, #0f172a 100%);
|
||||
margin: 0;
|
||||
min-height: 100vh;
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
color: #e2e8f0;
|
||||
}
|
||||
.container {
|
||||
background: #1e293b;
|
||||
border: 1px solid #334155;
|
||||
padding: 40px;
|
||||
border-radius: 12px;
|
||||
box-shadow: 0 10px 30px rgba(0,0,0,0.4);
|
||||
max-width: 560px;
|
||||
width: 100%;
|
||||
}
|
||||
h1 {
|
||||
color: #f1f5f9;
|
||||
margin: 0 0 20px;
|
||||
font-size: 1.75em;
|
||||
font-weight: 600;
|
||||
}
|
||||
p {
|
||||
color: #cbd5e1;
|
||||
line-height: 1.7;
|
||||
margin: 0 0 12px;
|
||||
font-size: 1.05em;
|
||||
}
|
||||
.note {
|
||||
color: #94a3b8;
|
||||
font-size: 0.9em;
|
||||
margin-top: 24px;
|
||||
padding-top: 24px;
|
||||
border-top: 1px solid #334155;
|
||||
}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
<h1>This site is temporarily unavailable.</h1>
|
||||
<p>The site you are trying to reach is currently offline.</p>
|
||||
<p class="note">Site owners: please contact support to restore service.</p>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1 @@
|
||||
172.116.197.166
|
||||
@@ -0,0 +1 @@
|
||||
172.116.197.166 1
|
||||
Reference in New Issue
Block a user