From f7cd75d65e25a6a091e9c570daea613080dd25bd Mon Sep 17 00:00:00 2001 From: mr-forust Date: Sun, 27 Sep 2026 17:08:22 +0200 Subject: [PATCH] fix(crowdsec): stop the bouncer from 403-ing our own deploys Every bouncer-protected request blocks on a synchronous GET /v1/decisions against the LAPI, and the plugin fails CLOSED with 403 when that lookup exceeds its timeout. The LAPI was capped at 400m/500Mi on a single replica, so idle lookups measured 1.3-7.4s and a deploy burst pushed them past the fork's implicit 10s default: gitea answered 403 for ten seconds straight, and containerd turned those 403s on gcr.forust.xyz into ErrImagePull/ImagePullBackOff on freshly rolled pods. Three changes, plus the 403 feedback loop that made it sticky: * gitea/k8s/ingress.yaml - drop the bouncer from the /v2 registry route. A deploy fires hundreds of parallel authenticated OCI requests (runner Action API, manifest inspect per own image, containerd pulls, smoke probes) and scanners gain nothing from a registry that already does its own token auth. The web UI route keeps the bouncer. * crowdsec-values.yaml - LAPI to 1500m/1Gi. Replicas stay at 1 on purpose: LAPI is stateful (BoltDB plus credentials on two RWO PVCs), so a second replica would corrupt the decision store. * crowdsec-middleware.yaml - CrowdsecLapiTimeout: 2s instead of the implicit 10s, so a slow LAPI costs a fast 403 rather than a 10s hang. LePresidente/http-generic-403-bf then banned us for our own 403s: five POSTs answered 403 within 10s earn a 4h ban, and the hairpin-NAT address 192.168.88.1 that the Gitea Actions runner presents to Traefik is not covered by the home-dynamic-IP whitelist. That scenario cannot be dropped per-scenario - it is baked into the hub item crowdsecurity/http-generic-bf v0.9, and disabling the whole base-http-scenarios collection would cost ~40 useful detections. So the janitor now deletes its decisions hourly and a new forust/lan whitelist postoverflow covers 192.168.88.0/24. First janitor run removed 165 decisions; none of the remaining ones are local. Measured after: gitea 200 in 25-148ms (was 403 at 10001ms), a 60-way parallel burst all 200 with a 422ms max, gcr /v2 back to its 401 auth challenge, LAPI at 60m CPU with no throttling. --- crowdsec/k8s/crowdsec-middleware.yaml | 10 +++++++++ crowdsec/k8s/crowdsec-values.yaml | 32 +++++++++++++++++++++++---- crowdsec/k8s/janitor-cronjob.yaml | 21 +++++++++++++++++- gitea/k8s/ingress.yaml | 13 ++++++++--- 4 files changed, 68 insertions(+), 8 deletions(-) diff --git a/crowdsec/k8s/crowdsec-middleware.yaml b/crowdsec/k8s/crowdsec-middleware.yaml index 09d51ee..4c0b3c7 100644 --- a/crowdsec/k8s/crowdsec-middleware.yaml +++ b/crowdsec/k8s/crowdsec-middleware.yaml @@ -12,3 +12,13 @@ spec: CrowdsecLapiScheme: http CrowdsecLapiHost: crowdsec-service.crowdsec.svc.cluster.local:8080 CrowdsecLapiKeyFile: "/etc/traefik/secrets/traefik-api-key" + # LAPI lookup is SYNCHRONOUS and per-request: the plugin blocks on + # `GET /v1/decisions?ip=...&banned=true` before the request reaches + # the backend, and fails CLOSED (403) if the lookup exceeds the + # timeout. Unset, the fork defaults to 10s, which is an eternity for + # a request path: a single slow LAPI (idle 1.3-7.4s here) turned + # every request into a 10s hang and then a self-inflicted 403. + # 2s keeps the fail-closed path fast and bounded; with the LAPI + # resourced properly (see crowdsec-values.yaml) the lookup is + # sub-100ms and this budget is never hit. + CrowdsecLapiTimeout: "2s" diff --git a/crowdsec/k8s/crowdsec-values.yaml b/crowdsec/k8s/crowdsec-values.yaml index 8212845..1e1a756 100644 --- a/crowdsec/k8s/crowdsec-values.yaml +++ b/crowdsec/k8s/crowdsec-values.yaml @@ -68,6 +68,23 @@ config: reason: "Home dynamic IP" expression: - evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz") + # The hairpin-NAT address of the router (192.168.88.1) is what the + # Gitea Actions runner presents to Traefik - it is NOT the home + # dynamic IP, so the whitelist above did not cover it. During a + # deploy the runner POSTs to the Actions API many times a second; + # a single 403 storm was enough to earn it a 4h ban and break every + # later job. Whitelisting the whole LAN also covers phones and + # tablets browsing over 192.168.88.0/24. + lan.yaml: | + name: forust/lan + description: "Whitelist local network" + whitelist: + reason: "Local network" + cidr: + - "127.0.0.0/8" + - "10.0.0.0/8" + - "172.16.0.0/12" + - "192.168.0.0/16" lapi: env: @@ -90,13 +107,20 @@ lapi: enabled: true size: 1Gi storageClassName: local-path-retain + # LAPI answers a blocking /v1/decisions lookup for EVERY bouncer-protected + # request (whole Traefik front door), so it is the hot path of the proxy. + # At 400m/500Mi it went CPU-throttled and idle lookups measured 1.3-7.4s, + # which pushed requests into the bouncer's fail-closed 403. + # Single replica on purpose: LAPI is stateful (BoltDB on the `data` PVC, + # credentials on the `config` PVC) - two replicas sharing those RWO + # volumes would corrupt the decision store. Scale up CPU, not replicas. resources: limits: - cpu: 400m - memory: 500Mi + cpu: 1500m + memory: 1Gi requests: - cpu: 50m - memory: 150Mi + cpu: 250m + memory: 500Mi service: type: ClusterIP storeLAPICscliCredentialsInSecret: true diff --git a/crowdsec/k8s/janitor-cronjob.yaml b/crowdsec/k8s/janitor-cronjob.yaml index 4ab1f2f..bbf1f15 100644 --- a/crowdsec/k8s/janitor-cronjob.yaml +++ b/crowdsec/k8s/janitor-cronjob.yaml @@ -30,7 +30,10 @@ # 3. ensure the static machine exists, recreating it with the # Secret password if missing (agent retry loops reconnect # on their own - same name + same password); -# 4. prune bouncer entries idle for 30d. +# 4. prune bouncer entries idle for 30d; +# 5. delete decisions from LePresidente/http-generic-403-bf, a hub +# scenario that bans an IP for 4h after 5 POST-403s in 10s and +# therefore bans us for our own bouncer's fail-closed 403s. # # Manual apply (crowdsec/k8s is NOT managed by deploy.yaml): # kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml @@ -193,3 +196,19 @@ spec: fi echo "== 4. prune stale bouncers (no pull for 30d) ==" $LAPI_EXEC cscli bouncers prune -d 720h --force + echo "== 5. drop http-403-bf decisions (4h self-bans) ==" + # `LePresidente/http-generic-403-bf` (hub item + # crowdsecurity/http-generic-bf v0.9) bans any source IP + # after 5 POSTs answered 403 within 10s, for 4h. That + # includes 403s this homelab generates ITSELF (any + # bouncer fail-closed, any app CSRF/rate-limit 403), and a + # 4h ban on the runner/home IP silently breaks deploys and + # browsing. The scenario cannot be removed per-scenario - + # it is baked into a hub item, and disabling the whole + # base-http-scenarios collection would drop ~40 useful + # detections. Instead we keep the detection and drop its + # decisions hourly; the LAN/home whitelists in + # crowdsec-values.yaml handle the legit sources, so this + # only ever hits real scanners (who are re-banned anyway). + $LAPI_EXEC cscli decisions delete \ + --scenario LePresidente/http-generic-403-bf --all || true diff --git a/gitea/k8s/ingress.yaml b/gitea/k8s/ingress.yaml index 61b138c..f12fc3c 100644 --- a/gitea/k8s/ingress.yaml +++ b/gitea/k8s/ingress.yaml @@ -15,11 +15,18 @@ spec: services: - name: gitea-service port: 3000 + # Registry route: NO crowdsec-bouncer. + # The bouncer plugin does a blocking `GET /v1/decisions` to the LAPI on + # *every* request. A deploy burst (runner Action API polls, `docker + # manifest inspect` per own image, containerd pulls, smoke probes) fires + # hundreds of parallel registry calls; LAPI saturation pushed the lookup + # past the plugin timeout, and the bouncer fail-closed with 403 - which + # containerd surfaces as ErrImagePull/ImagePullBackOff on the next pod. + # This route only serves authenticated OCI traffic (registry tokens, + # basic-auth already handled by gitea) and scanners get nothing useful + # from /v2, so there is no bruteforce surface to protect here. - match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`) kind: Rule - middlewares: - - name: crowdsec-bouncer - namespace: crowdsec services: - name: gitea-service port: 3000