diff --git a/crowdsec/k8s/crowdsec-middleware.yaml b/crowdsec/k8s/crowdsec-middleware.yaml index 4c0b3c7..748e863 100644 --- a/crowdsec/k8s/crowdsec-middleware.yaml +++ b/crowdsec/k8s/crowdsec-middleware.yaml @@ -8,17 +8,40 @@ spec: crowdsec-bouncer: enabled: true LogLevel: INFO - CrowdsecMode: live + # `live` blocked on a `GET /v1/decisions` per request, so a burst + # saturated the LAPI and the plugin 403'd IPs that were never banned. + # v1.3.3 ignores UpdateMaxFailure in `live`, so fail-open is only + # reachable in stream mode, which polls into a cache instead - no + # per-request call to saturate. 15s rather than the 60s default: the + # deploy runner shares one public IP with the house, so this bounds + # both how late a ban lands and how long a lifted one lingers. + CrowdsecMode: stream + UpdateIntervalSeconds: 15 + # -1 = never block because the LAPI is unreachable. In v1.3.3 + # handleStreamTicker only clears isCrowdsecStreamHealthy when + # updateMaxFailure != -1, and ServeHTTP 403s once it is false, so this + # makes a CrowdSec outage mean "no protection", not "every site 403". + UpdateMaxFailure: -1 CrowdsecLapiScheme: http CrowdsecLapiHost: crowdsec-service.crowdsec.svc.cluster.local:8080 CrowdsecLapiKeyFile: "/etc/traefik/secrets/traefik-api-key" - # LAPI lookup is SYNCHRONOUS and per-request: the plugin blocks on - # `GET /v1/decisions?ip=...&banned=true` before the request reaches - # the backend, and fails CLOSED (403) if the lookup exceeds the - # timeout. Unset, the fork defaults to 10s, which is an eternity for - # a request path: a single slow LAPI (idle 1.3-7.4s here) turned - # every request into a 10s hang and then a self-inflicted 403. - # 2s keeps the fail-closed path fast and bounded; with the LAPI - # resourced properly (see crowdsec-values.yaml) the lookup is - # sub-100ms and this budget is never hit. - CrowdsecLapiTimeout: "2s" + # Bypasses the bouncer and the decision cache, no LAPI round-trip. + # Keep in sync with forust/local-network in crowdsec-values.yaml. + ClientTrustedIPs: + - "127.0.0.0/8" + - "10.0.0.0/8" + - "172.16.0.0/12" + - "192.168.0.0/16" + - "100.64.0.0/10" + - "169.254.0.0/16" + - "fc00::/7" + - "fe80::/10" + # The name is HTTPTimeoutSeconds, an int in seconds (min 1) - there is + # no CrowdsecLapiTimeout, and an unrecognised key is silently dropped, + # which is how this sat at the 10s default. Nothing rides on it per + # request any more, so this only bounds the stream pull - and too low + # is the dangerous direction: the LAPI needs ~2s to answer + # /v1/decisions/stream, and a pull that times out leaves the ban cache + # frozen at its startup contents ("failed sending new decisions"), + # i.e. new bans silently never apply. Keep it above the pull latency. + HTTPTimeoutSeconds: 10 diff --git a/crowdsec/k8s/crowdsec-values.yaml b/crowdsec/k8s/crowdsec-values.yaml index 1e1a756..ed51279 100644 --- a/crowdsec/k8s/crowdsec-values.yaml +++ b/crowdsec/k8s/crowdsec-values.yaml @@ -58,26 +58,15 @@ config: reason: "Mobile IP whitelist" cidr: - "84.245.64.0/18" - - postoverflows: - s01-whitelist: - home-dynamic-ip.yaml: | - name: forust/home-dynamic-ip - description: "Whitelist home dynamic IP" - whitelist: - reason: "Home dynamic IP" - expression: - - evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz") - # The hairpin-NAT address of the router (192.168.88.1) is what the - # Gitea Actions runner presents to Traefik - it is NOT the home - # dynamic IP, so the whitelist above did not cover it. During a - # deploy the runner POSTs to the Actions API many times a second; - # a single 403 storm was enough to earn it a 4h ban and break every - # later job. Whitelisting the whole LAN also covers phones and - # tablets browsing over 192.168.88.0/24. - lan.yaml: | - name: forust/lan - description: "Whitelist local network" + # CrowdSec's own guidance: CIDR allowlisting belongs at the parser stage. + # A parser whitelist discards the event before it reaches a bucket, so + # these addresses never produce an overflow and never become a decision. + # A postoverflow whitelist is checked only *after* the ban exists, and + # the bouncer answers 403 for as long as it does - which is a window we + # do not want the deploy sitting in. + local-network.yaml: | + name: forust/local-network + description: "Whitelist loopback, private and VPN networks" whitelist: reason: "Local network" cidr: @@ -85,6 +74,81 @@ config: - "10.0.0.0/8" - "172.16.0.0/12" - "192.168.0.0/16" + # CGNAT range (RFC 6598). The workstation and the k0s node live + # here on WireGuard, and 100.64.0.0/10 is not covered by the + # RFC 1918 blocks above. + - "100.64.0.0/10" + - "169.254.0.0/16" + - "fc00::/7" + - "fe80::/10" + + postoverflows: + s01-whitelist: + # The one whitelist that has to stay here: resolving a hostname is a + # network call, and the docs put expensive lookups in postoverflows on + # purpose - it runs only when a bucket actually overflows. + # ddns.forust.xyz is the public home address, not a private one, so + # forust/local-network does not cover it. + home-dynamic-ip.yaml: | + name: forust/home-dynamic-ip + description: "Whitelist home dynamic IP" + whitelist: + reason: "Home dynamic IP" + expression: + - evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz") + + # LAPI-only main config override, merged over config.yaml. NOTE: the + # chart's own default for this key is REPLACED, not merged, so its + # auto_registration block is repeated verbatim below - drop it and the + # agent can no longer register itself. + config.yaml.local: | + api: + server: + auto_registration: # Activate if not using TLS for authentication + enabled: true + token: "${REGISTRATION_TOKEN}" # /!\ Do not modify this variable (auto-generated and handled by the chart) + allowed_ranges: # /!\ Make sure to adapt to the pod IP ranges used by your cluster + - "127.0.0.1/32" + - "192.168.0.0/16" + - "10.0.0.0/8" + - "172.16.0.0/12" + # This homelab has no egress to console.crowdsec.cloud: DNS does + # not resolve. The LAPI kept trying anyway ("Signal push: N + # signals to push", "capi metrics: sending" every 10s) and each + # attempt sat on a resolver timeout WHILE HOLDING A WRITE + # TRANSACTION, which is what kept stalling per-request decision + # lookups even with WAL enabled. Nothing to share and nothing to + # pull - turn the Central API off instead of letting it block the + # only database writer we have. + online_client: + sharing: false + pull: + community: false + blocklists: false + disable_usage_metrics_export: true + db_config: + # SQLite without WAL serialises every reader behind the writer's + # rollback journal, and the LAPI writes constantly: the agent pushes + # Traefik alerts read from Loki, the metrics collector counts + # decisions, the bouncer touches "last pull" on every request. + # Symptom: decision lookups taking 10-30s (and a second connection + # that could not even open the database) while the LAPI sat at 28m + # CPU - the process was blocked in fsync, not computing. Every + # bouncer-protected request then blew through the plugin timeout and + # fail-closed with 403, on every site at once. + # The PVC is local-path-retain (hostPath), not a network share, so + # WAL is safe here; the crowdsec docs recommend it for exactly this + # ("allowing more concurrency in SQLite that will improve + # performances in most scenarios"). + use_wal: true + # Keeps the alert table bounded. At the 5000/7d default the file + # reached 54MB in 15 days off the Traefik access log alone, and the + # metrics collector counts decisions on a timer; a smaller working + # set means fewer full scans. Crowdsec only prunes - SQLite never + # shrinks the file, so the size stays until a manual VACUUM. + flush: + max_items: 1000 + max_age: 24h lapi: env: diff --git a/crowdsec/k8s/janitor-cronjob.yaml b/crowdsec/k8s/janitor-cronjob.yaml index bbf1f15..0494517 100644 --- a/crowdsec/k8s/janitor-cronjob.yaml +++ b/crowdsec/k8s/janitor-cronjob.yaml @@ -30,10 +30,14 @@ # 3. ensure the static machine exists, recreating it with the # Secret password if missing (agent retry loops reconnect # on their own - same name + same password); -# 4. prune bouncer entries idle for 30d; -# 5. delete decisions from LePresidente/http-generic-403-bf, a hub -# scenario that bans an IP for 4h after 5 POST-403s in 10s and -# therefore bans us for our own bouncer's fail-closed 403s. +# 4. prune bouncer entries idle for 30d. +# +# It used to also delete LePresidente/http-generic-403-bf decisions hourly. +# That was a workaround for the bouncer failing closed on a slow LAPI and +# 403-ing the deploy runner into a 4h ban. The bouncer now polls decisions +# into a cache and never blocks on an unreachable LAPI, so it cannot +# manufacture those 403s any more, and the scenario only fires against real +# scanners - deleting their decisions hourly was undoing a working ban. # # Manual apply (crowdsec/k8s is NOT managed by deploy.yaml): # kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml @@ -196,19 +200,3 @@ spec: fi echo "== 4. prune stale bouncers (no pull for 30d) ==" $LAPI_EXEC cscli bouncers prune -d 720h --force - echo "== 5. drop http-403-bf decisions (4h self-bans) ==" - # `LePresidente/http-generic-403-bf` (hub item - # crowdsecurity/http-generic-bf v0.9) bans any source IP - # after 5 POSTs answered 403 within 10s, for 4h. That - # includes 403s this homelab generates ITSELF (any - # bouncer fail-closed, any app CSRF/rate-limit 403), and a - # 4h ban on the runner/home IP silently breaks deploys and - # browsing. The scenario cannot be removed per-scenario - - # it is baked into a hub item, and disabling the whole - # base-http-scenarios collection would drop ~40 useful - # detections. Instead we keep the detection and drop its - # decisions hourly; the LAN/home whitelists in - # crowdsec-values.yaml handle the legit sources, so this - # only ever hits real scanners (who are re-banned anyway). - $LAPI_EXEC cscli decisions delete \ - --scenario LePresidente/http-generic-403-bf --all || true diff --git a/gitea/k8s/ingress.yaml b/gitea/k8s/ingress.yaml index f12fc3c..3b7bec8 100644 --- a/gitea/k8s/ingress.yaml +++ b/gitea/k8s/ingress.yaml @@ -16,15 +16,12 @@ spec: - name: gitea-service port: 3000 # Registry route: NO crowdsec-bouncer. - # The bouncer plugin does a blocking `GET /v1/decisions` to the LAPI on - # *every* request. A deploy burst (runner Action API polls, `docker - # manifest inspect` per own image, containerd pulls, smoke probes) fires - # hundreds of parallel registry calls; LAPI saturation pushed the lookup - # past the plugin timeout, and the bouncer fail-closed with 403 - which - # containerd surfaces as ErrImagePull/ImagePullBackOff on the next pod. - # This route only serves authenticated OCI traffic (registry tokens, - # basic-auth already handled by gitea) and scanners get nothing useful - # from /v2, so there is no bruteforce surface to protect here. + # A deploy burst (runner Action API polls, `docker manifest inspect` per + # own image, containerd pulls, smoke probes) fires hundreds of parallel + # registry calls, and a ban on the runner breaks every later job. This + # route only serves authenticated OCI traffic - registry tokens and + # basic-auth are already handled by gitea - and scanners get nothing + # useful from /v2, so there is no bruteforce surface to protect here. - match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`) kind: Rule services: diff --git a/netbird/k8s/ingress.yaml b/netbird/k8s/ingress.yaml index a1d762e..663148c 100644 --- a/netbird/k8s/ingress.yaml +++ b/netbird/k8s/ingress.yaml @@ -7,12 +7,18 @@ spec: entryPoints: - websecure routes: + # NO crowdsec-bouncer on the API routes. These are the mesh client's own + # endpoints: gRPC-gateway management calls plus signal/relay long-polling, + # authenticated by NetBird's token rather than by a login form. A ban here + # is self-defeating - the client needs the mesh to reach anything else, so + # CrowdSec banning it locks the peer out of the network it needs to + # function. It also backfires: a banned peer keeps retrying, every retry + # is another 403, and LePresidente/http-generic-403-bf turns five 403s in + # ten seconds into a 4h ban, so one 403 loop kept re-arming the ban. + # netbird-local below has always been exempt; this makes prod match. - match: Host(`nb.forust.xyz`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`)) kind: Rule priority: 100 - middlewares: - - name: crowdsec-bouncer - namespace: crowdsec services: - name: netbird-server-service port: 80 @@ -20,9 +26,6 @@ spec: - match: Host(`nb.forust.xyz`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`)) kind: Rule priority: 100 - middlewares: - - name: crowdsec-bouncer - namespace: crowdsec services: - name: netbird-server-service port: 80