container_runtime: containerd agent: acquisition: [] additionalAcquisition: - labels: type: traefik limit: 1000 query: | {namespace="traefik"} source: loki url: http://loki.prometheus.svc.cluster.local:3100/ wait_for_ready: 30s env: - name: COLLECTIONS value: crowdsecurity/traefik crowdsecurity/base-http-scenarios - name: DISABLE_COLLECTIONS value: crowdsecurity/sshd # Bans on 401/403 bursts hurt more than they protect: with L3 enforcement # a false positive cuts the IP off everything (SSH included), and past # incidents show legit automation (deploy runner, mesh peers, registry # pulls) tripping this probe. Probing/XSS/SQLi/CVE scenarios stay. - name: DISABLE_SCENARIOS value: crowdsecurity/http-generic-bf metrics: enabled: true serviceMonitor: additionalLabels: release: prometheus-stack enabled: true # Static machine identity: agent pods mount pre-created LAPI credentials # (Secret crowdsec-agent-credentials, key local_api_credentials.yaml) # at the exact path the agent entrypoint expects. Together with the # patched register-init (enforced by janitor-cronjob.yaml) the agent # never calls `cscli lapi register` in steady state, so pod names, # restarts and reboots can no longer break it. extraVolumes: - name: static-creds secret: secretName: crowdsec-agent-credentials items: - key: local_api_credentials.yaml path: local_api_credentials.yaml extraVolumeMounts: - name: static-creds mountPath: /tmp_config/local_api_credentials.yaml subPath: local_api_credentials.yaml readOnly: true resources: limits: cpu: 200m memory: 500Mi requests: cpu: 50m memory: 100Mi config: parsers: s02-enrich: mobile-whitelist.yaml: | name: forust/mobile-whitelist description: "Whitelist SWAN/4ka mobile network" whitelist: reason: "Mobile IP whitelist" cidr: - "84.245.64.0/18" # CrowdSec's own guidance: CIDR allowlisting belongs at the parser stage. # A parser whitelist discards the event before it reaches a bucket, so # these addresses never produce an overflow and never become a decision. # A postoverflow whitelist is checked only *after* the ban exists, and # the bouncer answers 403 for as long as it does - which is a window we # do not want the deploy sitting in. local-network.yaml: | name: forust/local-network description: "Whitelist loopback, private and VPN networks" whitelist: reason: "Local network" cidr: - "127.0.0.0/8" - "10.0.0.0/8" - "172.16.0.0/12" - "192.168.0.0/16" # CGNAT range (RFC 6598). The workstation and the k0s node live # here on WireGuard, and 100.64.0.0/10 is not covered by the # RFC 1918 blocks above. - "100.64.0.0/10" - "169.254.0.0/16" - "fc00::/7" - "fe80::/10" vps-whitelist.yaml: | name: forust/vps-whitelist description: "Whitelist static VPS" whitelist: reason: "VPS" ip: - "193.181.211.79" postoverflows: s01-whitelist: # The one whitelist that has to stay here: resolving a hostname is a # network call, and the docs put expensive lookups in postoverflows on # purpose - it runs only when a bucket actually overflows. # ddns.forust.xyz is the public home address, not a private one, so # forust/local-network does not cover it. home-dynamic-ip.yaml: | name: forust/home-dynamic-ip description: "Whitelist home dynamic IP" whitelist: reason: "Home dynamic IP" expression: - evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz") # LAPI-only main config override, merged over config.yaml. NOTE: the # chart's own default for this key is REPLACED, not merged, so its # auto_registration block is repeated verbatim below - drop it and the # agent can no longer register itself. config.yaml.local: | api: server: auto_registration: # Activate if not using TLS for authentication enabled: true token: "${REGISTRATION_TOKEN}" # /!\ Do not modify this variable (auto-generated and handled by the chart) allowed_ranges: # /!\ Make sure to adapt to the pod IP ranges used by your cluster - "127.0.0.1/32" - "192.168.0.0/16" - "10.0.0.0/8" - "172.16.0.0/12" # This homelab has no egress to console.crowdsec.cloud: DNS does # not resolve. The LAPI kept trying anyway ("Signal push: N # signals to push", "capi metrics: sending" every 10s) and each # attempt sat on a resolver timeout WHILE HOLDING A WRITE # TRANSACTION, which is what kept stalling per-request decision # lookups even with WAL enabled. Nothing to share and nothing to # pull - turn the Central API off instead of letting it block the # only database writer we have. online_client: sharing: false pull: community: false blocklists: false disable_usage_metrics_export: true db_config: # SQLite without WAL serialises every reader behind the writer's # rollback journal, and the LAPI writes constantly: the agent pushes # Traefik alerts read from Loki, the metrics collector counts # decisions, the bouncer touches "last pull" on every request. # Symptom: decision lookups taking 10-30s (and a second connection # that could not even open the database) while the LAPI sat at 28m # CPU - the process was blocked in fsync, not computing. Every # bouncer-protected request then blew through the plugin timeout and # fail-closed with 403, on every site at once. # The PVC is local-path-retain (hostPath), not a network share, so # WAL is safe here; the crowdsec docs recommend it for exactly this # ("allowing more concurrency in SQLite that will improve # performances in most scenarios"). use_wal: true # Keeps the alert table bounded. At the 5000/7d default the file # reached 54MB in 15 days off the Traefik access log alone, and the # metrics collector counts decisions on a timer; a smaller working # set means fewer full scans. Crowdsec only prunes - SQLite never # shrinks the file, so the size stays until a manual VACUUM. flush: max_items: 1000 max_age: 24h lapi: env: - name: COLLECTIONS value: crowdsecurity/traefik crowdsecurity/base-http-scenarios - name: DISABLE_COLLECTIONS value: crowdsecurity/linux crowdsecurity/sshd metrics: enabled: true serviceMonitor: additionalLabels: release: prometheus-stack enabled: true persistentVolume: config: enabled: true size: 100Mi storageClassName: local-path-retain data: enabled: true size: 1Gi storageClassName: local-path-retain # LAPI answers a blocking /v1/decisions lookup for EVERY bouncer-protected # request (whole Traefik front door), so it is the hot path of the proxy. # At 400m/500Mi it went CPU-throttled and idle lookups measured 1.3-7.4s, # which pushed requests into the bouncer's fail-closed 403. # Single replica on purpose: LAPI is stateful (BoltDB on the `data` PVC, # credentials on the `config` PVC) - two replicas sharing those RWO # volumes would corrupt the decision store. Scale up CPU, not replicas. resources: limits: cpu: 1500m memory: 1Gi requests: cpu: 250m memory: 500Mi service: type: ClusterIP storeLAPICscliCredentialsInSecret: true