Compare commits
17
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5c2cba3ac1 | ||
|
|
7f0bd5f609 | ||
|
|
043923fc64 | ||
|
|
f0f8a35b0f | ||
|
|
f5f389b440 | ||
|
|
7cca330438 | ||
|
|
5936de3e56 | ||
|
|
c98ea8b957 | ||
|
|
34c33697bb | ||
|
|
82949613db | ||
|
|
47d788ca13 | ||
|
|
1e66b5f344 | ||
|
|
d638a2c1f9 | ||
|
|
b8512c6033 | ||
|
|
3e057ea18d | ||
|
|
77113fb629 | ||
|
|
a3a0ab92b7 |
No files matched your search
@@ -48,5 +48,5 @@ jobs:
|
|||||||
docker run --rm \
|
docker run --rm \
|
||||||
-v "$PWD:/work" \
|
-v "$PWD:/work" \
|
||||||
-w /work \
|
-w /work \
|
||||||
renovate/renovate:44.83.2 \
|
renovate/renovate:44.97.2 \
|
||||||
renovate-config-validator renovate.json
|
renovate-config-validator renovate.json
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
name: renovate-run
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
inputs:
|
||||||
|
repositories:
|
||||||
|
description: "Repositories to scan (comma-separated)"
|
||||||
|
required: false
|
||||||
|
default: "forust/homelab"
|
||||||
|
log_level:
|
||||||
|
description: "Renovate log level"
|
||||||
|
required: false
|
||||||
|
default: "info"
|
||||||
|
type: choice
|
||||||
|
options:
|
||||||
|
- info
|
||||||
|
- debug
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: renovate-run
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
run-renovate:
|
||||||
|
runs-on: [self-hosted, linux, arch, homelab]
|
||||||
|
steps:
|
||||||
|
- name: Checkout repository
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Run Renovate
|
||||||
|
shell: bash
|
||||||
|
env:
|
||||||
|
RENOVATE_TOKEN: ${{ secrets.RENOVATE_TOKEN }}
|
||||||
|
RENOVATE_GITHUB_COM_TOKEN: ${{ secrets.RENOVATE_GITHUB_COM_TOKEN }}
|
||||||
|
RENOVATE_REPOSITORIES: ${{ inputs.repositories }}
|
||||||
|
LOG_LEVEL: ${{ inputs.log_level }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
: "${RENOVATE_TOKEN:?missing RENOVATE_TOKEN secret — add a renovate-bot PAT in repo/org Actions secrets}"
|
||||||
|
|
||||||
|
docker run --rm \
|
||||||
|
-v "$PWD/renovate/config.js:/opt/renovate/config.js:ro" \
|
||||||
|
-e RENOVATE_PLATFORM=gitea \
|
||||||
|
-e RENOVATE_ENDPOINT=https://gitea.forust.xyz/api/v1 \
|
||||||
|
-e RENOVATE_TOKEN="$RENOVATE_TOKEN" \
|
||||||
|
-e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \
|
||||||
|
-e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \
|
||||||
|
-e RENOVATE_CONFIG_FILE=/opt/renovate/config.js \
|
||||||
|
-e RENOVATE_BASE_DIR=/tmp/renovate \
|
||||||
|
-e LOG_LEVEL="${LOG_LEVEL:-info}" \
|
||||||
|
renovate/renovate:44.97.2
|
||||||
@@ -138,6 +138,21 @@ jobs:
|
|||||||
--values "$repo/prometheus-stack/k8s/grafana-values.yaml" \
|
--values "$repo/prometheus-stack/k8s/grafana-values.yaml" \
|
||||||
--wait
|
--wait
|
||||||
fi
|
fi
|
||||||
|
if [ -f "$repo/loki/k8s/active" ]; then
|
||||||
|
echo "== Upgrading loki/alloy =="
|
||||||
|
helm repo add grafana https://grafana.github.io/helm-charts >/dev/null 2>&1 || true
|
||||||
|
helm repo update grafana >/dev/null 2>&1 || true
|
||||||
|
helm upgrade --install loki grafana/loki \
|
||||||
|
--version 7.3.0 \
|
||||||
|
--namespace prometheus \
|
||||||
|
--values "$repo/loki/k8s/loki-values.yaml" \
|
||||||
|
--wait
|
||||||
|
helm upgrade --install alloy grafana/alloy \
|
||||||
|
--version 1.12.1 \
|
||||||
|
--namespace prometheus \
|
||||||
|
--values "$repo/loki/k8s/alloy-values.yaml" \
|
||||||
|
--wait
|
||||||
|
fi
|
||||||
|
|
||||||
if [ "${#other_files[@]}" -gt 0 ]; then
|
if [ "${#other_files[@]}" -gt 0 ]; then
|
||||||
echo " resources: ${other_files[*]}"
|
echo " resources: ${other_files[*]}"
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
secret.yaml
|
||||||
Whitespace-only changes.
@@ -0,0 +1,37 @@
|
|||||||
|
apiVersion: apps/v1
|
||||||
|
kind: Deployment
|
||||||
|
metadata:
|
||||||
|
name: cloudflared
|
||||||
|
labels:
|
||||||
|
app: cloudflared
|
||||||
|
spec:
|
||||||
|
replicas: 1
|
||||||
|
selector:
|
||||||
|
matchLabels:
|
||||||
|
app: cloudflared
|
||||||
|
template:
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app: cloudflared
|
||||||
|
spec:
|
||||||
|
containers:
|
||||||
|
- name: cloudflared
|
||||||
|
image: cloudflare/cloudflared:2026.1.1
|
||||||
|
imagePullPolicy: IfNotPresent
|
||||||
|
args:
|
||||||
|
- tunnel
|
||||||
|
- --no-autoupdate
|
||||||
|
- run
|
||||||
|
env:
|
||||||
|
- name: TUNNEL_TOKEN
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: cloudflared-secrets
|
||||||
|
key: TUNNEL_TOKEN
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
memory: "32Mi"
|
||||||
|
cpu: "30m"
|
||||||
|
limits:
|
||||||
|
memory: "128Mi"
|
||||||
|
cpu: "200m"
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
apiVersion: v1
|
||||||
|
kind: Secret
|
||||||
|
metadata:
|
||||||
|
name: cloudflared-secrets
|
||||||
|
type: Opaque
|
||||||
|
stringData:
|
||||||
|
TUNNEL_TOKEN: your_tunnel_token_here
|
||||||
@@ -1,57 +1,102 @@
|
|||||||
container_runtime: containerd
|
container_runtime: containerd
|
||||||
|
|
||||||
agent:
|
agent:
|
||||||
|
acquisition: []
|
||||||
|
additionalAcquisition:
|
||||||
|
- labels:
|
||||||
|
type: traefik
|
||||||
|
limit: 1000
|
||||||
|
query: |
|
||||||
|
{namespace="traefik"}
|
||||||
|
source: loki
|
||||||
|
url: http://loki.prometheus.svc.cluster.local:3100/
|
||||||
|
wait_for_ready: 30s
|
||||||
env:
|
env:
|
||||||
- name: COLLECTIONS
|
- name: COLLECTIONS
|
||||||
value: "crowdsecurity/traefik crowdsecurity/base-http-scenarios"
|
value: crowdsecurity/traefik crowdsecurity/base-http-scenarios
|
||||||
- name: DISABLE_COLLECTIONS
|
- name: DISABLE_COLLECTIONS
|
||||||
value: "crowdsecurity/linux crowdsecurity/sshd"
|
value: crowdsecurity/sshd
|
||||||
|
metrics:
|
||||||
acquisition:
|
enabled: true
|
||||||
- namespace: traefik
|
serviceMonitor:
|
||||||
podName: "*traefik*"
|
additionalLabels:
|
||||||
program: traefik
|
release: prometheus-stack
|
||||||
poll_without_inotify: true
|
enabled: true
|
||||||
|
# Static machine identity: agent pods mount pre-created LAPI credentials
|
||||||
|
# (Secret crowdsec-agent-credentials, key local_api_credentials.yaml)
|
||||||
|
# at the exact path the agent entrypoint expects. Together with the
|
||||||
|
# patched register-init (enforced by janitor-cronjob.yaml) the agent
|
||||||
|
# never calls `cscli lapi register` in steady state, so pod names,
|
||||||
|
# restarts and reboots can no longer break it.
|
||||||
|
extraVolumes:
|
||||||
|
- name: static-creds
|
||||||
|
secret:
|
||||||
|
secretName: crowdsec-agent-credentials
|
||||||
|
items:
|
||||||
|
- key: local_api_credentials.yaml
|
||||||
|
path: local_api_credentials.yaml
|
||||||
|
extraVolumeMounts:
|
||||||
|
- name: static-creds
|
||||||
|
mountPath: /tmp_config/local_api_credentials.yaml
|
||||||
|
subPath: local_api_credentials.yaml
|
||||||
|
readOnly: true
|
||||||
resources:
|
resources:
|
||||||
requests:
|
|
||||||
cpu: 50m
|
|
||||||
memory: 100Mi
|
|
||||||
limits:
|
limits:
|
||||||
cpu: 200m
|
cpu: 200m
|
||||||
memory: 500Mi
|
memory: 500Mi
|
||||||
|
requests:
|
||||||
|
cpu: 50m
|
||||||
|
memory: 100Mi
|
||||||
|
|
||||||
|
config:
|
||||||
|
parsers:
|
||||||
|
s02-enrich:
|
||||||
|
mobile-whitelist.yaml: |
|
||||||
|
name: forust/mobile-whitelist
|
||||||
|
description: "Whitelist SWAN/4ka mobile network"
|
||||||
|
whitelist:
|
||||||
|
reason: "Mobile IP whitelist"
|
||||||
|
cidr:
|
||||||
|
- "84.245.64.0/18"
|
||||||
|
|
||||||
|
postoverflows:
|
||||||
|
s01-whitelist:
|
||||||
|
home-dynamic-ip.yaml: |
|
||||||
|
name: forust/home-dynamic-ip
|
||||||
|
description: "Whitelist home dynamic IP"
|
||||||
|
whitelist:
|
||||||
|
reason: "Home dynamic IP"
|
||||||
|
expression:
|
||||||
|
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
|
||||||
|
|
||||||
lapi:
|
lapi:
|
||||||
env:
|
env:
|
||||||
- name: COLLECTIONS
|
- name: COLLECTIONS
|
||||||
value: "crowdsecurity/traefik crowdsecurity/base-http-scenarios"
|
value: crowdsecurity/traefik crowdsecurity/base-http-scenarios
|
||||||
- name: DISABLE_COLLECTIONS
|
- name: DISABLE_COLLECTIONS
|
||||||
value: "crowdsecurity/linux crowdsecurity/sshd"
|
value: crowdsecurity/linux crowdsecurity/sshd
|
||||||
service:
|
metrics:
|
||||||
type: ClusterIP
|
enabled: true
|
||||||
persistentVolume:
|
serviceMonitor:
|
||||||
data:
|
additionalLabels:
|
||||||
|
release: prometheus-stack
|
||||||
enabled: true
|
enabled: true
|
||||||
storageClassName: local-path-retain
|
persistentVolume:
|
||||||
size: 1Gi
|
|
||||||
config:
|
config:
|
||||||
enabled: true
|
enabled: true
|
||||||
storageClassName: local-path-retain
|
|
||||||
size: 100Mi
|
size: 100Mi
|
||||||
storeLAPICscliCredentialsInSecret: true
|
storageClassName: local-path-retain
|
||||||
|
data:
|
||||||
|
enabled: true
|
||||||
|
size: 1Gi
|
||||||
|
storageClassName: local-path-retain
|
||||||
resources:
|
resources:
|
||||||
requests:
|
|
||||||
cpu: 50m
|
|
||||||
memory: 150Mi
|
|
||||||
limits:
|
limits:
|
||||||
cpu: 400m
|
cpu: 400m
|
||||||
memory: 500Mi
|
memory: 500Mi
|
||||||
|
requests:
|
||||||
metrics:
|
cpu: 50m
|
||||||
enabled: true
|
memory: 150Mi
|
||||||
serviceMonitor:
|
service:
|
||||||
additionalLabels:
|
type: ClusterIP
|
||||||
release: prometheus-stack
|
storeLAPICscliCredentialsInSecret: true
|
||||||
enabled: true
|
|
||||||
interval: 30s
|
|
||||||
scrapeTimeout: 10s
|
|
||||||
namespace: prometheus
|
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
apiVersion: v1
|
||||||
|
data:
|
||||||
|
crowdsec-overview.json: "{\n \"__inputs\": [\n {\n \"name\": \"DS_PROMETHEUS\",\n \"label\": \"Prometheus\",\n \"description\": \"\",\n \"type\": \"datasource\",\n \"pluginId\": \"prometheus\",\n \"pluginName\": \"Prometheus\"\n }\n ],\n \"__requires\": [\n {\n \"type\": \"grafana\",\n \"id\": \"grafana\",\n \"name\": \"Grafana\",\n \"version\": \"8.1.2\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"graph\",\n \"name\": \"Graph (old)\",\n \"version\": \"\"\n },\n {\n \"type\": \"datasource\",\n \"id\": \"prometheus\",\n \"name\": \"Prometheus\",\n \"version\": \"1.0.0\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"stat\",\n \"name\": \"Stat\",\n \"version\": \"\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"timeseries\",\n \"name\": \"Time series\",\n \"version\": \"\"\n }\n ],\n \"annotations\": {\n \"list\": [\n {\n \"builtIn\": 1,\n \"datasource\": \"-- Grafana --\",\n \"enable\": true,\n \"hide\": true,\n \"iconColor\": \"rgba(0, 211, 255, 1)\",\n \"name\": \"Annotations & Alerts\",\n \"target\": {\n \"limit\": 100,\n \"matchAny\": false,\n \"tags\": [],\n \"type\": \"dashboard\"\n },\n \"type\": \"dashboard\"\n }\n ]\n },\n \"editable\": true,\n \"gnetId\": null,\n \"graphTooltip\": 0,\n \"id\": null,\n \"links\": [],\n \"panels\": [\n {\n \"collapsed\": false,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 0\n },\n \"id\": 24,\n \"panels\": [],\n \"title\": \"Summary\",\n \"type\": \"row\"\n },\n {\n \"cacheTimeout\": null,\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [\n {\n \"options\": {\n \"match\": \"null\",\n \"result\": {\n \"text\": \"N/A\"\n }\n },\n \"type\": \"special\"\n }\n ],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"#E02F44\",\n \"value\": null\n },\n {\n \"color\": \"#E02F44\",\n \"value\": 10\n },\n {\n \"color\": \"#299c46\",\n \"value\": 10\n }\n ]\n },\n \"unit\": \"none\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 8,\n \"w\": 6,\n \"x\": 0,\n \"y\": 1\n },\n \"id\": 2,\n \"interval\": null,\n \"links\": [],\n \"maxDataPoints\": 100,\n \"options\": {\n \"colorMode\": \"background\",\n \"graphMode\": \"none\",\n \"justifyMode\": \"auto\",\n \"orientation\": \"horizontal\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"text\": {},\n \"textMode\": \"auto\"\n },\n \"pluginVersion\": \"8.1.2\",\n \"targets\": [\n {\n \"exemplar\": true,\n \"expr\": \"count(cs_info)\",\n \"interval\": \"\",\n \"legendFormat\": \"\",\n \"refId\": \"A\"\n }\n ],\n \"timeFrom\": null,\n \"timeShift\": null,\n \"title\": \"Running Crowdsec\",\n \"transparent\": true,\n \"type\": \"stat\"\n },\n {\n \"aliasColors\": {},\n \"bars\": false,\n \"dashLength\": 10,\n \"dashes\": false,\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"decimals\": 1,\n \"fieldConfig\": {\n \"defaults\": {\n \"links\": []\n },\n \"overrides\": []\n },\n \"fill\": 1,\n \"fillGradient\": 0,\n \"gridPos\": {\n \"h\": 8,\n \"w\": 18,\n \"x\": 6,\n \"y\": 1\n },\n \"hiddenSeries\": false,\n \"id\": 8,\n \"legend\": {\n \"alignAsTable\": true,\n \"avg\": false,\n \"current\": false,\n \"max\": false,\n \"min\": false,\n \"rightSide\": true,\n \"show\": true,\n \"sort\": \"total\",\n \"sortDesc\": true,\n \"total\": true,\n \"values\": true\n },\n \"lines\": true,\n \"linewidth\": 1,\n \"nullPointMode\": \"null\",\n \"options\": {\n \"alertThreshold\": true\n },\n \"percentage\": false,\n \"pluginVersion\": \"8.1.2\",\n \"pointradius\": 2,\n \"points\": false,\n \"renLine truncated
|
||||||
|
kind: ConfigMap
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/managed-by: manual
|
||||||
|
grafana_dashboard: "1"
|
||||||
|
name: crowdsec-crowdsec-overview
|
||||||
|
namespace: prometheus
|
||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
data:
|
||||||
|
crowdsec-lapi-metrics.json: "{\n \"__inputs\": [\n {\n \"name\": \"DS_PROMETHEUS\",\n \"label\": \"Prometheus\",\n \"description\": \"\",\n \"type\": \"datasource\",\n \"pluginId\": \"prometheus\",\n \"pluginName\": \"Prometheus\"\n }\n ],\n \"__requires\": [\n {\n \"type\": \"panel\",\n \"id\": \"bargauge\",\n \"name\": \"Bar gauge\",\n \"version\": \"\"\n },\n {\n \"type\": \"grafana\",\n \"id\": \"grafana\",\n \"name\": \"Grafana\",\n \"version\": \"8.1.2\"\n },\n {\n \"type\": \"datasource\",\n \"id\": \"prometheus\",\n \"name\": \"Prometheus\",\n \"version\": \"1.0.0\"\n }\n ],\n \"annotations\": {\n \"list\": [\n {\n \"builtIn\": 1,\n \"datasource\": \"-- Grafana --\",\n \"enable\": true,\n \"hide\": true,\n \"iconColor\": \"rgba(0, 211, 255, 1)\",\n \"name\": \"Annotations & Alerts\",\n \"target\": {\n \"limit\": 100,\n \"matchAny\": false,\n \"tags\": [],\n \"type\": \"dashboard\"\n },\n \"type\": \"dashboard\"\n }\n ]\n },\n \"editable\": true,\n \"gnetId\": null,\n \"graphTooltip\": 0,\n \"id\": null,\n \"iteration\": 1655915193937,\n \"links\": [],\n \"panels\": [\n {\n \"collapsed\": false,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 0\n },\n \"id\": 10,\n \"panels\": [],\n \"title\": \"Agents\",\n \"type\": \"row\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n },\n {\n \"color\": \"red\",\n \"value\": 80\n }\n ]\n }\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 8,\n \"w\": 12,\n \"x\": 0,\n \"y\": 1\n },\n \"id\": 2,\n \"options\": {\n \"displayMode\": \"gradient\",\n \"orientation\": \"vertical\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"showUnfilled\": false,\n \"text\": {}\n },\n \"pluginVersion\": \"8.1.2\",\n \"repeat\": \"query0\",\n \"repeatDirection\": \"h\",\n \"targets\": [\n {\n \"exemplar\": false,\n \"expr\": \"sum(rate(cs_lapi_request_duration_seconds_bucket{endpoint=\\\"/v1/watchers/login\\\", instance=\\\"$lapi\\\"}[$__rate_interval])) by (le)\",\n \"format\": \"heatmap\",\n \"interval\": \"\",\n \"legendFormat\": \"{{le}}\",\n \"refId\": \"A\"\n }\n ],\n \"title\": \"Agents Login\",\n \"type\": \"heatmap\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n }\n ]\n },\n \"unit\": \"none\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 8,\n \"w\": 12,\n \"x\": 12,\n \"y\": 1\n },\n \"id\": 6,\n \"options\": {\n \"displayMode\": \"gradient\",\n \"orientation\": \"auto\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"showUnfilled\": false,\n \"text\": {}\n },\n \"pluginVersion\": \"8.1.2\",\n \"targets\": [\n {\n \"exemplar\": true,\n \"expr\": \"sum(rate(cs_lapi_request_duration_seconds_bucket{endpoint=\\\"/v1/watchers/login\\\"}[$__rate_interval])) by (le)\",\n \"format\": \"heatmap\",\n \"interval\": \"\",\n \"legendFormat\": \"{{le}}\",\n \"refId\": \"A\"\n }\n ],\n \"title\": \"Heartbeat\",\n \"type\": \"heatmap\"\n },\n {\n \"collapsed\": false,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 9\n },\n \"id\": 12,\n \"panels\": [],\n \"title\": \"Decisions\",\n \"type\": \"row\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n Line truncated
|
||||||
|
kind: ConfigMap
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/managed-by: manual
|
||||||
|
grafana_dashboard: "1"
|
||||||
|
name: crowdsec-crowdsec-lapi-metrics
|
||||||
|
namespace: prometheus
|
||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
data:
|
||||||
|
crowdsec-insight.json: "{\n \"__inputs\": [\n {\n \"name\": \"DS_PROMETHEUS\",\n \"label\": \"Prometheus\",\n \"description\": \"\",\n \"type\": \"datasource\",\n \"pluginId\": \"prometheus\",\n \"pluginName\": \"Prometheus\"\n }\n ],\n \"__requires\": [\n {\n \"type\": \"panel\",\n \"id\": \"bargauge\",\n \"name\": \"Bar gauge\",\n \"version\": \"\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"gauge\",\n \"name\": \"Gauge\",\n \"version\": \"\"\n },\n {\n \"type\": \"grafana\",\n \"id\": \"grafana\",\n \"name\": \"Grafana\",\n \"version\": \"8.1.2\"\n },\n {\n \"type\": \"datasource\",\n \"id\": \"prometheus\",\n \"name\": \"Prometheus\",\n \"version\": \"1.0.0\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"stat\",\n \"name\": \"Stat\",\n \"version\": \"\"\n }\n ],\n \"annotations\": {\n \"list\": [\n {\n \"builtIn\": 1,\n \"datasource\": \"-- Grafana --\",\n \"enable\": true,\n \"hide\": true,\n \"iconColor\": \"rgba(0, 211, 255, 1)\",\n \"name\": \"Annotations & Alerts\",\n \"target\": {\n \"limit\": 100,\n \"matchAny\": false,\n \"tags\": [],\n \"type\": \"dashboard\"\n },\n \"type\": \"dashboard\"\n }\n ]\n },\n \"editable\": true,\n \"gnetId\": null,\n \"graphTooltip\": 0,\n \"id\": null,\n \"iteration\": 1655915159751,\n \"links\": [],\n \"panels\": [\n {\n \"collapsed\": true,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 0\n },\n \"id\": 22,\n \"panels\": [\n {\n \"cacheTimeout\": null,\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [\n {\n \"options\": {\n \"match\": \"null\",\n \"result\": {\n \"text\": \"N/A\"\n }\n },\n \"type\": \"special\"\n }\n ],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n },\n {\n \"color\": \"red\",\n \"value\": 80\n }\n ]\n },\n \"unit\": \"dateTimeAsIso\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 9,\n \"w\": 5,\n \"x\": 2,\n \"y\": 1\n },\n \"id\": 2,\n \"interval\": null,\n \"links\": [],\n \"maxDataPoints\": 100,\n \"options\": {\n \"colorMode\": \"none\",\n \"graphMode\": \"none\",\n \"justifyMode\": \"auto\",\n \"orientation\": \"horizontal\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"text\": {},\n \"textMode\": \"auto\"\n },\n \"pluginVersion\": \"8.1.2\",\n \"targets\": [\n {\n \"exemplar\": true,\n \"expr\": \"(process_start_time_seconds{instance=\\\"$instance\\\"})*1000\",\n \"interval\": \"\",\n \"legendFormat\": \"{{instance}}\",\n \"refId\": \"A\"\n }\n ],\n \"timeFrom\": null,\n \"timeShift\": null,\n \"title\": \"Up since\",\n \"type\": \"stat\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"displayName\": \"\",\n \"mappings\": [],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n }\n ]\n },\n \"unit\": \"decbytes\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 9,\n \"w\": 5,\n \"x\": 7,\n \"y\": 1\n },\n \"id\": 4,\n \"options\": {\n \"orientation\": \"auto\",\n \"reduceOptions\": {\n \"calcs\": [\n \"mean\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\Line truncated
|
||||||
|
kind: ConfigMap
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/managed-by: manual
|
||||||
|
grafana_dashboard: "1"
|
||||||
|
name: crowdsec-crowdsec-insight
|
||||||
|
namespace: prometheus
|
||||||
@@ -0,0 +1,195 @@
|
|||||||
|
# CrowdSec self-healing: static machine identity + enforcement loops.
|
||||||
|
#
|
||||||
|
# Problem it fixes: the chart's agent init container runs
|
||||||
|
# `cscli lapi register --machine "$POD_NAME" ...`
|
||||||
|
# unconditionally. Credentials live in an emptyDir, the machine row lives
|
||||||
|
# in LAPI's persistent DB. Any init re-run for an already-known pod name
|
||||||
|
# (kubelet restart, node reboot) dies with
|
||||||
|
# 403 Forbidden: user '<pod>' already exist
|
||||||
|
# and the DaemonSet pod sticks in Init forever. Every DS restart also
|
||||||
|
# leaves an orphan machine row that is never cleaned.
|
||||||
|
#
|
||||||
|
# Design (name-independent):
|
||||||
|
# * Agent identity is a STATIC machine `crowdsec-agent-workstation`
|
||||||
|
# whose password lives in Secret `crowdsec-agent-credentials`
|
||||||
|
# (created once, manually - like all other secrets in this repo).
|
||||||
|
# The secret is mounted into agent pods at
|
||||||
|
# /tmp_config/local_api_credentials.yaml (see extraVolumeMounts in
|
||||||
|
# crowdsec-values.yaml), which is exactly the path the agent's main
|
||||||
|
# container copies into place at startup.
|
||||||
|
# * The DS init command is patched (strategic merge, by container name)
|
||||||
|
# to SKIP registration when that file exists, keeping the legacy
|
||||||
|
# register path only as fallback. Detection marker in the patched
|
||||||
|
# command: `[ -s /tmp_config`.
|
||||||
|
# * This CronJob enforces the desired state hourly, so recovery is
|
||||||
|
# automatic even after `helm upgrade` reverts the DS patch or the
|
||||||
|
# LAPI database is wiped:
|
||||||
|
# 1. patch DS init if it still has the unconditional register
|
||||||
|
# (no-op otherwise - no restart churn);
|
||||||
|
# 2. prune machines with no heartbeat for 2h (orphan hygiene);
|
||||||
|
# 3. ensure the static machine exists, recreating it with the
|
||||||
|
# Secret password if missing (agent retry loops reconnect
|
||||||
|
# on their own - same name + same password);
|
||||||
|
# 4. prune bouncer entries idle for 30d.
|
||||||
|
#
|
||||||
|
# Manual apply (crowdsec/k8s is NOT managed by deploy.yaml):
|
||||||
|
# kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml
|
||||||
|
# Force a run:
|
||||||
|
# kubectl create job -n crowdsec --from=cronjob/crowdsec-janitor janitor-now
|
||||||
|
#
|
||||||
|
# Helm upgrades: the janitor's strategic patch puts the DS field under
|
||||||
|
# the `kubectl-patch` field manager, so a plain `helm upgrade` FAILS
|
||||||
|
# with an SSA conflict on initContainers[].command. Procedure:
|
||||||
|
# 1. revert init to chart state (kills the conflict):
|
||||||
|
# helm template crowdsec crowdsec/crowdsec --version <ver> \
|
||||||
|
# -n crowdsec -f crowdsec/k8s/crowdsec-values.yaml > /tmp/r.yaml
|
||||||
|
# python3 -c "import yaml,json; ..." # build revert patch from
|
||||||
|
# the rendered DaemonSet init command, then
|
||||||
|
# kubectl patch ds crowdsec-agent -n crowdsec \
|
||||||
|
# --type strategic -p "\$(cat /tmp/revert_patch.json)"
|
||||||
|
# 2. helm upgrade --install crowdsec ... (no --force needed)
|
||||||
|
# 3. janitor-now right away (upgrade reverts init; new pods would
|
||||||
|
# sit in Init until the next hourly run otherwise).
|
||||||
|
#
|
||||||
|
# One-time bootstrap (order matters):
|
||||||
|
# 1. Create Secret + static machine (see commands in chat).
|
||||||
|
# 2. Apply this file, trigger janitor-now, wait for agent 1/1.
|
||||||
|
# 3. One-time orphan cleanup:
|
||||||
|
# kubectl exec -n crowdsec deploy/crowdsec-lapi -- \
|
||||||
|
# cscli machines prune --duration 1h --force
|
||||||
|
# 4. Only then `helm upgrade` crowdsec with the extraVolumes values.
|
||||||
|
# Upgrade reverts the DS patch; trigger janitor-now right after it
|
||||||
|
# (otherwise new pods sit in Init until the next hourly run, then
|
||||||
|
# self-heal anyway).
|
||||||
|
#
|
||||||
|
# Password rotation: update the Secret, delete the machine
|
||||||
|
# (`cscli machines delete crowdsec-agent-workstation`), trigger
|
||||||
|
# janitor-now (recreates it), then `kubectl rollout restart
|
||||||
|
# ds/crowdsec-agent -n crowdsec` (agent reads the file at startup only).
|
||||||
|
apiVersion: v1
|
||||||
|
kind: ServiceAccount
|
||||||
|
metadata:
|
||||||
|
name: crowdsec-janitor
|
||||||
|
namespace: crowdsec
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/part-of: crowdsec
|
||||||
|
---
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
kind: Role
|
||||||
|
metadata:
|
||||||
|
name: crowdsec-janitor
|
||||||
|
namespace: crowdsec
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/part-of: crowdsec
|
||||||
|
rules:
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["pods"]
|
||||||
|
verbs: ["get", "list"]
|
||||||
|
- apiGroups: [""]
|
||||||
|
resources: ["pods/exec"]
|
||||||
|
verbs: ["create"]
|
||||||
|
- apiGroups: ["apps"]
|
||||||
|
resources: ["daemonsets"]
|
||||||
|
verbs: ["get", "patch"]
|
||||||
|
# `kubectl exec deploy/<name>` resolves deploy -> replicaset -> pod,
|
||||||
|
# which needs read access to these (exec itself is pods/exec above).
|
||||||
|
- apiGroups: ["apps"]
|
||||||
|
resources: ["deployments", "replicasets"]
|
||||||
|
verbs: ["get", "list"]
|
||||||
|
---
|
||||||
|
apiVersion: rbac.authorization.k8s.io/v1
|
||||||
|
kind: RoleBinding
|
||||||
|
metadata:
|
||||||
|
name: crowdsec-janitor
|
||||||
|
namespace: crowdsec
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/part-of: crowdsec
|
||||||
|
subjects:
|
||||||
|
- kind: ServiceAccount
|
||||||
|
name: crowdsec-janitor
|
||||||
|
namespace: crowdsec
|
||||||
|
roleRef:
|
||||||
|
kind: Role
|
||||||
|
name: crowdsec-janitor
|
||||||
|
apiGroup: rbac.authorization.k8s.io
|
||||||
|
---
|
||||||
|
apiVersion: batch/v1
|
||||||
|
kind: CronJob
|
||||||
|
metadata:
|
||||||
|
name: crowdsec-janitor
|
||||||
|
namespace: crowdsec
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/part-of: crowdsec
|
||||||
|
spec:
|
||||||
|
schedule: "17 * * * *"
|
||||||
|
concurrencyPolicy: Forbid
|
||||||
|
successfulJobsHistoryLimit: 3
|
||||||
|
failedJobsHistoryLimit: 3
|
||||||
|
jobTemplate:
|
||||||
|
spec:
|
||||||
|
activeDeadlineSeconds: 300
|
||||||
|
template:
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app.kubernetes.io/part-of: crowdsec
|
||||||
|
spec:
|
||||||
|
serviceAccountName: crowdsec-janitor
|
||||||
|
restartPolicy: OnFailure
|
||||||
|
containers:
|
||||||
|
- name: janitor
|
||||||
|
# Same image the chart itself uses for registration jobs;
|
||||||
|
# IfNotPresent so it works while the node is offline
|
||||||
|
# (layer cached from the chart install).
|
||||||
|
image: alpine/kubectl:latest
|
||||||
|
imagePullPolicy: IfNotPresent
|
||||||
|
env:
|
||||||
|
- name: AGENT_PASSWORD
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: crowdsec-agent-credentials
|
||||||
|
key: password
|
||||||
|
command:
|
||||||
|
- /bin/sh
|
||||||
|
- -c
|
||||||
|
- |
|
||||||
|
set -eu
|
||||||
|
LAPI_EXEC="kubectl exec -n crowdsec deploy/crowdsec-lapi --"
|
||||||
|
echo "== 1. enforce patched agent init =="
|
||||||
|
CUR=$(kubectl get ds crowdsec-agent -n crowdsec \
|
||||||
|
-o jsonpath='{.spec.template.spec.initContainers[0].command[2]}')
|
||||||
|
case "$CUR" in
|
||||||
|
*'-s /tmp_config'*)
|
||||||
|
echo "init already patched"
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
echo "patching init"
|
||||||
|
WAIT='until nc "$LAPI_HOST" "$LAPI_PORT" -z'
|
||||||
|
WAIT="$WAIT; do echo waiting for lapi to start; sleep 5; done"
|
||||||
|
LINK='ln -s /staging/etc/crowdsec /etc/crowdsec'
|
||||||
|
REG='cscli lapi register --machine "$USERNAME"'
|
||||||
|
REG="$REG -u \"\$LAPI_URL\" --token \"\$REGISTRATION_TOKEN\""
|
||||||
|
CREDS=/tmp_config/local_api_credentials.yaml
|
||||||
|
CMD="$WAIT; $LINK; [ -s $CREDS ] || {"
|
||||||
|
CMD="$CMD $REG && cp"
|
||||||
|
CMD="$CMD /etc/crowdsec/local_api_credentials.yaml $CREDS; }"
|
||||||
|
ESC=$(printf '%s' "$CMD" | sed 's/"/\\"/g')
|
||||||
|
PATCH='{"spec":{"template":{"spec":{"initContainers":'
|
||||||
|
PATCH=$PATCH'[{"name":"wait-for-lapi-and-register",'
|
||||||
|
PATCH=$PATCH'"command":["sh","-c","'$ESC'"]}]}}}}'
|
||||||
|
kubectl patch ds crowdsec-agent -n crowdsec \
|
||||||
|
--type strategic -p "$PATCH"
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
echo "== 2. prune orphan machines (no heartbeat for 2h) =="
|
||||||
|
$LAPI_EXEC cscli machines prune --duration 2h --force
|
||||||
|
echo "== 3. ensure static machine exists =="
|
||||||
|
if $LAPI_EXEC cscli machines inspect \
|
||||||
|
crowdsec-agent-workstation >/dev/null 2>&1; then
|
||||||
|
echo "static machine present"
|
||||||
|
else
|
||||||
|
echo "recreating static machine"
|
||||||
|
$LAPI_EXEC cscli machines add crowdsec-agent-workstation \
|
||||||
|
--password "$AGENT_PASSWORD" --force
|
||||||
|
fi
|
||||||
|
echo "== 4. prune stale bouncers (no pull for 30d) =="
|
||||||
|
$LAPI_EXEC cscli bouncers prune -d 720h --force
|
||||||
@@ -25,3 +25,10 @@ spec:
|
|||||||
ports:
|
ports:
|
||||||
- protocol: TCP
|
- protocol: TCP
|
||||||
port: 8080
|
port: 8080
|
||||||
|
- from:
|
||||||
|
- namespaceSelector:
|
||||||
|
matchLabels:
|
||||||
|
kubernetes.io/metadata.name: prometheus
|
||||||
|
ports:
|
||||||
|
- protocol: TCP
|
||||||
|
port: 6060
|
||||||
Whitespace-only changes.
@@ -0,0 +1,57 @@
|
|||||||
|
# Pinned chart: grafana/alloy 1.12.1 (app v1.19.2).
|
||||||
|
# Install (deferred to deploy task, namespace prometheus):
|
||||||
|
# helm upgrade --install alloy grafana/alloy --version 1.12.1 \
|
||||||
|
# --namespace prometheus --values loki/k8s/alloy-values.yaml --wait
|
||||||
|
# DaemonSet ships k8s pod logs (API-tailed, no hostPath mounts) to Loki.
|
||||||
|
# Scope phase 1: k8s only, compose leftovers out.
|
||||||
|
|
||||||
|
controller:
|
||||||
|
type: daemonset
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
memory: "128Mi"
|
||||||
|
cpu: "50m"
|
||||||
|
limits:
|
||||||
|
memory: "512Mi"
|
||||||
|
cpu: "500m"
|
||||||
|
|
||||||
|
image:
|
||||||
|
tag: "v1.19.2"
|
||||||
|
|
||||||
|
alloy:
|
||||||
|
configMap:
|
||||||
|
create: true
|
||||||
|
content: |
|
||||||
|
discovery.kubernetes "pods" {
|
||||||
|
role = "pod"
|
||||||
|
}
|
||||||
|
|
||||||
|
discovery.relabel "pods" {
|
||||||
|
targets = discovery.kubernetes.pods.targets
|
||||||
|
|
||||||
|
rule {
|
||||||
|
source_labels = ["__meta_kubernetes_namespace"]
|
||||||
|
target_label = "namespace"
|
||||||
|
}
|
||||||
|
|
||||||
|
rule {
|
||||||
|
source_labels = ["__meta_kubernetes_pod_name"]
|
||||||
|
target_label = "pod"
|
||||||
|
}
|
||||||
|
|
||||||
|
rule {
|
||||||
|
source_labels = ["__meta_kubernetes_pod_container_name"]
|
||||||
|
target_label = "container"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
loki.source.kubernetes "pods" {
|
||||||
|
targets = discovery.relabel.pods.output
|
||||||
|
forward_to = [loki.write.default.receiver]
|
||||||
|
}
|
||||||
|
|
||||||
|
loki.write "default" {
|
||||||
|
endpoint {
|
||||||
|
url = "http://loki-gateway.prometheus.svc.cluster.local/loki/api/v1/push"
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,97 @@
|
|||||||
|
# Pinned chart: grafana/loki 7.3.0 (app 3.6.12).
|
||||||
|
# Install (deferred to deploy task, namespace prometheus):
|
||||||
|
# helm upgrade --install loki grafana/loki --version 7.3.0 \
|
||||||
|
# --namespace prometheus --values loki/k8s/loki-values.yaml --wait
|
||||||
|
# SingleBinary, filesystem storage on local-path-retain, 14d retention.
|
||||||
|
# No IngressRoute: Loki is cluster-internal, queried via Grafana datasource.
|
||||||
|
|
||||||
|
deploymentMode: SingleBinary
|
||||||
|
|
||||||
|
loki:
|
||||||
|
# Multitenancy off: single-node homelab, gateway + Alloy + Grafana talk to one tenant.
|
||||||
|
auth_enabled: false
|
||||||
|
image:
|
||||||
|
tag: "3.6.12"
|
||||||
|
commonConfig:
|
||||||
|
# Single replica: default RF=3 would require 3 ingesters and fail all writes.
|
||||||
|
replication_factor: 1
|
||||||
|
storage:
|
||||||
|
type: filesystem
|
||||||
|
schemaConfig:
|
||||||
|
configs:
|
||||||
|
- from: "2024-04-01"
|
||||||
|
store: tsdb
|
||||||
|
object_store: filesystem
|
||||||
|
schema: v13
|
||||||
|
index:
|
||||||
|
prefix: index_
|
||||||
|
period: 24h
|
||||||
|
compactor:
|
||||||
|
retention_enabled: true
|
||||||
|
delete_request_store: filesystem
|
||||||
|
limits_config:
|
||||||
|
retention_period: 336h
|
||||||
|
rulerConfig:
|
||||||
|
wal:
|
||||||
|
dir: /var/loki/ruler-wal
|
||||||
|
storage:
|
||||||
|
type: local
|
||||||
|
local:
|
||||||
|
directory: /var/loki/rules
|
||||||
|
|
||||||
|
singleBinary:
|
||||||
|
replicas: 1
|
||||||
|
persistence:
|
||||||
|
enabled: true
|
||||||
|
size: 20Gi
|
||||||
|
storageClass: local-path-retain
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
memory: "512Mi"
|
||||||
|
cpu: "200m"
|
||||||
|
limits:
|
||||||
|
memory: "2Gi"
|
||||||
|
cpu: "1000m"
|
||||||
|
|
||||||
|
# Zeroed: unused in SingleBinary mode (chart validation requires it).
|
||||||
|
write:
|
||||||
|
replicas: 0
|
||||||
|
read:
|
||||||
|
replicas: 0
|
||||||
|
backend:
|
||||||
|
replicas: 0
|
||||||
|
|
||||||
|
gateway:
|
||||||
|
replicas: 1
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
memory: "64Mi"
|
||||||
|
cpu: "50m"
|
||||||
|
limits:
|
||||||
|
memory: "256Mi"
|
||||||
|
cpu: "300m"
|
||||||
|
|
||||||
|
monitoring:
|
||||||
|
serviceMonitor:
|
||||||
|
enabled: true
|
||||||
|
labels:
|
||||||
|
release: prometheus-stack
|
||||||
|
interval: 15s
|
||||||
|
rules:
|
||||||
|
enabled: true
|
||||||
|
namespace: prometheus
|
||||||
|
labels:
|
||||||
|
release: prometheus-stack
|
||||||
|
|
||||||
|
# Disabled: memcached caches don't fit a memory-tight single node.
|
||||||
|
# SingleBinary works without them (slower repeated queries, fine at homelab scale).
|
||||||
|
resultsCache:
|
||||||
|
enabled: false
|
||||||
|
chunksCache:
|
||||||
|
enabled: false
|
||||||
|
|
||||||
|
# Disabled: synthetic canary traffic + helm test pod, noise on a single node.
|
||||||
|
lokiCanary:
|
||||||
|
enabled: false
|
||||||
|
test:
|
||||||
|
enabled: false
|
||||||
@@ -1,5 +1,10 @@
|
|||||||
route:
|
route:
|
||||||
receiver: default
|
receiver: default
|
||||||
|
routes:
|
||||||
|
- receiver: "null"
|
||||||
|
matchers:
|
||||||
|
- alertname="Watchdog"
|
||||||
|
|
||||||
receivers:
|
receivers:
|
||||||
- name: default
|
- name: default
|
||||||
|
- name: "null"
|
||||||
@@ -1,3 +1,6 @@
|
|||||||
|
templates:
|
||||||
|
- "/etc/alertmanager/config/telegram.tmpl"
|
||||||
|
|
||||||
route:
|
route:
|
||||||
receiver: telegram
|
receiver: telegram
|
||||||
group_by:
|
group_by:
|
||||||
@@ -7,10 +10,10 @@ route:
|
|||||||
group_interval: 5m
|
group_interval: 5m
|
||||||
repeat_interval: 12h
|
repeat_interval: 12h
|
||||||
routes:
|
routes:
|
||||||
- receiver: null
|
- receiver: "null"
|
||||||
matchers:
|
matchers:
|
||||||
- alertname="InfoInhibitor"
|
- alertname="InfoInhibitor"
|
||||||
- receiver: null
|
- receiver: "null"
|
||||||
matchers:
|
matchers:
|
||||||
- alertname="Watchdog"
|
- alertname="Watchdog"
|
||||||
|
|
||||||
@@ -35,4 +38,5 @@ receivers:
|
|||||||
chat_id: REPLACE_WITH_TELEGRAM_CHAT_ID
|
chat_id: REPLACE_WITH_TELEGRAM_CHAT_ID
|
||||||
parse_mode: HTML
|
parse_mode: HTML
|
||||||
send_resolved: true
|
send_resolved: true
|
||||||
- name: null
|
message: '{{ template "telegram.forust.message" . }}'
|
||||||
|
- name: "null"
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
{{- define "telegram.forust.message" -}}
|
||||||
|
{{- $statusEmoji := "🚨" -}}
|
||||||
|
{{- if ne .Status "firing" -}}{{- $statusEmoji = "✅" -}}{{- end -}}
|
||||||
|
{{- $count := len .Alerts.Firing -}}
|
||||||
|
{{- if eq .Status "resolved" -}}{{- $count = len .Alerts.Resolved -}}{{- end -}}
|
||||||
|
{{ $statusEmoji }} <b>{{ .Status | toUpper }} ({{ $count }})</b>
|
||||||
|
{{- range .Alerts }}
|
||||||
|
|
||||||
|
<b>{{ .Labels.alertname }}</b>
|
||||||
|
{{- $sev := .Labels.severity }}
|
||||||
|
{{- if eq $sev "critical" }} 🔥 critical
|
||||||
|
{{- else if eq $sev "warning" }} ⚠️ warning
|
||||||
|
{{- else if eq $sev "info" }} ℹ️ info
|
||||||
|
{{- else if $sev }} • {{ $sev }}
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Annotations.description }}
|
||||||
|
|
||||||
|
<i>{{ .Annotations.description }}</i>
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Labels.namespace }}
|
||||||
|
📦 Namespace: <code>{{ .Labels.namespace }}</code>
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Labels.pod }}
|
||||||
|
📦 Pod: <code>{{ .Labels.pod }}</code>
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Labels.container }}
|
||||||
|
🐳 Container: <code>{{ .Labels.container }}</code>
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Labels.node }}
|
||||||
|
🖥 Node: <code>{{ .Labels.node }}</code>
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Labels.instance }}
|
||||||
|
🖥 Instance: <code>{{ .Labels.instance }}</code>
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Labels.job }}
|
||||||
|
🔧 Job: <code>{{ .Labels.job }}</code>
|
||||||
|
{{- end }}
|
||||||
|
{{- if eq .Status "firing" }}
|
||||||
|
🕐 Since: <code>{{ .StartsAt | date "2006-01-02 15:04 MST" }}</code>
|
||||||
|
{{- else }}
|
||||||
|
🕐 Resolved: <code>{{ .EndsAt | date "2006-01-02 15:04 MST" }}</code>
|
||||||
|
{{- end }}
|
||||||
|
{{- end }}
|
||||||
|
{{- end }}
|
||||||
@@ -66,3 +66,21 @@ spec:
|
|||||||
annotations:
|
annotations:
|
||||||
summary: "Node memory pressure"
|
summary: "Node memory pressure"
|
||||||
description: "Node {{ $labels.instance }} has used more than 90% of memory for 15 minutes."
|
description: "Node {{ $labels.instance }} has used more than 90% of memory for 15 minutes."
|
||||||
|
|
||||||
|
- alert: LokiDown
|
||||||
|
expr: kube_statefulset_status_replicas_unavailable{statefulset="loki"} > 0 or kube_deployment_status_replicas_unavailable{deployment="loki-gateway"} > 0
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Loki is down"
|
||||||
|
description: "Loki in namespace {{ $labels.namespace }} has unavailable replicas for more than 10 minutes. Logs are not queryable."
|
||||||
|
|
||||||
|
- alert: AlloyDown
|
||||||
|
expr: kube_daemonset_status_number_unavailable{daemonset="alloy"} > 0
|
||||||
|
for: 10m
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
annotations:
|
||||||
|
summary: "Alloy is down"
|
||||||
|
description: "Alloy DaemonSet in namespace {{ $labels.namespace }} has {{ $value }} unavailable pods for more than 10 minutes. Pod logs are not being shipped to Loki."
|
||||||
@@ -14,8 +14,11 @@ grafana:
|
|||||||
|
|
||||||
persistence:
|
persistence:
|
||||||
enabled: true
|
enabled: true
|
||||||
storageClassName: local-path-retain
|
# Matches live PVC (10Gi/local-path). Retain migration is a separate task
|
||||||
size: 20Gi
|
# with data migration (see storage-audit doc); do NOT change SC/size here
|
||||||
|
# without migrating, helm upgrade fails on immutable PVC fields.
|
||||||
|
storageClassName: local-path
|
||||||
|
size: 10Gi
|
||||||
|
|
||||||
ingress:
|
ingress:
|
||||||
enabled: false
|
enabled: false
|
||||||
@@ -23,6 +26,12 @@ grafana:
|
|||||||
service:
|
service:
|
||||||
port: 80
|
port: 80
|
||||||
|
|
||||||
|
additionalDataSources:
|
||||||
|
- name: Loki
|
||||||
|
type: loki
|
||||||
|
url: http://loki-gateway.prometheus.svc.cluster.local
|
||||||
|
access: proxy
|
||||||
|
|
||||||
prometheus:
|
prometheus:
|
||||||
prometheusSpec:
|
prometheusSpec:
|
||||||
retention: 60d
|
retention: 60d
|
||||||
@@ -30,7 +39,8 @@ prometheus:
|
|||||||
storageSpec:
|
storageSpec:
|
||||||
volumeClaimTemplate:
|
volumeClaimTemplate:
|
||||||
spec:
|
spec:
|
||||||
storageClassName: "local-path-retain"
|
# Matches live PVC, see note on grafana.persistence above.
|
||||||
|
storageClassName: "local-path"
|
||||||
accessModes:
|
accessModes:
|
||||||
- ReadWriteOnce
|
- ReadWriteOnce
|
||||||
resources:
|
resources:
|
||||||
@@ -48,12 +58,13 @@ alertmanager:
|
|||||||
storage:
|
storage:
|
||||||
volumeClaimTemplate:
|
volumeClaimTemplate:
|
||||||
spec:
|
spec:
|
||||||
storageClassName: local-path-retain
|
# Matches live PVC (20Gi), see note on grafana.persistence above.
|
||||||
|
storageClassName: local-path
|
||||||
accessModes:
|
accessModes:
|
||||||
- ReadWriteOnce
|
- ReadWriteOnce
|
||||||
resources:
|
resources:
|
||||||
requests:
|
requests:
|
||||||
storage: 10Gi
|
storage: 20Gi
|
||||||
|
|
||||||
defaultRules:
|
defaultRules:
|
||||||
disabled:
|
disabled:
|
||||||
|
|||||||
@@ -12,7 +12,6 @@ spec:
|
|||||||
middlewares:
|
middlewares:
|
||||||
- name: crowdsec-bouncer
|
- name: crowdsec-bouncer
|
||||||
namespace: crowdsec
|
namespace: crowdsec
|
||||||
- name: "security-chain@file"
|
|
||||||
services:
|
services:
|
||||||
- name: prometheus-stack-grafana
|
- name: prometheus-stack-grafana
|
||||||
port: 80
|
port: 80
|
||||||
|
|||||||
+4
-1
@@ -1,7 +1,10 @@
|
|||||||
{
|
{
|
||||||
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
|
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
|
||||||
"extends": ["config:recommended"],
|
"extends": ["config:recommended"],
|
||||||
"enabledManagers": ["docker-compose", "kubernetes"],
|
"enabledManagers": ["docker-compose", "kubernetes", "helm-values"],
|
||||||
|
"helm-values": {
|
||||||
|
"managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"]
|
||||||
|
},
|
||||||
"kubernetes": {
|
"kubernetes": {
|
||||||
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
|
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
|
||||||
},
|
},
|
||||||
|
|||||||
+14
-1
@@ -24,12 +24,25 @@ The `renovate/k8s/active` marker makes the normal deployment workflow include
|
|||||||
the namespace, ConfigMap, and CronJob. The Secret is intentionally excluded
|
the namespace, ConfigMap, and CronJob. The Secret is intentionally excluded
|
||||||
from Git and must be applied separately after every new cluster.
|
from Git and must be applied separately after every new cluster.
|
||||||
|
|
||||||
Run it immediately instead of waiting for the six-hour schedule:
|
Run it immediately instead of waiting for the six-hour schedule.
|
||||||
|
|
||||||
|
Two options, both use the same `renovate/config.js`:
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
kubectl create job --from=cronjob/renovate renovate-manual-$(date +%s) -n renovate
|
kubectl create job --from=cronjob/renovate renovate-manual-$(date +%s) -n renovate
|
||||||
```
|
```
|
||||||
|
|
||||||
|
or the `renovate-run` Actions workflow (Actions tab → `renovate-run` →
|
||||||
|
Run workflow). It runs `renovate/renovate:44.97.2` on the self-hosted
|
||||||
|
runner via Docker. Required Actions secrets (repo or org settings):
|
||||||
|
|
||||||
|
- `RENOVATE_TOKEN` — renovate-bot PAT (repository + issue read/write).
|
||||||
|
- `RENOVATE_GITHUB_COM_TOKEN` — optional, for changelogs and GitHub rate limits.
|
||||||
|
|
||||||
|
Inputs: `repositories` (default `forust/homelab`), `log_level`
|
||||||
|
(`info`/`debug`). Only one run at a time (concurrency group
|
||||||
|
`renovate-run`), same as the CronJob `Forbid` policy.
|
||||||
|
|
||||||
Inspect runs with:
|
Inspect runs with:
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
|
|||||||
+4
-1
@@ -1,7 +1,10 @@
|
|||||||
module.exports = {
|
module.exports = {
|
||||||
platform: 'gitea',
|
platform: 'gitea',
|
||||||
endpoint: process.env.RENOVATE_ENDPOINT,
|
endpoint: process.env.RENOVATE_ENDPOINT,
|
||||||
enabledManagers: ['docker-compose', 'kubernetes'],
|
enabledManagers: ['docker-compose', 'kubernetes', 'helm-values'],
|
||||||
|
'helm-values': {
|
||||||
|
managerFilePatterns: ['/k8s/.+values\\.ya?ml$/'],
|
||||||
|
},
|
||||||
kubernetes: {
|
kubernetes: {
|
||||||
managerFilePatterns: ['/k8s/.+\\.ya?ml$/'],
|
managerFilePatterns: ['/k8s/.+\\.ya?ml$/'],
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -8,7 +8,10 @@ data:
|
|||||||
module.exports = {
|
module.exports = {
|
||||||
platform: 'gitea',
|
platform: 'gitea',
|
||||||
endpoint: process.env.RENOVATE_ENDPOINT,
|
endpoint: process.env.RENOVATE_ENDPOINT,
|
||||||
enabledManagers: ['docker-compose', 'kubernetes'],
|
enabledManagers: ['docker-compose', 'kubernetes', 'helm-values'],
|
||||||
|
'helm-values': {
|
||||||
|
managerFilePatterns: ['/k8s/.+values\\.ya?ml$/'],
|
||||||
|
},
|
||||||
kubernetes: {
|
kubernetes: {
|
||||||
managerFilePatterns: ['/k8s/.+\\.ya?ml$/'],
|
managerFilePatterns: ['/k8s/.+\\.ya?ml$/'],
|
||||||
},
|
},
|
||||||
@@ -22,7 +25,10 @@ data:
|
|||||||
dependencyDashboard: true,
|
dependencyDashboard: true,
|
||||||
prCreation: 'immediate',
|
prCreation: 'immediate',
|
||||||
labels: ['dependencies', 'automated'],
|
labels: ['dependencies', 'automated'],
|
||||||
extends: ['config:recommended', ':dependencyDashboard'],
|
extends: [
|
||||||
|
'config:recommended',
|
||||||
|
':dependencyDashboard',
|
||||||
|
],
|
||||||
packageRules: [
|
packageRules: [
|
||||||
{
|
{
|
||||||
description: 'Do not update private homelab images',
|
description: 'Do not update private homelab images',
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ spec:
|
|||||||
restartPolicy: Never
|
restartPolicy: Never
|
||||||
containers:
|
containers:
|
||||||
- name: renovate
|
- name: renovate
|
||||||
image: renovate/renovate:44.102.0
|
image: renovate/renovate:44.103.0
|
||||||
env:
|
env:
|
||||||
- name: RENOVATE_PLATFORM
|
- name: RENOVATE_PLATFORM
|
||||||
value: gitea
|
value: gitea
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
services:
|
services:
|
||||||
renovate:
|
renovate:
|
||||||
image: renovate/renovate:44.83.2
|
image: renovate/renovate:44.97.2
|
||||||
container_name: renovate
|
container_name: renovate
|
||||||
restart: "no"
|
restart: "no"
|
||||||
env_file:
|
env_file:
|
||||||
|
|||||||
Reference in new issue
Block a user