Compare commits
132
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3c4274c8ce | ||
|
|
9021eddbc3 | ||
|
|
2adf17c307 | ||
|
|
1add5b5cd7 | ||
|
|
34f10211ab | ||
|
|
02447f2946 | ||
|
|
8799962b1c | ||
|
|
fb80024fa2 | ||
|
|
0f1a788874 | ||
|
|
39df442e60 | ||
|
|
62a451773e | ||
|
|
f196099491 | ||
|
|
66502b8279 | ||
|
|
9018c091fa | ||
|
|
114608af2f | ||
|
|
51a73fb213 | ||
|
|
939fad4a23 | ||
|
|
12d5cc6202 | ||
|
|
a91862d288 | ||
|
|
a6b84d86b4 | ||
|
|
f2dd9b3fbc | ||
|
|
f3a5cedc80 | ||
|
|
b0282e24f3 | ||
|
|
e208ccae62 | ||
|
|
c65fee7b29 | ||
|
|
d3892d2ed5 | ||
|
|
fb025d221b | ||
|
|
7164b48275 | ||
|
|
3f484a37c6 | ||
|
|
956596e6aa | ||
|
|
3320018232 | ||
|
|
65709e005b | ||
|
|
c24bf90629 | ||
|
|
78e6363fe2 | ||
|
|
22c2f1e108 | ||
|
|
64ba4ce9f1 | ||
|
|
7a81b4ea8b | ||
|
|
0859479c0f | ||
|
|
872f64b887 | ||
|
|
a90fb19ed4 | ||
|
|
6f4cd03f4b | ||
|
|
e56662194b | ||
|
|
4856348a6e | ||
|
|
49d0da1dd0 | ||
|
|
6bc938436f | ||
|
|
04b33d0736 | ||
|
|
5dcad7eb38 | ||
|
|
d0872bc918 | ||
|
|
91dd749a29 | ||
|
|
b6e1dc0362 | ||
|
|
a2af854867 | ||
|
|
314ec0cda7 | ||
|
|
e45ae10204 | ||
|
|
357ee26111 | ||
|
|
436fd1ecae | ||
|
|
a564dd8b67 | ||
|
|
c2c90c490d | ||
|
|
da0f9e84c3 | ||
|
|
3ecc12300a | ||
|
|
19029fa012 | ||
|
|
2a5d690e34 | ||
|
|
b53ce36d89 | ||
|
|
a8c4e9bcbe | ||
|
|
761f97ef8e | ||
|
|
0c9743e241 | ||
|
|
2b3c28a46b | ||
|
|
2cb06debc5 | ||
|
|
a6af69dca0 | ||
|
|
76f39da90c | ||
|
|
86730ff0c4 | ||
|
|
af66d3e7fd | ||
|
|
deeaefa695 | ||
|
|
6999dd2728 | ||
|
|
742d78944b | ||
|
|
e451c97dfc | ||
|
|
33c54ac830 | ||
|
|
b09d718310 | ||
|
|
b4f76373bb | ||
|
|
74adf38d63 | ||
|
|
f5b2f89f38 | ||
|
|
8e63284240 | ||
|
|
1d81410cd8 | ||
|
|
2f891a5d31 | ||
|
|
cda0022d81 | ||
|
|
dde6b1c743 | ||
|
|
7c4843c88c | ||
|
|
f9e4623ade | ||
|
|
2ad4fa1b82 | ||
|
|
2a4f215546 | ||
|
|
16aaeb60c1 | ||
|
|
a5409edbf2 | ||
|
|
c70d2db3a1 | ||
|
|
24dd82e801 | ||
|
|
11e92fdf4e | ||
|
|
b1f98fc148 | ||
|
|
9b91b5847e | ||
|
|
892790822d | ||
|
|
f7cd75d65e | ||
|
|
6a9a460769 | ||
|
|
ac0f845da6 | ||
|
|
f54589a05c | ||
|
|
f49d91b63d | ||
|
|
aafa74b70a | ||
|
|
4f74fe1778 | ||
|
|
a5d384a4d8 | ||
|
|
af9a22fea9 | ||
|
|
baedea504d | ||
|
|
30995ee009 | ||
|
|
3a05d86e3e | ||
|
|
c00a4724f5 | ||
|
|
284e19ef88 | ||
|
|
0ae0df7473 | ||
|
|
fddd82704f | ||
|
|
0691536f28 | ||
|
|
2b9e34ba4a | ||
|
|
30d2b83efe | ||
|
|
db7bccfd89 | ||
|
|
1505b638ce | ||
|
|
7ce727bc8a | ||
|
|
f22793e32e | ||
|
|
4a8d4feea0 | ||
|
|
1d9a85bef9 | ||
|
|
b625d30568 | ||
|
|
d018a441af | ||
|
|
ecb254017d | ||
|
|
41f18ea993 | ||
|
|
91c344fe2c | ||
|
|
cab6ef2102 | ||
|
|
a2ff9515a3 | ||
|
|
62d39ee4f1 | ||
|
|
f1f7dd4a0a | ||
|
|
27ab6b859e |
No files matched your search
@@ -0,0 +1,10 @@
|
||||
# actionlint configuration. Passed explicitly from the ci workflow:
|
||||
# actionlint -config-file .gitea/actionlint.yaml .gitea/workflows/*.yaml
|
||||
#
|
||||
# The self-hosted act_runner registers custom labels that actionlint cannot know
|
||||
# about, so declare them here instead of silencing the whole runner-label check.
|
||||
self-hosted-runner:
|
||||
labels:
|
||||
- arch
|
||||
- homelab
|
||||
- prod
|
||||
+303
-103
@@ -7,6 +7,12 @@ on:
|
||||
pull_request:
|
||||
workflow_dispatch:
|
||||
|
||||
# Every job here is checkout plus local tools. The token needs to read the tree
|
||||
# and nothing else, and saying so keeps a future step that reaches for the API
|
||||
# from quietly holding a token that can write to the repository.
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.ref != 'refs/heads/main' }}
|
||||
@@ -15,8 +21,87 @@ env:
|
||||
REGISTRY: gcr.forust.xyz
|
||||
|
||||
jobs:
|
||||
lint-compose:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
# Structure check for every committed Compose file, active or not.
|
||||
# Interpolation, env-file and bind-mount resolution are all switched off,
|
||||
# because inactive stacks have no .env here and would only fail on their
|
||||
# ${VAR:?} guards. Active stacks get the full check with interpolation in
|
||||
# the deploy workflow, where the real .env files live.
|
||||
- name: Validate Compose files
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
source .gitea/workflows/compose-lint.sh
|
||||
|
||||
mapfile -t safe_flags < <(compose_safe_flags)
|
||||
echo "docker compose config ${safe_flags[*]-}"
|
||||
|
||||
mapfile -t files < <(compose_files)
|
||||
if [ "${#files[@]}" -eq 0 ]; then
|
||||
echo "No Compose files found."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
failed=0
|
||||
for f in "${files[@]}"; do
|
||||
if ! out="$(validate_compose_file "$f" ${safe_flags[@]+"${safe_flags[@]}"} 2>&1)"; then
|
||||
failed=1
|
||||
echo "::error file=${f}::$(printf '%s' "$out" | head -1)"
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$failed" -ne 0 ]; then
|
||||
echo "Compose validation failed."
|
||||
exit 1
|
||||
fi
|
||||
echo "checked ${#files[@]} Compose file(s)"
|
||||
|
||||
lint-actionlint:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Lint Gitea Actions workflows with actionlint
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml
|
||||
|
||||
lint-shellcheck:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Lint shell scripts with ShellCheck
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
mapfile -t scripts < <(
|
||||
git ls-files '*.sh' ':(glob)**/*.bash'
|
||||
)
|
||||
if [ "${#scripts[@]}" -eq 0 ]; then
|
||||
echo "No shell scripts found."
|
||||
exit 0
|
||||
fi
|
||||
shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}"
|
||||
|
||||
lint-prettier:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
@@ -24,6 +109,10 @@ jobs:
|
||||
- name: Check formatting with Prettier
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
|
||||
mapfile -t prettier_files < <(
|
||||
git ls-files \
|
||||
| grep -E '\.(md|json|ya?ml|html|css)$' \
|
||||
@@ -39,17 +128,23 @@ jobs:
|
||||
|
||||
lint-ruff:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Lint Python with Ruff
|
||||
- name: Lint and format-check Python with Ruff
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
ruff check .
|
||||
ruff format --check .
|
||||
|
||||
lint-yaml:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
@@ -57,6 +152,10 @@ jobs:
|
||||
- name: Lint YAML syntax
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
|
||||
mapfile -t yaml_files < <(
|
||||
git ls-files '*.yaml' '*.yml' \
|
||||
':!node_modules/**' \
|
||||
@@ -72,6 +171,7 @@ jobs:
|
||||
|
||||
lint-dockerfiles:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
@@ -79,6 +179,10 @@ jobs:
|
||||
- name: Lint Dockerfiles
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
|
||||
mapfile -t dockerfiles < <(
|
||||
git ls-files ':(glob)**/Dockerfile' ':(glob)**/Dockerfile.*'
|
||||
)
|
||||
@@ -92,13 +196,18 @@ jobs:
|
||||
|
||||
validate:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Validate Kubernetes manifests
|
||||
- name: Validate Kubernetes manifests against JSON schemas
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
|
||||
mapfile -t manifests < <(
|
||||
git ls-files ':(glob)**/k8s/**/*.yaml' ':(glob)**/k8s/**/*.yml' \
|
||||
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
|
||||
@@ -115,10 +224,102 @@ jobs:
|
||||
-summary \
|
||||
"${manifests[@]}"
|
||||
|
||||
# kubeconform has no schemas for CRDs, so every IngressRoute, Certificate,
|
||||
# PrometheusRule, Middleware, ServersTransport and ServiceMonitor is silently
|
||||
# skipped above. The live API server knows the real CRD schemas (and runs the
|
||||
# cert-manager / Traefik admission webhooks), so validate there too.
|
||||
#
|
||||
# Only services marked with a k8s/active marker are checked: server-side
|
||||
# dry-run needs the target namespace to exist, and inactive services are not
|
||||
# deployed. Services being enabled for the first time are still covered by
|
||||
# the JSON-schema pass above.
|
||||
#
|
||||
# Main pushes only. `--dry-run=server` persists nothing, but it does execute
|
||||
# the admission webhooks of the production API server, so anyone able to open
|
||||
# a pull request would be able to run arbitrary manifest content through
|
||||
# cert-manager and Traefik. A pull request has nothing to gain from it either:
|
||||
# only main is ever deployed, and this job runs to completion before the
|
||||
# deploy workflow is allowed to start, so a bad CRD is still caught before
|
||||
# anything reaches the cluster -- just on the push rather than on the PR.
|
||||
- name: Note the server-side check is not running here
|
||||
if: github.event_name == 'pull_request' || github.ref != 'refs/heads/main'
|
||||
shell: bash
|
||||
run: |
|
||||
echo "::notice::Skipping the server-side dry-run. It executes the cert-manager and" \
|
||||
"Traefik admission webhooks against the production API server, so it is limited" \
|
||||
"to pushes to main. CRDs are still schema-checked by kubeconform above, and the" \
|
||||
"server-side pass still runs on main before the deploy."
|
||||
|
||||
- name: Validate active manifests against the live API server
|
||||
if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main'
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
if ! kubectl get --raw='/readyz' --request-timeout=10s >/dev/null 2>&1; then
|
||||
echo "::warning::Cluster unreachable — skipped server-side validation of CRDs (IngressRoute, Certificate, PrometheusRule). Review manifest changes manually."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
mapfile -t k8s_dirs < <(
|
||||
git ls-files '*.yaml' '*.yml' \
|
||||
| grep -E '(^|/)k8s/' \
|
||||
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
|
||||
| sort -u
|
||||
)
|
||||
|
||||
manifests=()
|
||||
kustomize_apps=()
|
||||
for dir in "${k8s_dirs[@]}"; do
|
||||
if [ ! -f "${dir}/active" ]; then
|
||||
echo "skip (no k8s/active): ${dir}"
|
||||
continue
|
||||
fi
|
||||
if [ -f "${dir}/overlays/prod/kustomization.yaml" ]; then
|
||||
kustomize_apps+=("${dir}/overlays/prod")
|
||||
elif [ -f "${dir}/base/kustomization.yaml" ]; then
|
||||
kustomize_apps+=("${dir}/base")
|
||||
else
|
||||
while IFS= read -r f; do
|
||||
[ -n "$f" ] && manifests+=("$f")
|
||||
done < <(
|
||||
git ls-files "${dir}/*.yaml" "${dir}/*.yml" \
|
||||
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
|
||||
)
|
||||
fi
|
||||
done
|
||||
|
||||
echo "server-side dry-run: ${#manifests[@]} manifests, ${#kustomize_apps[@]} kustomize apps"
|
||||
failed=0
|
||||
for m in ${manifests[@]+"${manifests[@]}"}; do
|
||||
if ! out="$(kubectl apply --dry-run=server -f "$m" 2>&1)"; then
|
||||
failed=1
|
||||
echo "::error file=${m}::$(printf '%s' "$out" | head -1)"
|
||||
fi
|
||||
done
|
||||
for k in ${kustomize_apps[@]+"${kustomize_apps[@]}"}; do
|
||||
if ! out="$(kubectl apply -k "$k" --dry-run=server 2>&1)"; then
|
||||
failed=1
|
||||
echo "::error file=${k}::$(printf '%s' "$out" | head -1)"
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$failed" -ne 0 ]; then
|
||||
echo "Server-side validation failed. The API server (or an admission webhook) rejected these manifests."
|
||||
exit 1
|
||||
fi
|
||||
echo "server-side dry-run: all active manifests accepted by the API server"
|
||||
|
||||
build:
|
||||
needs: [lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
|
||||
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev')
|
||||
needs:
|
||||
# The panel's scan-deps/test-backend/test-frontend jobs gated here until
|
||||
# userbot moved to its own repo; upstream's code is upstream's gate now.
|
||||
# The rule is unchanged: publishing and passing the checks are the same
|
||||
# gate, so a commit that fails any of these still cannot move :prod.
|
||||
[lint-actionlint, lint-shellcheck, lint-compose, lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
|
||||
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev') && !startsWith(github.ref_name, 'renovate/')
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 60
|
||||
outputs:
|
||||
services: ${{ steps.services.outputs.services }}
|
||||
steps:
|
||||
@@ -131,12 +332,20 @@ jobs:
|
||||
id: services
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
base="${{ github.event.before }}"
|
||||
if [ -z "$base" ] || [ "$base" = "0000000000000000000000000000000000000000" ]; then
|
||||
base="$(git rev-list --max-parents=0 HEAD)"
|
||||
fi
|
||||
|
||||
mapfile -t changed_files < <(git diff --name-only "$base" "${GITHUB_SHA}")
|
||||
# A failed diff used to leave changed_files empty, which reads exactly
|
||||
# like "nothing to build": the job went green having built nothing and
|
||||
# the tag never moved. The status is checked, not assumed.
|
||||
if ! changed="$(git diff --name-only "$base" "${GITHUB_SHA}")"; then
|
||||
echo "::error::cannot diff ${base}..${GITHUB_SHA}"
|
||||
exit 1
|
||||
fi
|
||||
mapfile -t changed_files <<<"$changed"
|
||||
|
||||
services=()
|
||||
|
||||
@@ -156,15 +365,9 @@ jobs:
|
||||
|
||||
for file in "${changed_files[@]}"; do
|
||||
case "$file" in
|
||||
dtek_notif/*)
|
||||
add_service dtek_notif
|
||||
;;
|
||||
errorpages/*)
|
||||
add_service errorpages
|
||||
;;
|
||||
userbot/*)
|
||||
add_service userbot
|
||||
;;
|
||||
homepages/*)
|
||||
add_service homepages
|
||||
;;
|
||||
@@ -184,55 +387,63 @@ jobs:
|
||||
echo "services=$(paste -sd, /tmp/services.txt)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Log in to registry
|
||||
if: steps.services.outputs.services != ''
|
||||
# The pin step below also writes (manifest PUTs), and it runs on every
|
||||
# main push — including manifest-only ones where services is empty. A
|
||||
# stale persistent login on the old runner used to mask this; a clean
|
||||
# runner pushes anonymously and gets 401.
|
||||
if: steps.services.outputs.services != '' || github.ref_name == 'main'
|
||||
shell: bash
|
||||
# Through env, not by substitution into the script. A secret written
|
||||
# into a run: block is pasted into the shell source before bash parses
|
||||
# it, so a password containing a quote, a backtick or $(...) becomes
|
||||
# code that runs. Masking the value in the log does not prevent that.
|
||||
env:
|
||||
REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }}
|
||||
REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }}
|
||||
run: |
|
||||
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login "${REGISTRY}" \
|
||||
-u "${{ secrets.REGISTRY_USERNAME }}" \
|
||||
set -euo pipefail
|
||||
printf '%s' "$REGISTRY_PASSWORD" | docker login "${REGISTRY}" \
|
||||
-u "$REGISTRY_USERNAME" \
|
||||
--password-stdin
|
||||
|
||||
- name: Build and push changed images
|
||||
if: steps.services.outputs.services != ''
|
||||
shell: bash
|
||||
run: |
|
||||
# This step was the one run: block in the workflow without it, and it
|
||||
# is the one that cannot afford it: a docker push that failed partway
|
||||
# through the loop used to be followed by more pushes, the loop's exit
|
||||
# status came from the last one, and the job went green with half the
|
||||
# images missing from the registry.
|
||||
set -euo pipefail
|
||||
IFS=, read -r -a services <<< "${{ steps.services.outputs.services }}"
|
||||
|
||||
# Tags for this push. The commit-pinned name is the point of this
|
||||
# step: the deploy resolves it in preference to :prod, so a deploy
|
||||
# that sat in the queue behind a later push still gets the build of
|
||||
# the commit CI validated, instead of whatever :prod points at by the
|
||||
# time it runs. See render_pinned in deploy-lib.sh.
|
||||
commit_tag=""
|
||||
if [ "${GITHUB_REF_NAME}" = "main" ]; then
|
||||
commit_tag="sha-${GITHUB_SHA:0:12}"
|
||||
fi
|
||||
|
||||
set_tags() {
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main) tags+=("main" "prod") ;;
|
||||
dev) tags+=("dev") ;;
|
||||
esac
|
||||
if [ -n "$commit_tag" ]; then
|
||||
tags+=("$commit_tag")
|
||||
fi
|
||||
}
|
||||
|
||||
for service in "${services[@]}"; do
|
||||
case "$service" in
|
||||
dtek_notif)
|
||||
image="${REGISTRY}/forust/dtek-notif"
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build \
|
||||
--cache-from "type=registry,ref=${image}:buildcache" \
|
||||
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
|
||||
"${build_args[@]}" dtek_notif
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
;;
|
||||
errorpages)
|
||||
image="${REGISTRY}/forust/error-pages"
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
set_tags
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
@@ -245,43 +456,9 @@ jobs:
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
;;
|
||||
userbot)
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
for target in runtime panel; do
|
||||
case "$target" in
|
||||
runtime)
|
||||
context="userbot"
|
||||
image="${REGISTRY}/forust/userbot"
|
||||
;;
|
||||
panel)
|
||||
context="userbot/panel"
|
||||
image="${REGISTRY}/forust/userbot-panel"
|
||||
;;
|
||||
esac
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build \
|
||||
--cache-from "type=registry,ref=${image}:buildcache" \
|
||||
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
|
||||
"${build_args[@]}" "$context"
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
done
|
||||
;;
|
||||
homepages)
|
||||
for service in forust xdfnx; do
|
||||
case "$service" in
|
||||
for variant in forust xdfnx; do
|
||||
case "$variant" in
|
||||
forust)
|
||||
image="${REGISTRY}/forust/forust-homepage"
|
||||
;;
|
||||
@@ -289,15 +466,7 @@ jobs:
|
||||
image="${REGISTRY}/forust/xdfnx-homepage"
|
||||
;;
|
||||
esac
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
set_tags
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
@@ -305,15 +474,15 @@ jobs:
|
||||
docker build \
|
||||
--cache-from "type=registry,ref=${image}:buildcache" \
|
||||
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
|
||||
"${build_args[@]}" -f "homepages/Dockerfile.${service}" homepages
|
||||
"${build_args[@]}" -f "homepages/Dockerfile.${variant}" homepages
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
done
|
||||
;;
|
||||
edu_master)
|
||||
for service in session-keeper webinar-checker; do
|
||||
case "$service" in
|
||||
for variant in session-keeper webinar-checker; do
|
||||
case "$variant" in
|
||||
session-keeper)
|
||||
context="edu_master/phpsessid-bot"
|
||||
image="${REGISTRY}/forust/session-keeper"
|
||||
@@ -323,15 +492,7 @@ jobs:
|
||||
image="${REGISTRY}/forust/webinar-checker"
|
||||
;;
|
||||
esac
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
set_tags
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
@@ -347,3 +508,42 @@ jobs:
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
# Every image the tree names has to carry the commit-pinned name, not only
|
||||
# the ones this push rebuilt. A push that touches nothing but manifests
|
||||
# builds nothing, and its deploy would then find no commit-pinned tag to
|
||||
# resolve and quietly fall back to the moving :prod - which is the whole
|
||||
# failure the commit-pinned name exists to remove.
|
||||
#
|
||||
# Re-tagging copies the manifest list and transfers no layers, so pinning
|
||||
# six images that already exist costs six registry writes.
|
||||
#
|
||||
# The list is derived from the tree rather than written out here, so an
|
||||
# image added to a manifest is covered without a second place to update.
|
||||
- name: Pin the commit name on the images this push did not rebuild
|
||||
if: github.ref_name == 'main'
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
commit_tag="sha-${GITHUB_SHA:0:12}"
|
||||
mapfile -t repos < <(
|
||||
git grep -hoE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+' -- '*.yaml' '*.yml' \
|
||||
| sort -u
|
||||
)
|
||||
if [ "${#repos[@]}" -eq 0 ]; then
|
||||
echo "No own images referenced by the tree."
|
||||
exit 0
|
||||
fi
|
||||
echo "pinning ${#repos[@]} image(s) to $commit_tag"
|
||||
for repo in "${repos[@]}"; do
|
||||
if docker buildx imagetools inspect "$repo:$commit_tag" >/dev/null 2>&1; then
|
||||
echo " already built by this push: ${repo##*/}"
|
||||
continue
|
||||
fi
|
||||
if ! docker buildx imagetools inspect "$repo:prod" >/dev/null 2>&1; then
|
||||
echo " WARNING: ${repo##*/} has no :prod to pin and no build produced it"
|
||||
continue
|
||||
fi
|
||||
docker buildx imagetools create --tag "$repo:$commit_tag" "$repo:prod"
|
||||
echo " pinned ${repo##*/}"
|
||||
done
|
||||
@@ -0,0 +1,46 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared helpers for validating Compose files. Sourced both by steps in
|
||||
# .gitea/workflows/ci.yaml and by deploy-lib.sh on the workstation.
|
||||
#
|
||||
# Two levels of checking, matching how the repo is structured:
|
||||
#
|
||||
# general every committed Compose file, active or not. Pure structure check:
|
||||
# no ${VAR} interpolation, no .env lookup, no bind-mount path
|
||||
# resolution. Disabled stacks deliberately have no .env in the repo
|
||||
# and no values on the CI runner, so a full `config` run would fail on
|
||||
# their `${VAR:?}` guards for reasons that have nothing to do with the
|
||||
# change under review.
|
||||
#
|
||||
# full active stacks only, with interpolation and env-file resolution, so
|
||||
# required variables and referenced files are actually resolved. Needs
|
||||
# the gitignored .env files, so this only runs in the deploy workflow
|
||||
# on the workstation.
|
||||
#
|
||||
# This file is meant to be sourced, not executed.
|
||||
|
||||
# All committed Compose files, including the ones deploy never starts.
|
||||
compose_files() {
|
||||
git ls-files \
|
||||
'*/compose.yaml' '*/compose.yml' 'compose.yaml' 'compose.yml' \
|
||||
'*/docker-compose.yaml' '*/docker-compose.yml'
|
||||
}
|
||||
|
||||
# Prints the flags that turn `docker compose config` into the general check.
|
||||
# Probed rather than hardcoded so an older Compose without --no-env-resolution
|
||||
# still gets the flags it does support.
|
||||
compose_safe_flags() {
|
||||
local help flag
|
||||
help="$(docker compose config --help 2>/dev/null || true)"
|
||||
for flag in --no-interpolate --no-env-resolution --no-path-resolution; do
|
||||
if printf '%s' "$help" | grep -q -- "$flag"; then
|
||||
printf '%s\n' "$flag"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
# validate_compose_file <file> [extra docker compose config flags...]
|
||||
validate_compose_file() {
|
||||
local file="$1"
|
||||
shift
|
||||
docker compose -f "$file" config --quiet "$@"
|
||||
}
|
||||
+944
-46
File diff suppressed because it is too large.
Load diff
@@ -1,13 +1,26 @@
|
||||
name: deploy
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
# Deploy only what CI already validated. workflow_run is used instead of
|
||||
# workflow_dispatch so a red lint/validate run can never reach the cluster.
|
||||
workflow_run:
|
||||
workflows: [ci]
|
||||
types: [completed]
|
||||
workflow_dispatch:
|
||||
|
||||
# The deploy jobs read the tree, then reach the cluster over SSH with the
|
||||
# deploy key. The Actions token itself is not part of that path, so it gets
|
||||
# read-only contents and no more.
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: deploy-main
|
||||
# Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and
|
||||
# takes the verify job down with it, so a superseded deploy would leave the
|
||||
# cluster half-applied and unchecked — the exact failure the verify job exists
|
||||
# to catch. kubectl apply and docker compose up are both idempotent, so letting
|
||||
# the older run finish and then deploying the newer commit costs little.
|
||||
cancel-in-progress: false
|
||||
|
||||
env:
|
||||
@@ -17,10 +30,26 @@ env:
|
||||
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
|
||||
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
|
||||
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
|
||||
# workflow_run's own GITHUB_SHA points at the branch head, not at the commit the
|
||||
# finished ci run checked. Pin the exact validated commit instead, so a push
|
||||
# landing mid-deploy cannot make the workstation deploy something else. Also
|
||||
# what the verify job checks the snapshot against. Empty for workflow_dispatch,
|
||||
# which falls back to the current origin/main.
|
||||
DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }}
|
||||
|
||||
jobs:
|
||||
preflight:
|
||||
# Autodeploy defaults to OFF: pushes deploy only when the AUTODEPLOY repo
|
||||
# variable is set to 'true' (Settings -> Actions -> Variables). A manual
|
||||
# Run workflow always bypasses the switch: dispatching it is the explicit
|
||||
# intent to deploy.
|
||||
if: >-
|
||||
(vars.AUTODEPLOY == 'true' || github.event_name == 'workflow_dispatch') &&
|
||||
(github.event_name != 'workflow_run' ||
|
||||
(github.event.workflow_run.conclusion == 'success' &&
|
||||
github.event.workflow_run.head_branch == 'main'))
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
@@ -34,6 +63,7 @@ jobs:
|
||||
validate:
|
||||
needs: [preflight]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
@@ -47,6 +77,36 @@ jobs:
|
||||
apply-k8s:
|
||||
needs: [validate]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
# Apply only, no verification, so this is just the work itself: snapshot,
|
||||
# then sequential `helm upgrade --install --wait --rollback-on-failure --timeout 10m`, then the apply loop.
|
||||
# Verification has its own job and its own budget.
|
||||
#
|
||||
# 45 is roughly four times the measured cost of the stage, which is
|
||||
# deliberately not raised on a theory:
|
||||
#
|
||||
# helm, healthy 3 no-op upgrades ~3-5 min
|
||||
# helm, one release bad rollback-on-failure spends its 10m, ~10-15 min
|
||||
# then rolls that one back
|
||||
# apply loop ~40 manifests, 4 of which ~1 min
|
||||
# resolve an image digest
|
||||
# restart_stale_images 7.6s to find 8 workloads, ~0.5 min
|
||||
# 9.8s to resolve their digests
|
||||
#
|
||||
# The helm figure is one release, not three: `set -e` aborts
|
||||
# upgrade_helm_releases on the first failure, so a broken release costs
|
||||
# 10m and the other two are never attempted. Multiplying 10m by three
|
||||
# overstates the worst case by 20 minutes.
|
||||
#
|
||||
# The 45 minutes this was last raised to 45 were still not enough, and the
|
||||
# job logs for those runs no longer exist, so what actually consumed the
|
||||
# budget is not known - the two measurable candidates above account for
|
||||
# ~15 of it. The unbounded `docker manifest inspect` against the registry's
|
||||
# known hang mode is now bounded inside registry_digest (25s timeout, 3
|
||||
# attempts): a dead registry fails each owned image after ~85s instead of
|
||||
# hanging the stage, and a blinking one is retried instead of failing the
|
||||
# whole apply file. Still open: make the stage announce which manifest it
|
||||
# is working on, so a killed run leaves a diagnosable last line.
|
||||
timeout-minutes: 45
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
@@ -60,6 +120,7 @@ jobs:
|
||||
apply-compose:
|
||||
needs: [validate]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
@@ -69,3 +130,68 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh apply-compose
|
||||
|
||||
# Watches the workloads this deploy changed and rolls back the ones that never
|
||||
# became healthy. Runs even when the apply jobs failed, timed out or were
|
||||
# cancelled — that is the whole point of splitting it out. `always()` is what
|
||||
# lets it start after a failed dependency; the needs on apply-compose are a
|
||||
# barrier, so verification begins only once both applies are done.
|
||||
verify-k8s:
|
||||
needs: [apply-k8s, apply-compose]
|
||||
if: >-
|
||||
always() &&
|
||||
needs.apply-k8s.result != 'skipped' &&
|
||||
needs.apply-compose.result != 'skipped'
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
# Not raised, because the arithmetic does not close.
|
||||
#
|
||||
# 32 workloads are under management and the wave width is 8, so the verify
|
||||
# itself is 4 waves of ROLLOUT_TIMEOUT (300s) = 20 minutes worst case, when
|
||||
# every rollout times out rather than converging. That is already 20 of 30.
|
||||
#
|
||||
# The other 10 would have to absorb rollback, and rollback_workloads is a
|
||||
# serial `while read` loop at 300s per failed workload. 10 minutes buys two.
|
||||
# Any larger number is buying a bigger multiple of an unbounded term rather
|
||||
# than covering a known cost: 60 minutes buys eight, and 60 minutes is
|
||||
# therefore not a bound, it is a guess with two digits.
|
||||
#
|
||||
# The number becomes derivable the moment rollback uses the same wave width
|
||||
# as the verify: 32 failures then cost 4 waves = 20 minutes instead of 160,
|
||||
# and 45 covers verify plus rollback at full width. That change is to the
|
||||
# recovery path and is not folded into a timeout edit.
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Verify workloads and roll back on failure
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh verify-k8s
|
||||
|
||||
# Asks the public route of every active service whether it is actually
|
||||
# serving, which the rollout check above structurally cannot: a pod can
|
||||
# converge and still be crash-looping, or be listening on a port no Service
|
||||
# points at, or answer 500.
|
||||
#
|
||||
# `always()` for the same reason verify-k8s has it, and it runs after that job
|
||||
# specifically because a rollback is when a route most needs re-checking. The
|
||||
# needs is a barrier, not a filter: whether verify-k8s passed, failed or was
|
||||
# cancelled, the probes are what say whether the cluster is serving, and
|
||||
# suppressing them on a rollback would hide the one run where the answer
|
||||
# matters most.
|
||||
smoke:
|
||||
needs: [verify-k8s]
|
||||
if: always() && needs.verify-k8s.result != 'skipped'
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Probe the public route of every active service
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh smoke
|
||||
Executable
+254
@@ -0,0 +1,254 @@
|
||||
#!/usr/bin/env bash
|
||||
# Installs the pinned CI tools into "$TOOLS_DIR/bin" and echoes that directory
|
||||
# on stdout, so callers can do:
|
||||
#
|
||||
# export PATH="$(bash .gitea/workflows/install-ci-tools.sh kubeconform shellcheck):$PATH"
|
||||
#
|
||||
# Versions come from tool-versions.env next to this script and are kept fresh by
|
||||
# Renovate. Re-running is cheap: an already-installed tool at the pinned version
|
||||
# is left alone.
|
||||
set -euo pipefail
|
||||
|
||||
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
# shellcheck source=tool-versions.env
|
||||
. "$here/tool-versions.env"
|
||||
|
||||
TOOLS_DIR="${TOOLS_DIR:-${RUNNER_TEMP:-/tmp}/homelab-tools}"
|
||||
BIN_DIR="$TOOLS_DIR/bin"
|
||||
mkdir -p "$BIN_DIR"
|
||||
# The just-installed tools must resolve inside this script too: callers only
|
||||
# prepend BIN_DIR to PATH after the script exits, so a bare `uv` below would
|
||||
# miss the binary install_uv just placed (exit 127 on a clean runner).
|
||||
export PATH="$BIN_DIR:$PATH"
|
||||
|
||||
arch="$(uname -m)"
|
||||
# Upstream projects disagree on arch spelling: kubeconform and actionlint use
|
||||
# Go names (amd64/arm64), shellcheck uses uname names (x86_64/aarch64), node
|
||||
# uses neither (x64/arm64), and hadolint mixes the two in a single release
|
||||
# (x86_64 but arm64).
|
||||
case "$arch" in
|
||||
x86_64 | amd64)
|
||||
goarch=amd64
|
||||
sharch=x86_64
|
||||
nodearch=x64
|
||||
hadolintarch=x86_64
|
||||
;;
|
||||
aarch64 | arm64)
|
||||
goarch=arm64
|
||||
sharch=aarch64
|
||||
nodearch=arm64
|
||||
hadolintarch=arm64
|
||||
;;
|
||||
*)
|
||||
echo "install-ci-tools: unsupported architecture: $arch" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
fetch() {
|
||||
# fetch <url> <dest>
|
||||
if command -v curl >/dev/null 2>&1; then
|
||||
curl -sSLf --retry 3 -o "$2" "$1"
|
||||
elif command -v wget >/dev/null 2>&1; then
|
||||
wget -q -O "$2" "$1"
|
||||
else
|
||||
echo "install-ci-tools: neither curl nor wget is available" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
# resolve <command>
|
||||
# Absolute path to use for invoking a tool: the copy in BIN_DIR when present,
|
||||
# otherwise the name for PATH lookup. Every version check and every in-script
|
||||
# invocation goes through this, so a tool missing from both places reads as
|
||||
# "not installed" instead of dying with 127 under `set -e`.
|
||||
resolve() {
|
||||
if [ -x "$BIN_DIR/$1" ]; then
|
||||
printf '%s' "$BIN_DIR/$1"
|
||||
else
|
||||
printf '%s' "$1"
|
||||
fi
|
||||
}
|
||||
|
||||
# installed_version <command>
|
||||
# Prints the version of an already-installed tool, or nothing. Each tool spells
|
||||
# its version flag differently, hence the case.
|
||||
installed_version() {
|
||||
local bin out
|
||||
bin="$(resolve "$1")"
|
||||
if ! command -v "$bin" >/dev/null 2>&1; then
|
||||
return 0
|
||||
fi
|
||||
case "$1" in
|
||||
kubeconform) out="$("$bin" -v 2>/dev/null | head -1 || true)" ;;
|
||||
*) out="$("$bin" --version 2>/dev/null | head -1 || true)" ;;
|
||||
esac
|
||||
printf '%s' "$out"
|
||||
}
|
||||
|
||||
# at_version <command> <expected>
|
||||
at_version() {
|
||||
case "$(installed_version "$1")" in
|
||||
*"$2"*) return 0 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
install_kubeconform() {
|
||||
if at_version kubeconform "v${KUBECONFORM_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
local tmp
|
||||
tmp="$(mktemp -d)"
|
||||
fetch "https://github.com/yannh/kubeconform/releases/download/v${KUBECONFORM_VERSION}/kubeconform-linux-${goarch}.tar.gz" \
|
||||
"$tmp/kubeconform.tar.gz"
|
||||
tar -xzf "$tmp/kubeconform.tar.gz" -C "$tmp" kubeconform
|
||||
install -m 0755 "$tmp/kubeconform" "$BIN_DIR/kubeconform"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
install_shellcheck() {
|
||||
if at_version shellcheck "${SHELLCHECK_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
local tmp
|
||||
tmp="$(mktemp -d)"
|
||||
fetch "https://github.com/koalaman/shellcheck/releases/download/v${SHELLCHECK_VERSION}/shellcheck-v${SHELLCHECK_VERSION}.linux.${sharch}.tar.xz" \
|
||||
"$tmp/shellcheck.tar.xz"
|
||||
tar -xJf "$tmp/shellcheck.tar.xz" -C "$tmp" --strip-components=1 "shellcheck-v${SHELLCHECK_VERSION}/shellcheck"
|
||||
install -m 0755 "$tmp/shellcheck" "$BIN_DIR/shellcheck"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
install_uv() {
|
||||
if at_version uv "${UV_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
local tmp
|
||||
tmp="$(mktemp -d)"
|
||||
# uv release tags carry no leading v, unlike every other tool installed here.
|
||||
fetch "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-${sharch}-unknown-linux-gnu.tar.gz" \
|
||||
"$tmp/uv.tar.gz"
|
||||
tar -xzf "$tmp/uv.tar.gz" -C "$tmp" --strip-components=1 "uv-${sharch}-unknown-linux-gnu/uv"
|
||||
install -m 0755 "$tmp/uv" "$BIN_DIR/uv"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
install_hadolint() {
|
||||
if at_version hadolint "${HADOLINT_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
# A bare binary, no archive: hadolint ships one file per platform.
|
||||
fetch "https://github.com/hadolint/hadolint/releases/download/v${HADOLINT_VERSION}/hadolint-linux-${hadolintarch}" \
|
||||
"$BIN_DIR/hadolint"
|
||||
chmod 0755 "$BIN_DIR/hadolint"
|
||||
}
|
||||
|
||||
# ruff and yamllint both come from PyPI as wheels, which uv unpacks for us.
|
||||
install_uv_tool() {
|
||||
# <package> <pinned version>
|
||||
if at_version "$1" "$2"; then
|
||||
return 0
|
||||
fi
|
||||
install_uv
|
||||
UV_TOOL_BIN_DIR="$BIN_DIR" "$BIN_DIR/uv" tool install --force "$1==$2" >/dev/null
|
||||
}
|
||||
|
||||
install_ruff() {
|
||||
install_uv_tool ruff "${RUFF_VERSION}"
|
||||
}
|
||||
|
||||
install_yamllint() {
|
||||
install_uv_tool yamllint "${YAMLLINT_VERSION}"
|
||||
}
|
||||
|
||||
install_pip_audit() {
|
||||
install_uv_tool pip-audit "${PIP_AUDIT_VERSION}"
|
||||
}
|
||||
|
||||
install_prettier() {
|
||||
if at_version prettier "${PRETTIER_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
# Not a standalone binary: prettier's entry point requires ../package.json
|
||||
# relative to its own real path, so the package directory has to survive
|
||||
# next to it. Hence a versioned directory plus a relative symlink, rather
|
||||
# than copying the one file out as the other installers do.
|
||||
local dir="$BIN_DIR/prettier-${PRETTIER_VERSION}"
|
||||
if [ ! -f "$dir/package/package.json" ]; then
|
||||
rm -rf "$dir"
|
||||
mkdir -p "$dir"
|
||||
fetch "https://registry.npmjs.org/prettier/-/prettier-${PRETTIER_VERSION}.tgz" "$dir/prettier.tgz"
|
||||
tar -xzf "$dir/prettier.tgz" -C "$dir"
|
||||
rm -f "$dir/prettier.tgz"
|
||||
# npm strips the exec bit from bin/ on the way into the tarball.
|
||||
chmod 0755 "$dir/package/bin/prettier.cjs"
|
||||
fi
|
||||
# Relative, so the whole tree stays valid if TOOLS_DIR is relocated.
|
||||
ln -sfn "prettier-${PRETTIER_VERSION}/package/bin/prettier.cjs" "$BIN_DIR/prettier"
|
||||
}
|
||||
|
||||
install_node() {
|
||||
# npm gets checked by running it, not by looking it up: what matters is that
|
||||
# it answers, so a stub, a half-removed Arch package or a name that resolves
|
||||
# to something broken all have to read as "not installed". The runner's npm
|
||||
# is a symlink into /usr/lib/node_modules/npm, which is exactly the kind of
|
||||
# thing that disappears between runs.
|
||||
if at_version node "v${NODE_VERSION}" && [ -n "$(installed_version npm)" ]; then
|
||||
return 0
|
||||
fi
|
||||
# Same shape as prettier above: the tarball's bin/npm and bin/npx are links
|
||||
# into lib/node_modules, so the whole tree has to survive next to them.
|
||||
local dir="$BIN_DIR/node-${NODE_VERSION}"
|
||||
if [ ! -x "$dir/bin/node" ]; then
|
||||
rm -rf "$dir"
|
||||
mkdir -p "$dir"
|
||||
fetch "https://nodejs.org/dist/v${NODE_VERSION}/node-v${NODE_VERSION}-linux-${nodearch}.tar.xz" \
|
||||
"$dir/node.tar.xz"
|
||||
tar -xJf "$dir/node.tar.xz" -C "$dir" --strip-components=1 "node-v${NODE_VERSION}-linux-${nodearch}"
|
||||
rm -f "$dir/node.tar.xz"
|
||||
fi
|
||||
# Relative, so the whole tree stays valid if TOOLS_DIR is relocated.
|
||||
for bin in node npm npx; do
|
||||
ln -sfn "node-${NODE_VERSION}/bin/${bin}" "$BIN_DIR/${bin}"
|
||||
done
|
||||
}
|
||||
|
||||
install_actionlint() {
|
||||
if at_version actionlint "${ACTIONLINT_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
local tmp
|
||||
tmp="$(mktemp -d)"
|
||||
fetch "https://github.com/rhysd/actionlint/releases/download/v${ACTIONLINT_VERSION}/actionlint_${ACTIONLINT_VERSION}_linux_${goarch}.tar.gz" \
|
||||
"$tmp/actionlint.tar.gz"
|
||||
tar -xzf "$tmp/actionlint.tar.gz" -C "$tmp" actionlint
|
||||
install -m 0755 "$tmp/actionlint" "$BIN_DIR/actionlint"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
wanted=("$@")
|
||||
if [ "${#wanted[@]}" -eq 0 ]; then
|
||||
wanted=(kubeconform shellcheck actionlint prettier ruff yamllint hadolint)
|
||||
fi
|
||||
|
||||
for tool in "${wanted[@]}"; do
|
||||
case "$tool" in
|
||||
kubeconform) install_kubeconform ;;
|
||||
shellcheck) install_shellcheck ;;
|
||||
actionlint) install_actionlint ;;
|
||||
prettier) install_prettier ;;
|
||||
ruff) install_ruff ;;
|
||||
yamllint) install_yamllint ;;
|
||||
pip-audit) install_pip_audit ;;
|
||||
hadolint) install_hadolint ;;
|
||||
node) install_node ;;
|
||||
uv) install_uv ;;
|
||||
*)
|
||||
echo "install-ci-tools: unknown tool: $tool" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
printf '%s\n' "$BIN_DIR"
|
||||
@@ -7,33 +7,58 @@ on:
|
||||
- main
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
validate-renovate:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Validate Renovate Compose draft
|
||||
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag,
|
||||
# so the same version that runs in the cluster is the one validated here.
|
||||
- name: Resolve the deployed Renovate image
|
||||
id: image
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
trap 'rm -f renovate/.env' EXIT
|
||||
printf '%s\n' \
|
||||
'RENOVATE_ENDPOINT=https://gitea.example/api/v1' \
|
||||
'RENOVATE_TOKEN=test-token' \
|
||||
'RENOVATE_REPOSITORIES=forust/homelab' \
|
||||
> renovate/.env
|
||||
docker compose -f renovate/renovate-compose.yaml config --quiet
|
||||
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
|
||||
renovate/k8s/cronjob.yaml | head -1)"
|
||||
if [ -z "$image" ]; then
|
||||
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
|
||||
exit 1
|
||||
fi
|
||||
echo "using $image"
|
||||
echo "image=$image" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Validate Kubernetes manifests
|
||||
- name: Validate Renovate repository config
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
ghcr.io/yannh/kubeconform:latest \
|
||||
-v "$PWD/renovate:/opt/renovate:ro" \
|
||||
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
|
||||
"${{ steps.image.outputs.image }}" \
|
||||
renovate-config-validator /opt/renovate/renovate.json
|
||||
|
||||
# The CronJob cannot read the repository, so renovate/k8s/configmap.yaml
|
||||
# carries an inlined copy of the config. Fail if it no longer matches.
|
||||
- name: Check the generated Renovate ConfigMap
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/sync-renovate-configmap.sh --check
|
||||
|
||||
- name: Validate Renovate Kubernetes manifests
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
kubeconform \
|
||||
-strict \
|
||||
-ignore-missing-schemas \
|
||||
-summary \
|
||||
@@ -41,12 +66,11 @@ jobs:
|
||||
renovate/k8s/configmap.yaml \
|
||||
renovate/k8s/cronjob.yaml
|
||||
|
||||
- name: Validate Renovate repository config
|
||||
- name: Validate Renovate Compose file
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
renovate/renovate:44.103.0 \
|
||||
renovate-config-validator renovate.json
|
||||
source .gitea/workflows/compose-lint.sh
|
||||
mapfile -t safe_flags < <(compose_safe_flags)
|
||||
validate_compose_file renovate/renovate-compose.yaml \
|
||||
${safe_flags[@]+"${safe_flags[@]}"}
|
||||
@@ -21,6 +21,11 @@ on:
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
# Renovate writes through its own bot PAT, passed in as RENOVATE_TOKEN, so the
|
||||
# Actions token is only ever used to read the checkout.
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: renovate-run
|
||||
cancel-in-progress: false
|
||||
@@ -28,18 +33,36 @@ concurrency:
|
||||
jobs:
|
||||
run-renovate:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag.
|
||||
# Reading it here means this workflow validates and runs the exact version
|
||||
# that is deployed, instead of a copy that silently goes stale.
|
||||
- name: Resolve the deployed Renovate image
|
||||
id: image
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
|
||||
renovate/k8s/cronjob.yaml | head -1)"
|
||||
if [ -z "$image" ]; then
|
||||
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
|
||||
exit 1
|
||||
fi
|
||||
echo "using $image"
|
||||
echo "image=$image" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Validate Renovate config
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker run --rm \
|
||||
-v "$PWD/renovate/config.js:/opt/renovate/config.js:ro" \
|
||||
-e RENOVATE_CONFIG_FILE=/opt/renovate/config.js \
|
||||
renovate/renovate:44.103.0 \
|
||||
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
|
||||
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
|
||||
"${{ steps.image.outputs.image }}" \
|
||||
renovate-config-validator
|
||||
|
||||
- name: Run Renovate
|
||||
@@ -56,14 +79,14 @@ jobs:
|
||||
: "${RENOVATE_TOKEN:?missing RENOVATE_TOKEN secret — add a renovate-bot PAT in repo/org Actions secrets}"
|
||||
|
||||
docker run --rm \
|
||||
-v "$PWD/renovate/config.js:/opt/renovate/config.js:ro" \
|
||||
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
|
||||
-e RENOVATE_PLATFORM=gitea \
|
||||
-e RENOVATE_ENDPOINT=https://gitea.forust.xyz/api/v1 \
|
||||
-e RENOVATE_ENDPOINT=https://git.forust.xyz/api/v1 \
|
||||
-e RENOVATE_TOKEN="$RENOVATE_TOKEN" \
|
||||
-e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \
|
||||
-e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \
|
||||
-e RENOVATE_DRY_RUN="${RENOVATE_DRY_RUN:-}" \
|
||||
-e RENOVATE_CONFIG_FILE=/opt/renovate/config.js \
|
||||
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
|
||||
-e RENOVATE_BASE_DIR=/tmp/renovate \
|
||||
-e LOG_LEVEL="${LOG_LEVEL:-info}" \
|
||||
renovate/renovate:44.103.0
|
||||
"${{ steps.image.outputs.image }}"
|
||||
@@ -11,15 +11,61 @@ deploy_port="${DEPLOY_PORT:-22}"
|
||||
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
|
||||
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)"
|
||||
|
||||
ssh_key="$RUNNER_TEMP/deploy_key"
|
||||
mkdir -p "$RUNNER_TEMP"
|
||||
# The private key is written to a per-run directory that is removed on exit, so a
|
||||
# failed or cancelled job cannot leave deploy credentials in the runner's temp
|
||||
# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel.
|
||||
key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")"
|
||||
trap 'rm -rf "$key_dir"' EXIT INT TERM
|
||||
|
||||
ssh_key="$key_dir/deploy_key"
|
||||
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
|
||||
chmod 600 "$ssh_key"
|
||||
|
||||
ssh -i "$ssh_key" -p "$deploy_port" \
|
||||
-o BatchMode=yes -o StrictHostKeyChecking=accept-new \
|
||||
"${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} STAGE=$1 bash -se" <<'EOF'
|
||||
# A connection that died silently used to hang until the job timeout, and the
|
||||
# stage was never re-run: one flaky TCP session cost a whole 45-minute apply.
|
||||
# ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds
|
||||
# setup. Only exit 255 - ssh's own transport failures - is retried. A stage that
|
||||
# fails on its own merits exits with the remote's status, so a real failure
|
||||
# still surfaces its own log instead of burning three attempts. The stages are
|
||||
# declarative applies, so re-running one that had already committed is harmless.
|
||||
ssh_opts=(
|
||||
-i "$ssh_key" -p "$deploy_port"
|
||||
-o BatchMode=yes -o StrictHostKeyChecking=accept-new
|
||||
-o ConnectTimeout=15
|
||||
-o ServerAliveInterval=15 -o ServerAliveCountMax=4
|
||||
)
|
||||
|
||||
rc=0
|
||||
# apply-k8s and apply-compose are separate workflow jobs so the graph stays
|
||||
# intact for the verify job, but on a single node they must not run at once:
|
||||
# host docker churn on top of cluster churn is what melts the node (load 40+,
|
||||
# netbird/ssh die, helm is left pending-*). Serialize them on the workstation
|
||||
# with a shared lock; whoever arrives second waits.
|
||||
remote_cmd=(bash -se)
|
||||
case "$1" in
|
||||
apply-k8s | apply-compose)
|
||||
remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se)
|
||||
;;
|
||||
esac
|
||||
for attempt in 1 2 3; do
|
||||
if [ "$attempt" -gt 1 ]; then
|
||||
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
|
||||
sleep $((attempt * 5))
|
||||
fi
|
||||
rc=0
|
||||
# shellcheck disable=SC2029 # remote_cmd/ssh_opts expand on the client on purpose: they select the local ssh invocation, only the heredoc runs remotely.
|
||||
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
|
||||
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
|
||||
"STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$?
|
||||
source "$REPO/.gitea/workflows/deploy-lib.sh"
|
||||
run_stage "$STAGE"
|
||||
EOF
|
||||
[ "$rc" -eq 0 ] && break
|
||||
[ "$rc" -ne 255 ] && break
|
||||
done
|
||||
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo ":: error::stage $1 failed over ssh (exit $rc)"
|
||||
fi
|
||||
exit "$rc"
|
||||
Executable
+55
@@ -0,0 +1,55 @@
|
||||
#!/usr/bin/env bash
|
||||
# Regenerates renovate/k8s/configmap.yaml from renovate/renovate.json.
|
||||
#
|
||||
# renovate/renovate.json is the single source of truth: the CronJob, the Compose
|
||||
# file and the renovate-run workflow all mount that exact file. A ConfigMap cannot
|
||||
# read a file from the repository, so the same bytes are inlined here as a literal
|
||||
# block. This script keeps the copy honest:
|
||||
#
|
||||
# .gitea/workflows/sync-renovate-configmap.sh # rewrite in place
|
||||
# .gitea/workflows/sync-renovate-configmap.sh --check # fail if out of date
|
||||
#
|
||||
# renovate-ci runs the --check form on every PR and push, so a config change that
|
||||
# forgets to regenerate the ConfigMap cannot be merged.
|
||||
set -euo pipefail
|
||||
|
||||
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
repo="$(git -C "$here" rev-parse --show-toplevel)"
|
||||
|
||||
src="$repo/renovate/renovate.json"
|
||||
dst="$repo/renovate/k8s/configmap.yaml"
|
||||
[ -f "$src" ] || {
|
||||
echo "missing $src" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
render() {
|
||||
cat <<'HEADER'
|
||||
# GENERATED FILE - do not edit by hand.
|
||||
# Source: renovate/renovate.json
|
||||
# Regenerate: .gitea/workflows/sync-renovate-configmap.sh
|
||||
# Verify: .gitea/workflows/sync-renovate-configmap.sh --check
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: renovate-config
|
||||
namespace: renovate
|
||||
data:
|
||||
renovate.json: |
|
||||
HEADER
|
||||
sed 's/^/ /' "$src"
|
||||
}
|
||||
|
||||
if [ "${1:-}" = "--check" ]; then
|
||||
if ! diff -u "$dst" <(render) >/dev/null 2>&1; then
|
||||
echo "ERROR: $dst is out of sync with renovate/renovate.json"
|
||||
echo "Run: .gitea/workflows/sync-renovate-configmap.sh"
|
||||
diff -u "$dst" <(render) || true
|
||||
exit 1
|
||||
fi
|
||||
echo "renovate/k8s/configmap.yaml is in sync with renovate/renovate.json"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
render >"$dst"
|
||||
echo "wrote $dst"
|
||||
@@ -0,0 +1,33 @@
|
||||
# Pinned versions of the CI tools installed by install-ci-tools.sh.
|
||||
# Renovate keeps these up to date (see customManagers in renovate/renovate.json).
|
||||
#
|
||||
# Every version here except NODE_VERSION matches what was already installed on
|
||||
# the runner, so pinning them changes what CI does not at all. It changes what
|
||||
# CI does when the runner is rebuilt with something else: today
|
||||
# install-ci-tools.sh finds the pinned version already on PATH and installs
|
||||
# nothing, and a runner that drifts gets the pinned one installed over it.
|
||||
#
|
||||
# The renovate image version is NOT pinned here: renovate/k8s/cronjob.yaml is the
|
||||
# single source of truth and the workflows read the tag from it, so there is
|
||||
# nothing to drift.
|
||||
ACTIONLINT_VERSION="1.7.7"
|
||||
SHELLCHECK_VERSION="0.11.0"
|
||||
KUBECONFORM_VERSION="0.8.0"
|
||||
PRETTIER_VERSION="3.8.1"
|
||||
RUFF_VERSION="0.16.8"
|
||||
YAMLLINT_VERSION="1.38.0"
|
||||
HADOLINT_VERSION="2.14.0"
|
||||
# pip-audit reads the advisory database over the network, so a floating version
|
||||
# would make the same commit report different things on different days. Pin it
|
||||
# like the rest: the advisories themselves are the moving part, not the tool.
|
||||
PIP_AUDIT_VERSION="2.10.1"
|
||||
# uv builds the throwaway venv the pytest job runs in, and unpacks the PyPI
|
||||
# wheels for ruff, yamllint and pip-audit.
|
||||
UV_VERSION="0.12.17"
|
||||
# node runs `npm ci` for the frontend tests and the npm audit, and it is the one
|
||||
# pin here that does NOT come from the runner: the runner's system node is a
|
||||
# rolling Arch package (it was node 26 with no npm at all when this was pinned),
|
||||
# and the panel image is node:22-alpine. Pinned to the image's major on purpose,
|
||||
# so the tree that gets tested is the tree that gets built. Renovate keeps this
|
||||
# in step with the Dockerfile's node: tag via the "node runtime" group.
|
||||
NODE_VERSION="22.23.3"
|
||||
@@ -96,6 +96,8 @@ replacements.txt
|
||||
# Temp files
|
||||
edu_master/temp/
|
||||
temp/*
|
||||
# Local-only tooling scratch space (pinned CI tools, verification scripts)
|
||||
tmp/
|
||||
|
||||
# Environment
|
||||
.env
|
||||
|
||||
@@ -58,6 +58,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: adguard
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -73,7 +75,7 @@ spec:
|
||||
memory: "1.5Gi"
|
||||
cpu: "300m"
|
||||
requests:
|
||||
memory: "500Mi"
|
||||
memory: "512Mi"
|
||||
cpu: "50m"
|
||||
ports:
|
||||
- containerPort: 3000
|
||||
@@ -82,6 +84,13 @@ spec:
|
||||
name: dns
|
||||
- containerPort: 853
|
||||
name: dot
|
||||
readinessProbe:
|
||||
tcpSocket:
|
||||
port: dns
|
||||
initialDelaySeconds: 5
|
||||
periodSeconds: 5
|
||||
successThreshold: 1
|
||||
failureThreshold: 3
|
||||
volumeMounts:
|
||||
- name: adguard-data
|
||||
mountPath: /opt/adguardhome/work
|
||||
|
||||
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`dns.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: adguard-service
|
||||
port: 3000
|
||||
|
||||
@@ -34,6 +34,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: authentik-server
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -52,8 +54,8 @@ spec:
|
||||
- containerPort: 9000
|
||||
resources:
|
||||
requests:
|
||||
memory: "700Mi"
|
||||
cpu: "300m"
|
||||
memory: "768Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
memory: "1.5Gi"
|
||||
cpu: "1000m"
|
||||
@@ -68,6 +70,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: authentik-worker
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -86,8 +90,8 @@ spec:
|
||||
name: authentik-secrets
|
||||
resources:
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
cpu: "300m"
|
||||
memory: "320Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
memory: "768Mi"
|
||||
cpu: "700m"
|
||||
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`auth.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: authentik-server-service
|
||||
port: 9000
|
||||
|
||||
@@ -9,6 +9,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: cfddns
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -22,10 +24,10 @@ spec:
|
||||
imagePullPolicy: Always
|
||||
resources:
|
||||
requests:
|
||||
memory: "20Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "30m"
|
||||
limits:
|
||||
memory: "64Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "50m"
|
||||
envFrom:
|
||||
- secretRef:
|
||||
|
||||
@@ -24,6 +24,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: checkmk
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
|
||||
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`cmk.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: checkmk-service
|
||||
port: 5000
|
||||
|
||||
@@ -9,6 +9,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: cloudflared
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -30,8 +32,8 @@ spec:
|
||||
key: TUNNEL_TOKEN
|
||||
resources:
|
||||
requests:
|
||||
memory: "32Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "30m"
|
||||
limits:
|
||||
memory: "128Mi"
|
||||
memory: "256Mi"
|
||||
cpu: "200m"
|
||||
@@ -1,7 +1,7 @@
|
||||
services:
|
||||
convertx:
|
||||
container_name: convertx
|
||||
image: ghcr.io/c4illin/convertx:v0.18.0
|
||||
image: ghcr.io/c4illin/convertx:v0.19.0
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "9992:3000"
|
||||
@@ -54,7 +54,7 @@ services:
|
||||
- "traefik.http.routers.bentopdf.tls.certresolver=letsencrypt"
|
||||
- "traefik.http.routers.bentopdf.tls=true"
|
||||
# Local router
|
||||
- "traefik.http.routers.bentopdf-local.rule=Host(`pdf.wokstation.internal`)"
|
||||
- "traefik.http.routers.bentopdf-local.rule=Host(`pdf.workstation.internal`)"
|
||||
- "traefik.http.routers.bentopdf-local.entrypoints=websecure"
|
||||
- "traefik.http.routers.bentopdf-local.tls=true"
|
||||
# Dev router
|
||||
|
||||
@@ -31,12 +31,13 @@ spec:
|
||||
name: bentopdf
|
||||
ports:
|
||||
- containerPort: 8080
|
||||
# p95 4M, max 11M over 7 days. Was 50Mi/700Mi.
|
||||
resources:
|
||||
requests:
|
||||
memory: "50Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "50m"
|
||||
ephemeral-storage: "100Mi"
|
||||
limits:
|
||||
memory: "700Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "700m"
|
||||
ephemeral-storage: "5Gi"
|
||||
@@ -20,13 +20,15 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: convertx
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: convertx
|
||||
spec:
|
||||
containers:
|
||||
- image: ghcr.io/c4illin/convertx:v0.18.0
|
||||
- image: ghcr.io/c4illin/convertx:v0.19.0
|
||||
name: convertx
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
@@ -38,13 +40,14 @@ spec:
|
||||
volumeMounts:
|
||||
- mountPath: /data
|
||||
name: data
|
||||
# p95 85M, max 136M over 7 days, spikes while converting. Was 250Mi/1.5Gi.
|
||||
resources:
|
||||
requests:
|
||||
memory: "250Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
cpu: "1500m"
|
||||
memory: "1.5Gi"
|
||||
memory: "512Mi"
|
||||
volumes:
|
||||
- name: data
|
||||
persistentVolumeClaim:
|
||||
|
||||
@@ -1,14 +0,0 @@
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: Middleware
|
||||
metadata:
|
||||
name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
spec:
|
||||
plugin:
|
||||
crowdsec-bouncer:
|
||||
enabled: true
|
||||
LogLevel: INFO
|
||||
CrowdsecMode: live
|
||||
CrowdsecLapiScheme: http
|
||||
CrowdsecLapiHost: crowdsec-service.crowdsec.svc.cluster.local:8080
|
||||
CrowdsecLapiKeyFile: "/etc/traefik/secrets/traefik-api-key"
|
||||
@@ -16,6 +16,12 @@ agent:
|
||||
value: crowdsecurity/traefik crowdsecurity/base-http-scenarios
|
||||
- name: DISABLE_COLLECTIONS
|
||||
value: crowdsecurity/sshd
|
||||
# Bans on 401/403 bursts hurt more than they protect: with L3 enforcement
|
||||
# a false positive cuts the IP off everything (SSH included), and past
|
||||
# incidents show legit automation (deploy runner, mesh peers, registry
|
||||
# pulls) tripping this probe. Probing/XSS/SQLi/CVE scenarios stay.
|
||||
- name: DISABLE_SCENARIOS
|
||||
value: crowdsecurity/http-generic-bf
|
||||
metrics:
|
||||
enabled: true
|
||||
serviceMonitor:
|
||||
@@ -58,9 +64,44 @@ config:
|
||||
reason: "Mobile IP whitelist"
|
||||
cidr:
|
||||
- "84.245.64.0/18"
|
||||
# CrowdSec's own guidance: CIDR allowlisting belongs at the parser stage.
|
||||
# A parser whitelist discards the event before it reaches a bucket, so
|
||||
# these addresses never produce an overflow and never become a decision.
|
||||
# A postoverflow whitelist is checked only *after* the ban exists, and
|
||||
# the bouncer answers 403 for as long as it does - which is a window we
|
||||
# do not want the deploy sitting in.
|
||||
local-network.yaml: |
|
||||
name: forust/local-network
|
||||
description: "Whitelist loopback, private and VPN networks"
|
||||
whitelist:
|
||||
reason: "Local network"
|
||||
cidr:
|
||||
- "127.0.0.0/8"
|
||||
- "10.0.0.0/8"
|
||||
- "172.16.0.0/12"
|
||||
- "192.168.0.0/16"
|
||||
# CGNAT range (RFC 6598). The workstation and the k0s node live
|
||||
# here on WireGuard, and 100.64.0.0/10 is not covered by the
|
||||
# RFC 1918 blocks above.
|
||||
- "100.64.0.0/10"
|
||||
- "169.254.0.0/16"
|
||||
- "fc00::/7"
|
||||
- "fe80::/10"
|
||||
vps-whitelist.yaml: |
|
||||
name: forust/vps-whitelist
|
||||
description: "Whitelist static VPS"
|
||||
whitelist:
|
||||
reason: "VPS"
|
||||
ip:
|
||||
- "193.181.211.79"
|
||||
|
||||
postoverflows:
|
||||
s01-whitelist:
|
||||
# The one whitelist that has to stay here: resolving a hostname is a
|
||||
# network call, and the docs put expensive lookups in postoverflows on
|
||||
# purpose - it runs only when a bucket actually overflows.
|
||||
# ddns.forust.xyz is the public home address, not a private one, so
|
||||
# forust/local-network does not cover it.
|
||||
home-dynamic-ip.yaml: |
|
||||
name: forust/home-dynamic-ip
|
||||
description: "Whitelist home dynamic IP"
|
||||
@@ -69,6 +110,59 @@ config:
|
||||
expression:
|
||||
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
|
||||
|
||||
# LAPI-only main config override, merged over config.yaml. NOTE: the
|
||||
# chart's own default for this key is REPLACED, not merged, so its
|
||||
# auto_registration block is repeated verbatim below - drop it and the
|
||||
# agent can no longer register itself.
|
||||
config.yaml.local: |
|
||||
api:
|
||||
server:
|
||||
auto_registration: # Activate if not using TLS for authentication
|
||||
enabled: true
|
||||
token: "${REGISTRATION_TOKEN}" # /!\ Do not modify this variable (auto-generated and handled by the chart)
|
||||
allowed_ranges: # /!\ Make sure to adapt to the pod IP ranges used by your cluster
|
||||
- "127.0.0.1/32"
|
||||
- "192.168.0.0/16"
|
||||
- "10.0.0.0/8"
|
||||
- "172.16.0.0/12"
|
||||
# This homelab has no egress to console.crowdsec.cloud: DNS does
|
||||
# not resolve. The LAPI kept trying anyway ("Signal push: N
|
||||
# signals to push", "capi metrics: sending" every 10s) and each
|
||||
# attempt sat on a resolver timeout WHILE HOLDING A WRITE
|
||||
# TRANSACTION, which is what kept stalling per-request decision
|
||||
# lookups even with WAL enabled. Nothing to share and nothing to
|
||||
# pull - turn the Central API off instead of letting it block the
|
||||
# only database writer we have.
|
||||
online_client:
|
||||
sharing: false
|
||||
pull:
|
||||
community: false
|
||||
blocklists: false
|
||||
disable_usage_metrics_export: true
|
||||
db_config:
|
||||
# SQLite without WAL serialises every reader behind the writer's
|
||||
# rollback journal, and the LAPI writes constantly: the agent pushes
|
||||
# Traefik alerts read from Loki, the metrics collector counts
|
||||
# decisions, the bouncer touches "last pull" on every request.
|
||||
# Symptom: decision lookups taking 10-30s (and a second connection
|
||||
# that could not even open the database) while the LAPI sat at 28m
|
||||
# CPU - the process was blocked in fsync, not computing. Every
|
||||
# bouncer-protected request then blew through the plugin timeout and
|
||||
# fail-closed with 403, on every site at once.
|
||||
# The PVC is local-path-retain (hostPath), not a network share, so
|
||||
# WAL is safe here; the crowdsec docs recommend it for exactly this
|
||||
# ("allowing more concurrency in SQLite that will improve
|
||||
# performances in most scenarios").
|
||||
use_wal: true
|
||||
# Keeps the alert table bounded. At the 5000/7d default the file
|
||||
# reached 54MB in 15 days off the Traefik access log alone, and the
|
||||
# metrics collector counts decisions on a timer; a smaller working
|
||||
# set means fewer full scans. Crowdsec only prunes - SQLite never
|
||||
# shrinks the file, so the size stays until a manual VACUUM.
|
||||
flush:
|
||||
max_items: 1000
|
||||
max_age: 24h
|
||||
|
||||
lapi:
|
||||
env:
|
||||
- name: COLLECTIONS
|
||||
@@ -90,13 +184,20 @@ lapi:
|
||||
enabled: true
|
||||
size: 1Gi
|
||||
storageClassName: local-path-retain
|
||||
# LAPI answers a blocking /v1/decisions lookup for EVERY bouncer-protected
|
||||
# request (whole Traefik front door), so it is the hot path of the proxy.
|
||||
# At 400m/500Mi it went CPU-throttled and idle lookups measured 1.3-7.4s,
|
||||
# which pushed requests into the bouncer's fail-closed 403.
|
||||
# Single replica on purpose: LAPI is stateful (BoltDB on the `data` PVC,
|
||||
# credentials on the `config` PVC) - two replicas sharing those RWO
|
||||
# volumes would corrupt the decision store. Scale up CPU, not replicas.
|
||||
resources:
|
||||
limits:
|
||||
cpu: 400m
|
||||
memory: 500Mi
|
||||
cpu: 1500m
|
||||
memory: 1Gi
|
||||
requests:
|
||||
cpu: 50m
|
||||
memory: 150Mi
|
||||
cpu: 250m
|
||||
memory: 500Mi
|
||||
service:
|
||||
type: ClusterIP
|
||||
storeLAPICscliCredentialsInSecret: true
|
||||
@@ -32,6 +32,13 @@
|
||||
# on their own - same name + same password);
|
||||
# 4. prune bouncer entries idle for 30d.
|
||||
#
|
||||
# It used to also delete LePresidente/http-generic-403-bf decisions hourly.
|
||||
# That was a workaround for the bouncer failing closed on a slow LAPI and
|
||||
# 403-ing the deploy runner into a 4h ban. The bouncer now polls decisions
|
||||
# into a cache and never blocks on an unreachable LAPI, so it cannot
|
||||
# manufacture those 403s any more, and the scenario only fires against real
|
||||
# scanners - deleting their decisions hourly was undoing a working ban.
|
||||
#
|
||||
# Manual apply (crowdsec/k8s is NOT managed by deploy.yaml):
|
||||
# kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml
|
||||
# Force a run:
|
||||
|
||||
@@ -18,8 +18,6 @@ spec:
|
||||
- match: Host(`dockmon.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
- name: security-headers@file
|
||||
services:
|
||||
- name: dockmon-service
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
services:
|
||||
downtify:
|
||||
container_name: downtify
|
||||
image: ghcr.io/henriquesebastiao/downtify:3.1.0
|
||||
image: ghcr.io/henriquesebastiao/downtify:3.4.0
|
||||
restart: unless-stopped
|
||||
# ports:
|
||||
# - '7077:8000'
|
||||
|
||||
@@ -20,6 +20,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: downtify
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -27,7 +29,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: downtify
|
||||
image: ghcr.io/henriquesebastiao/downtify:3.1.0
|
||||
image: ghcr.io/henriquesebastiao/downtify:3.4.0
|
||||
ports:
|
||||
- containerPort: 8000
|
||||
volumeMounts:
|
||||
|
||||
@@ -10,8 +10,6 @@ spec:
|
||||
- match: Host(`downtify.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
- name: security-chain@file
|
||||
services:
|
||||
- name: downtify-service
|
||||
|
||||
@@ -1,13 +0,0 @@
|
||||
FROM python:3.9-alpine
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Установка зависимостей
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
# Копирование кода
|
||||
COPY main.py .
|
||||
COPY .env .
|
||||
# Запуск бота
|
||||
CMD ["python", "-u", "main.py"]
|
||||
@@ -1,373 +0,0 @@
|
||||
Mozilla Public License Version 2.0
|
||||
==================================
|
||||
|
||||
1. Definitions
|
||||
--------------
|
||||
|
||||
1.1. "Contributor"
|
||||
means each individual or legal entity that creates, contributes to
|
||||
the creation of, or owns Covered Software.
|
||||
|
||||
1.2. "Contributor Version"
|
||||
means the combination of the Contributions of others (if any) used
|
||||
by a Contributor and that particular Contributor's Contribution.
|
||||
|
||||
1.3. "Contribution"
|
||||
means Covered Software of a particular Contributor.
|
||||
|
||||
1.4. "Covered Software"
|
||||
means Source Code Form to which the initial Contributor has attached
|
||||
the notice in Exhibit A, the Executable Form of such Source Code
|
||||
Form, and Modifications of such Source Code Form, in each case
|
||||
including portions thereof.
|
||||
|
||||
1.5. "Incompatible With Secondary Licenses"
|
||||
means
|
||||
|
||||
(a) that the initial Contributor has attached the notice described
|
||||
in Exhibit B to the Covered Software; or
|
||||
|
||||
(b) that the Covered Software was made available under the terms of
|
||||
version 1.1 or earlier of the License, but not also under the
|
||||
terms of a Secondary License.
|
||||
|
||||
1.6. "Executable Form"
|
||||
means any form of the work other than Source Code Form.
|
||||
|
||||
1.7. "Larger Work"
|
||||
means a work that combines Covered Software with other material, in
|
||||
a separate file or files, that is not Covered Software.
|
||||
|
||||
1.8. "License"
|
||||
means this document.
|
||||
|
||||
1.9. "Licensable"
|
||||
means having the right to grant, to the maximum extent possible,
|
||||
whether at the time of the initial grant or subsequently, any and
|
||||
all of the rights conveyed by this License.
|
||||
|
||||
1.10. "Modifications"
|
||||
means any of the following:
|
||||
|
||||
(a) any file in Source Code Form that results from an addition to,
|
||||
deletion from, or modification of the contents of Covered
|
||||
Software; or
|
||||
|
||||
(b) any new file in Source Code Form that contains any Covered
|
||||
Software.
|
||||
|
||||
1.11. "Patent Claims" of a Contributor
|
||||
means any patent claim(s), including without limitation, method,
|
||||
process, and apparatus claims, in any patent Licensable by such
|
||||
Contributor that would be infringed, but for the grant of the
|
||||
License, by the making, using, selling, offering for sale, having
|
||||
made, import, or transfer of either its Contributions or its
|
||||
Contributor Version.
|
||||
|
||||
1.12. "Secondary License"
|
||||
means either the GNU General Public License, Version 2.0, the GNU
|
||||
Lesser General Public License, Version 2.1, the GNU Affero General
|
||||
Public License, Version 3.0, or any later versions of those
|
||||
licenses.
|
||||
|
||||
1.13. "Source Code Form"
|
||||
means the form of the work preferred for making modifications.
|
||||
|
||||
1.14. "You" (or "Your")
|
||||
means an individual or a legal entity exercising rights under this
|
||||
License. For legal entities, "You" includes any entity that
|
||||
controls, is controlled by, or is under common control with You. For
|
||||
purposes of this definition, "control" means (a) the power, direct
|
||||
or indirect, to cause the direction or management of such entity,
|
||||
whether by contract or otherwise, or (b) ownership of more than
|
||||
fifty percent (50%) of the outstanding shares or beneficial
|
||||
ownership of such entity.
|
||||
|
||||
2. License Grants and Conditions
|
||||
--------------------------------
|
||||
|
||||
2.1. Grants
|
||||
|
||||
Each Contributor hereby grants You a world-wide, royalty-free,
|
||||
non-exclusive license:
|
||||
|
||||
(a) under intellectual property rights (other than patent or trademark)
|
||||
Licensable by such Contributor to use, reproduce, make available,
|
||||
modify, display, perform, distribute, and otherwise exploit its
|
||||
Contributions, either on an unmodified basis, with Modifications, or
|
||||
as part of a Larger Work; and
|
||||
|
||||
(b) under Patent Claims of such Contributor to make, use, sell, offer
|
||||
for sale, have made, import, and otherwise transfer either its
|
||||
Contributions or its Contributor Version.
|
||||
|
||||
2.2. Effective Date
|
||||
|
||||
The licenses granted in Section 2.1 with respect to any Contribution
|
||||
become effective for each Contribution on the date the Contributor first
|
||||
distributes such Contribution.
|
||||
|
||||
2.3. Limitations on Grant Scope
|
||||
|
||||
The licenses granted in this Section 2 are the only rights granted under
|
||||
this License. No additional rights or licenses will be implied from the
|
||||
distribution or licensing of Covered Software under this License.
|
||||
Notwithstanding Section 2.1(b) above, no patent license is granted by a
|
||||
Contributor:
|
||||
|
||||
(a) for any code that a Contributor has removed from Covered Software;
|
||||
or
|
||||
|
||||
(b) for infringements caused by: (i) Your and any other third party's
|
||||
modifications of Covered Software, or (ii) the combination of its
|
||||
Contributions with other software (except as part of its Contributor
|
||||
Version); or
|
||||
|
||||
(c) under Patent Claims infringed by Covered Software in the absence of
|
||||
its Contributions.
|
||||
|
||||
This License does not grant any rights in the trademarks, service marks,
|
||||
or logos of any Contributor (except as may be necessary to comply with
|
||||
the notice requirements in Section 3.4).
|
||||
|
||||
2.4. Subsequent Licenses
|
||||
|
||||
No Contributor makes additional grants as a result of Your choice to
|
||||
distribute the Covered Software under a subsequent version of this
|
||||
License (see Section 10.2) or under the terms of a Secondary License (if
|
||||
permitted under the terms of Section 3.3).
|
||||
|
||||
2.5. Representation
|
||||
|
||||
Each Contributor represents that the Contributor believes its
|
||||
Contributions are its original creation(s) or it has sufficient rights
|
||||
to grant the rights to its Contributions conveyed by this License.
|
||||
|
||||
2.6. Fair Use
|
||||
|
||||
This License is not intended to limit any rights You have under
|
||||
applicable copyright doctrines of fair use, fair dealing, or other
|
||||
equivalents.
|
||||
|
||||
2.7. Conditions
|
||||
|
||||
Sections 3.1, 3.2, 3.3, and 3.4 are conditions of the licenses granted
|
||||
in Section 2.1.
|
||||
|
||||
3. Responsibilities
|
||||
-------------------
|
||||
|
||||
3.1. Distribution of Source Form
|
||||
|
||||
All distribution of Covered Software in Source Code Form, including any
|
||||
Modifications that You create or to which You contribute, must be under
|
||||
the terms of this License. You must inform recipients that the Source
|
||||
Code Form of the Covered Software is governed by the terms of this
|
||||
License, and how they can obtain a copy of this License. You may not
|
||||
attempt to alter or restrict the recipients' rights in the Source Code
|
||||
Form.
|
||||
|
||||
3.2. Distribution of Executable Form
|
||||
|
||||
If You distribute Covered Software in Executable Form then:
|
||||
|
||||
(a) such Covered Software must also be made available in Source Code
|
||||
Form, as described in Section 3.1, and You must inform recipients of
|
||||
the Executable Form how they can obtain a copy of such Source Code
|
||||
Form by reasonable means in a timely manner, at a charge no more
|
||||
than the cost of distribution to the recipient; and
|
||||
|
||||
(b) You may distribute such Executable Form under the terms of this
|
||||
License, or sublicense it under different terms, provided that the
|
||||
license for the Executable Form does not attempt to limit or alter
|
||||
the recipients' rights in the Source Code Form under this License.
|
||||
|
||||
3.3. Distribution of a Larger Work
|
||||
|
||||
You may create and distribute a Larger Work under terms of Your choice,
|
||||
provided that You also comply with the requirements of this License for
|
||||
the Covered Software. If the Larger Work is a combination of Covered
|
||||
Software with a work governed by one or more Secondary Licenses, and the
|
||||
Covered Software is not Incompatible With Secondary Licenses, this
|
||||
License permits You to additionally distribute such Covered Software
|
||||
under the terms of such Secondary License(s), so that the recipient of
|
||||
the Larger Work may, at their option, further distribute the Covered
|
||||
Software under the terms of either this License or such Secondary
|
||||
License(s).
|
||||
|
||||
3.4. Notices
|
||||
|
||||
You may not remove or alter the substance of any license notices
|
||||
(including copyright notices, patent notices, disclaimers of warranty,
|
||||
or limitations of liability) contained within the Source Code Form of
|
||||
the Covered Software, except that You may alter any license notices to
|
||||
the extent required to remedy known factual inaccuracies.
|
||||
|
||||
3.5. Application of Additional Terms
|
||||
|
||||
You may choose to offer, and to charge a fee for, warranty, support,
|
||||
indemnity or liability obligations to one or more recipients of Covered
|
||||
Software. However, You may do so only on Your own behalf, and not on
|
||||
behalf of any Contributor. You must make it absolutely clear that any
|
||||
such warranty, support, indemnity, or liability obligation is offered by
|
||||
You alone, and You hereby agree to indemnify every Contributor for any
|
||||
liability incurred by such Contributor as a result of warranty, support,
|
||||
indemnity or liability terms You offer. You may include additional
|
||||
disclaimers of warranty and limitations of liability specific to any
|
||||
jurisdiction.
|
||||
|
||||
4. Inability to Comply Due to Statute or Regulation
|
||||
---------------------------------------------------
|
||||
|
||||
If it is impossible for You to comply with any of the terms of this
|
||||
License with respect to some or all of the Covered Software due to
|
||||
statute, judicial order, or regulation then You must: (a) comply with
|
||||
the terms of this License to the maximum extent possible; and (b)
|
||||
describe the limitations and the code they affect. Such description must
|
||||
be placed in a text file included with all distributions of the Covered
|
||||
Software under this License. Except to the extent prohibited by statute
|
||||
or regulation, such description must be sufficiently detailed for a
|
||||
recipient of ordinary skill to be able to understand it.
|
||||
|
||||
5. Termination
|
||||
--------------
|
||||
|
||||
5.1. The rights granted under this License will terminate automatically
|
||||
if You fail to comply with any of its terms. However, if You become
|
||||
compliant, then the rights granted under this License from a particular
|
||||
Contributor are reinstated (a) provisionally, unless and until such
|
||||
Contributor explicitly and finally terminates Your grants, and (b) on an
|
||||
ongoing basis, if such Contributor fails to notify You of the
|
||||
non-compliance by some reasonable means prior to 60 days after You have
|
||||
come back into compliance. Moreover, Your grants from a particular
|
||||
Contributor are reinstated on an ongoing basis if such Contributor
|
||||
notifies You of the non-compliance by some reasonable means, this is the
|
||||
first time You have received notice of non-compliance with this License
|
||||
from such Contributor, and You become compliant prior to 30 days after
|
||||
Your receipt of the notice.
|
||||
|
||||
5.2. If You initiate litigation against any entity by asserting a patent
|
||||
infringement claim (excluding declaratory judgment actions,
|
||||
counter-claims, and cross-claims) alleging that a Contributor Version
|
||||
directly or indirectly infringes any patent, then the rights granted to
|
||||
You by any and all Contributors for the Covered Software under Section
|
||||
2.1 of this License shall terminate.
|
||||
|
||||
5.3. In the event of termination under Sections 5.1 or 5.2 above, all
|
||||
end user license agreements (excluding distributors and resellers) which
|
||||
have been validly granted by You or Your distributors under this License
|
||||
prior to termination shall survive termination.
|
||||
|
||||
************************************************************************
|
||||
* *
|
||||
* 6. Disclaimer of Warranty *
|
||||
* ------------------------- *
|
||||
* *
|
||||
* Covered Software is provided under this License on an "as is" *
|
||||
* basis, without warranty of any kind, either expressed, implied, or *
|
||||
* statutory, including, without limitation, warranties that the *
|
||||
* Covered Software is free of defects, merchantable, fit for a *
|
||||
* particular purpose or non-infringing. The entire risk as to the *
|
||||
* quality and performance of the Covered Software is with You. *
|
||||
* Should any Covered Software prove defective in any respect, You *
|
||||
* (not any Contributor) assume the cost of any necessary servicing, *
|
||||
* repair, or correction. This disclaimer of warranty constitutes an *
|
||||
* essential part of this License. No use of any Covered Software is *
|
||||
* authorized under this License except under this disclaimer. *
|
||||
* *
|
||||
************************************************************************
|
||||
|
||||
************************************************************************
|
||||
* *
|
||||
* 7. Limitation of Liability *
|
||||
* -------------------------- *
|
||||
* *
|
||||
* Under no circumstances and under no legal theory, whether tort *
|
||||
* (including negligence), contract, or otherwise, shall any *
|
||||
* Contributor, or anyone who distributes Covered Software as *
|
||||
* permitted above, be liable to You for any direct, indirect, *
|
||||
* special, incidental, or consequential damages of any character *
|
||||
* including, without limitation, damages for lost profits, loss of *
|
||||
* goodwill, work stoppage, computer failure or malfunction, or any *
|
||||
* and all other commercial damages or losses, even if such party *
|
||||
* shall have been informed of the possibility of such damages. This *
|
||||
* limitation of liability shall not apply to liability for death or *
|
||||
* personal injury resulting from such party's negligence to the *
|
||||
* extent applicable law prohibits such limitation. Some *
|
||||
* jurisdictions do not allow the exclusion or limitation of *
|
||||
* incidental or consequential damages, so this exclusion and *
|
||||
* limitation may not apply to You. *
|
||||
* *
|
||||
************************************************************************
|
||||
|
||||
8. Litigation
|
||||
-------------
|
||||
|
||||
Any litigation relating to this License may be brought only in the
|
||||
courts of a jurisdiction where the defendant maintains its principal
|
||||
place of business and such litigation shall be governed by laws of that
|
||||
jurisdiction, without reference to its conflict-of-law provisions.
|
||||
Nothing in this Section shall prevent a party's ability to bring
|
||||
cross-claims or counter-claims.
|
||||
|
||||
9. Miscellaneous
|
||||
----------------
|
||||
|
||||
This License represents the complete agreement concerning the subject
|
||||
matter hereof. If any provision of this License is held to be
|
||||
unenforceable, such provision shall be reformed only to the extent
|
||||
necessary to make it enforceable. Any law or regulation which provides
|
||||
that the language of a contract shall be construed against the drafter
|
||||
shall not be used to construe this License against a Contributor.
|
||||
|
||||
10. Versions of the License
|
||||
---------------------------
|
||||
|
||||
10.1. New Versions
|
||||
|
||||
Mozilla Foundation is the license steward. Except as provided in Section
|
||||
10.3, no one other than the license steward has the right to modify or
|
||||
publish new versions of this License. Each version will be given a
|
||||
distinguishing version number.
|
||||
|
||||
10.2. Effect of New Versions
|
||||
|
||||
You may distribute the Covered Software under the terms of the version
|
||||
of the License under which You originally received the Covered Software,
|
||||
or under the terms of any subsequent version published by the license
|
||||
steward.
|
||||
|
||||
10.3. Modified Versions
|
||||
|
||||
If you create software not governed by this License, and you want to
|
||||
create a new license for such software, you may create and use a
|
||||
modified version of this License if you rename the license and remove
|
||||
any references to the name of the license steward (except to note that
|
||||
such modified license differs from this License).
|
||||
|
||||
10.4. Distributing Source Code Form that is Incompatible With Secondary
|
||||
Licenses
|
||||
|
||||
If You choose to distribute Source Code Form that is Incompatible With
|
||||
Secondary Licenses under the terms of this version of the License, the
|
||||
notice described in Exhibit B of this License must be attached.
|
||||
|
||||
Exhibit A - Source Code Form License Notice
|
||||
-------------------------------------------
|
||||
|
||||
This Source Code Form is subject to the terms of the Mozilla Public
|
||||
License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
||||
|
||||
If it is not possible or desirable to put the notice in a particular
|
||||
file, then You may include the notice in a location (such as a LICENSE
|
||||
file in a relevant directory) where a recipient would be likely to look
|
||||
for such a notice.
|
||||
|
||||
You may add additional accurate notices of copyright ownership.
|
||||
|
||||
Exhibit B - "Incompatible With Secondary Licenses" Notice
|
||||
---------------------------------------------------------
|
||||
|
||||
This Source Code Form is "Incompatible With Secondary Licenses", as
|
||||
defined by the Mozilla Public License, v. 2.0.
|
||||
@@ -1,15 +0,0 @@
|
||||
services:
|
||||
dtek_notif:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
image: gcr.forust.xyz/forust/dtek-notif:latest
|
||||
pull_policy: build
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
- TZ=Europe/Kyiv
|
||||
dns:
|
||||
- 1.1.1.1
|
||||
- 8.8.8.8
|
||||
networks:
|
||||
- default
|
||||
@@ -1,748 +0,0 @@
|
||||
import asyncio
|
||||
import contextlib
|
||||
import logging
|
||||
import os
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
import requests
|
||||
from aiogram import Bot, Dispatcher
|
||||
from aiogram.filters import Command
|
||||
from aiogram.types import KeyboardButton, Message
|
||||
from aiogram.utils.keyboard import ReplyKeyboardBuilder
|
||||
from bs4 import BeautifulSoup
|
||||
from dotenv import load_dotenv
|
||||
|
||||
# Загрузка переменных окружения
|
||||
load_dotenv()
|
||||
|
||||
# Настройки
|
||||
TELEGRAM_TOKEN = os.getenv('TELEGRAM_TOKEN', 'YOUR_TOKEN_HERE')
|
||||
ALLOWED_CHAT_IDS = list(map(int, os.getenv('ALLOWED_CHAT_IDS', '').split(','))) if os.getenv('ALLOWED_CHAT_IDS') else []
|
||||
CHECK_INTERVAL = int(os.getenv('CHECK_INTERVAL', '120'))
|
||||
|
||||
# Параметры для запроса
|
||||
VOE_CITY_ID = int(os.getenv('VOE_CITY_ID', 'VOE_CITY_ID'))
|
||||
VOE_STREET_ID = int(os.getenv('VOE_STREET_ID', 'VOE_STREET_ID'))
|
||||
VOE_HOUSE_ID = int(os.getenv('VOE_HOUSE_ID', 'VOE_HOUSE_ID'))
|
||||
|
||||
# Настройка логирования
|
||||
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Глобальные переменные
|
||||
bot = Bot(token=TELEGRAM_TOKEN)
|
||||
dp = Dispatcher()
|
||||
last_schedule: list[dict] | None = None
|
||||
last_notification_time: dict[str, datetime] = {}
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# УТИЛИТЫ
|
||||
# ============================================================================
|
||||
|
||||
|
||||
def format_time_duration(minutes: int) -> str:
|
||||
"""Форматирует время из минут в часы и минуты"""
|
||||
hours = minutes // 60
|
||||
mins = minutes % 60
|
||||
|
||||
if hours == 0:
|
||||
return f'{mins}м'
|
||||
elif mins == 0:
|
||||
return f'{hours}ч'
|
||||
return f'{hours}ч {mins}м'
|
||||
|
||||
|
||||
def get_day_statistics(day_blocks: list[dict]) -> dict[str, int]:
|
||||
"""Получает статистику по дню"""
|
||||
total_minutes = 0
|
||||
confirmed_minutes = 0
|
||||
possible_minutes = 0
|
||||
|
||||
for block in day_blocks:
|
||||
for half in [block['first_half'], block['second_half']]:
|
||||
if half['status'] == 'off':
|
||||
total_minutes += 30
|
||||
if half['confirmed']:
|
||||
confirmed_minutes += 30
|
||||
else:
|
||||
possible_minutes += 30
|
||||
|
||||
return {'total': total_minutes, 'confirmed': confirmed_minutes, 'possible': possible_minutes}
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# ПАРСИНГ ДАННЫХ
|
||||
# ============================================================================
|
||||
|
||||
|
||||
def parse_html(html: str) -> list[dict]:
|
||||
"""Парсит HTML с графиком отключений (логика от 15.11.2024)"""
|
||||
soup = BeautifulSoup(html, 'html.parser')
|
||||
cells = soup.select('.disconnection-detailed-table-cell.cell')
|
||||
|
||||
schedule = []
|
||||
current_hour = 0
|
||||
current_day = 0
|
||||
|
||||
for cell in cells:
|
||||
if 'legend' in cell.get('class', []) or 'head' in cell.get('class', []):
|
||||
continue
|
||||
|
||||
cell_classes = cell.get('class', [])
|
||||
|
||||
# ПРоверка статуса отключения на весь час
|
||||
full_hour_off = 'has_disconnection' in cell_classes and 'full_hour' in cell_classes
|
||||
|
||||
hour_block = cell.select_one('.hour_block')
|
||||
if not hour_block:
|
||||
continue
|
||||
|
||||
# Проверка подтверждённости отключения для всего часа
|
||||
cell_confirmed = None
|
||||
if 'confirm_1' in cell_classes:
|
||||
cell_confirmed = True
|
||||
elif 'confirm_0' in cell_classes:
|
||||
cell_confirmed = False
|
||||
|
||||
# Проверка половин часа
|
||||
left = hour_block.select_one('.half.left')
|
||||
right = hour_block.select_one('.half.right')
|
||||
|
||||
def parse_half(half, is_full_hour_off: bool, cell_confirmed: bool | None = None) -> dict:
|
||||
"""Парсит половину часа"""
|
||||
if not half:
|
||||
return {'status': 'on', 'queue': None, 'confirmed': None}
|
||||
|
||||
half_classes = half.get('class', [])
|
||||
|
||||
# Если вся ячейка full_hour - используем статус ячейки
|
||||
if is_full_hour_off:
|
||||
return {'status': 'off', 'queue': None, 'confirmed': cell_confirmed}
|
||||
|
||||
# Определяем статус половины
|
||||
if 'has_disconnection' in half_classes:
|
||||
status = 'off'
|
||||
elif 'no_disconnection' in half_classes:
|
||||
status = 'on'
|
||||
else:
|
||||
status = 'on' # По умолчанию считаем включенным
|
||||
|
||||
# Если выключено - ищем подробности
|
||||
queue = None
|
||||
confirmed = None
|
||||
|
||||
if status == 'off':
|
||||
disconnection_div = half.select_one('.disconnection')
|
||||
if disconnection_div:
|
||||
# Ищем номер черги в title
|
||||
if disconnection_div.has_attr('title'):
|
||||
title = disconnection_div['title']
|
||||
if 'Номер черги' in title or 'Номер черги:' in title:
|
||||
with contextlib.suppress(BaseException):
|
||||
queue = title.split(':')[-1].strip()
|
||||
|
||||
# Определяем подтверждение
|
||||
disc_classes = disconnection_div.get('class', [])
|
||||
if 'disconnection_confirm_1' in disc_classes:
|
||||
confirmed = True
|
||||
elif 'disconnection_confirm_0' in disc_classes:
|
||||
confirmed = False
|
||||
|
||||
return {'status': status, 'queue': queue, 'confirmed': confirmed}
|
||||
|
||||
first_half_data = parse_half(left, full_hour_off, cell_confirmed)
|
||||
second_half_data = parse_half(right, full_hour_off, cell_confirmed)
|
||||
|
||||
schedule.append(
|
||||
{
|
||||
'hour': current_hour,
|
||||
'day': current_day,
|
||||
'first_half': first_half_data,
|
||||
'second_half': second_half_data,
|
||||
}
|
||||
)
|
||||
|
||||
current_hour += 1
|
||||
if current_hour >= 24:
|
||||
current_hour = 0
|
||||
current_day += 1
|
||||
|
||||
return schedule
|
||||
|
||||
|
||||
def get_voe_html(city_id: int, street_id: int, house_id: int) -> str:
|
||||
"""Получает HTML с сайта VOE"""
|
||||
url = 'https://www.voe.com.ua/disconnection/detailed?ajax_form=1&_wrapper_format=drupal_ajax'
|
||||
headers = {
|
||||
'Content-Type': 'application/x-www-form-urlencoded; charset=UTF-8',
|
||||
'X-Requested-With': 'XMLHttpRequest',
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
|
||||
}
|
||||
data = {
|
||||
'search_type': 0,
|
||||
'city_id': city_id,
|
||||
'street_id': street_id,
|
||||
'house_id': house_id,
|
||||
'form_build_id': 'form-Irv5aHw1R2FT_Ik2apyHOZ47hTH5xPNH_LQnBrmpSTc',
|
||||
'form_id': 'disconnection_detailed_search_form',
|
||||
'_triggering_element_name': 'search',
|
||||
'_triggering_element_value': 'Показати',
|
||||
'_drupal_ajax': 1,
|
||||
}
|
||||
|
||||
try:
|
||||
response = requests.post(url, headers=headers, data=data, timeout=10)
|
||||
response.raise_for_status()
|
||||
resp_json = response.json()
|
||||
|
||||
insert_html = next((item['data'] for item in resp_json if item.get('command') == 'insert'), None)
|
||||
|
||||
if not insert_html:
|
||||
raise ValueError('HTML не найден в ответе')
|
||||
|
||||
return insert_html
|
||||
except requests.exceptions.RequestException as e:
|
||||
logger.error(f'Ошибка запроса VOE: {e}')
|
||||
raise
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# ФОРМАТИРОВАНИЕ СООБЩЕНИЙ
|
||||
# ============================================================================
|
||||
|
||||
|
||||
def get_main_keyboard():
|
||||
"""Создает главную клавиатуру"""
|
||||
builder = ReplyKeyboardBuilder()
|
||||
builder.row(KeyboardButton(text='📊 Графік'), KeyboardButton(text='🔄 Оновити'))
|
||||
builder.row(KeyboardButton(text='📅 Сьогодні'), KeyboardButton(text='📅 Завтра'))
|
||||
builder.row(KeyboardButton(text='ℹ️ Про бота'))
|
||||
return builder.as_markup(resize_keyboard=True)
|
||||
|
||||
|
||||
def format_schedule_message(schedule: list[dict], days_to_show: int = 2) -> str:
|
||||
"""Форматирует полный график на несколько дней"""
|
||||
lines = [
|
||||
'⚡️ <b>Графік відключень світла</b>',
|
||||
f'🕐 Оновлено: {datetime.now().strftime("%d.%m.%Y %H:%M:%S")}',
|
||||
'─' * 30,
|
||||
'',
|
||||
]
|
||||
|
||||
start_date = datetime.now()
|
||||
|
||||
for day in range(min(days_to_show, 2)):
|
||||
day_blocks = [b for b in schedule if b['day'] == day]
|
||||
if not day_blocks:
|
||||
continue
|
||||
|
||||
date_str = (start_date + timedelta(days=day)).strftime('%d.%m.%Y')
|
||||
day_name = '🌅 <b>Сьогодні</b>' if day == 0 else '🌄 <b>Завтра</b>'
|
||||
|
||||
lines.append(f'{day_name} ({date_str})')
|
||||
|
||||
# Статистика
|
||||
stats = get_day_statistics(day_blocks)
|
||||
if stats['total'] > 0:
|
||||
lines.append(f'⏱ Всього: <code>{format_time_duration(stats["total"])}</code>')
|
||||
if stats['confirmed'] > 0:
|
||||
lines.append(f'🔴 Підтверджено: <code>{format_time_duration(stats["confirmed"])}</code>')
|
||||
if stats['possible'] > 0:
|
||||
lines.append(f'🟠 Можливо: <code>{format_time_duration(stats["possible"])}</code>')
|
||||
else:
|
||||
lines.append('🟢 <b>Відключень немає!</b>')
|
||||
|
||||
lines.append('')
|
||||
|
||||
# Детальный список отключений
|
||||
disconnections = []
|
||||
current_status = None
|
||||
start_time = None
|
||||
current_confirmed = None
|
||||
current_queue = None
|
||||
|
||||
for block in day_blocks:
|
||||
hour = block['hour']
|
||||
|
||||
for half_idx, half in enumerate([block['first_half'], block['second_half']]):
|
||||
time_str = f'{hour:02d}:00' if half_idx == 0 else f'{hour:02d}:30'
|
||||
|
||||
if half['status'] == 'off':
|
||||
if current_status != 'off':
|
||||
start_time = time_str
|
||||
current_confirmed = half['confirmed']
|
||||
current_queue = half['queue']
|
||||
current_status = 'off'
|
||||
else:
|
||||
if current_status == 'off':
|
||||
icon = '🔴' if current_confirmed else '🟠'
|
||||
queue_text = f' (Ч{current_queue})' if current_queue else ''
|
||||
disconnections.append(f'{icon} <code>{start_time} - {time_str}</code>{queue_text}')
|
||||
current_status = half['status']
|
||||
|
||||
# Если день закончился на отключении
|
||||
if current_status == 'off':
|
||||
icon = '🔴' if current_confirmed else '🟠'
|
||||
queue_text = f' (Ч{current_queue})' if current_queue else ''
|
||||
next_hour = (day_blocks[-1]['hour'] + 1) % 24
|
||||
end_time = f'{next_hour:02d}:00'
|
||||
disconnections.append(f'{icon} <code>{start_time} - {end_time}</code>{queue_text}')
|
||||
|
||||
if disconnections:
|
||||
for idx, disc in enumerate(disconnections, 1):
|
||||
lines.append(f'{idx}. {disc}')
|
||||
|
||||
lines.append('')
|
||||
|
||||
lines.append('<i>🔴 = підтверджено • 🟠 = можливо • 🟢 = світло</i>')
|
||||
|
||||
return '\n'.join(lines)
|
||||
|
||||
|
||||
def format_single_day_schedule(schedule: list[dict], day: int) -> str:
|
||||
"""Форматирует график на один день"""
|
||||
day_blocks = [b for b in schedule if b['day'] == day]
|
||||
if not day_blocks:
|
||||
return '❌ Немає даних для цього дня'
|
||||
|
||||
start_date = datetime.now()
|
||||
date_str = (start_date + timedelta(days=day)).strftime('%d.%m.%Y')
|
||||
day_name = '🟠 <b>Сьогодні</b>' if day == 0 else '🔶 <b>Завтра</b>'
|
||||
|
||||
lines = [f'{day_name} • {date_str}', '']
|
||||
|
||||
# Статистика
|
||||
lines.append('<b>📊 Статистика</b>')
|
||||
stats = get_day_statistics(day_blocks)
|
||||
|
||||
if stats['total'] == 0:
|
||||
lines.append('└ 🟢 <b>Відключень немає!</b>')
|
||||
else:
|
||||
total_time = format_time_duration(stats['total'])
|
||||
lines.append(f'├ ⏱ Всього: <code>{total_time}</code>')
|
||||
|
||||
if stats['confirmed'] > 0:
|
||||
confirmed_time = format_time_duration(stats['confirmed'])
|
||||
lines.append(f'├ 🔴 Підтверджено: <code>{confirmed_time}</code>')
|
||||
|
||||
if stats['possible'] > 0:
|
||||
possible_time = format_time_duration(stats['possible'])
|
||||
lines.append(f'└ 🟠 Можливо: <code>{possible_time}</code>')
|
||||
else:
|
||||
lines.append('└ 🟢 Решта часу світло')
|
||||
|
||||
lines.append('')
|
||||
|
||||
# Детальный список отключений
|
||||
disconnections = []
|
||||
current_status = None
|
||||
start_time = None
|
||||
current_confirmed = None
|
||||
current_queue = None
|
||||
|
||||
for block in day_blocks:
|
||||
hour = block['hour']
|
||||
|
||||
for half_idx, half in enumerate([block['first_half'], block['second_half']]):
|
||||
time_str = f'{hour:02d}:00' if half_idx == 0 else f'{hour:02d}:30'
|
||||
|
||||
if half['status'] == 'off':
|
||||
if current_status != 'off':
|
||||
start_time = time_str
|
||||
current_confirmed = half['confirmed']
|
||||
current_queue = half['queue']
|
||||
current_status = 'off'
|
||||
else:
|
||||
if current_status == 'off':
|
||||
icon = '🔴' if current_confirmed else '🟠'
|
||||
queue_text = f' (Ч.{current_queue})' if current_queue else ''
|
||||
disconnections.append(f'{icon} <code>{start_time} - {time_str}</code>{queue_text}')
|
||||
current_status = half['status']
|
||||
|
||||
# Если день закончился на отключении
|
||||
if current_status == 'off':
|
||||
icon = '🔴' if current_confirmed else '🟠'
|
||||
queue_text = f' (Ч.{current_queue})' if current_queue else ''
|
||||
next_hour = (day_blocks[-1]['hour'] + 1) % 24
|
||||
end_time = f'{next_hour:02d}:00'
|
||||
disconnections.append(f'{icon} <code>{start_time} - {end_time}</code>{queue_text}')
|
||||
|
||||
if disconnections:
|
||||
lines.append('<b>⚡️ Розклад відключень</b>')
|
||||
for idx, disc in enumerate(disconnections, 1):
|
||||
lines.append(f'{idx}. {disc}')
|
||||
|
||||
lines.append('')
|
||||
lines.append('<i>🔴 підтверджено • 🟠 можливо • 🟢 світло</i>')
|
||||
|
||||
return '\n'.join(lines)
|
||||
|
||||
|
||||
def schedules_differ(old_schedule: list[dict] | None, new_schedule: list[dict] | None) -> bool:
|
||||
"""Проверяет отличия между графиками"""
|
||||
if old_schedule is None or new_schedule is None:
|
||||
return True
|
||||
|
||||
if len(old_schedule) != len(new_schedule):
|
||||
return True
|
||||
|
||||
for old, new in zip(old_schedule, new_schedule, strict=False):
|
||||
if old['day'] >= 2:
|
||||
break
|
||||
|
||||
if old['first_half'] != new['first_half'] or old['second_half'] != new['second_half']:
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# УВЕДОМЛЕНИЯ
|
||||
# ============================================================================
|
||||
|
||||
|
||||
async def send_to_all_users(message_text: str, parse_mode: str = 'HTML'):
|
||||
"""Отправляет сообщение всем пользователям"""
|
||||
if not ALLOWED_CHAT_IDS:
|
||||
logger.warning('Нет допущенных ID чатов для отправки уведомлений')
|
||||
return
|
||||
|
||||
for chat_id in ALLOWED_CHAT_IDS:
|
||||
try:
|
||||
await bot.send_message(chat_id, message_text, parse_mode=parse_mode)
|
||||
logger.info(f'✅ Сообщение отправлено пользователю {chat_id}')
|
||||
except Exception as e:
|
||||
logger.error(f'❌ Ошибка отправки пользователю {chat_id}: {e}')
|
||||
await asyncio.sleep(0.5)
|
||||
|
||||
|
||||
async def check_schedule():
|
||||
"""Проверяет график и отправляет уведомления"""
|
||||
global last_schedule
|
||||
|
||||
try:
|
||||
logger.info('🔍 Проверка графика...')
|
||||
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
|
||||
new_schedule = parse_html(html)
|
||||
|
||||
if schedules_differ(last_schedule, new_schedule):
|
||||
logger.info('✨ Обнаружены изменения!')
|
||||
message = format_schedule_message(new_schedule, days_to_show=2)
|
||||
|
||||
if last_schedule is not None:
|
||||
await send_to_all_users(f'🔄 <b>Графік оновлено!</b>\n\n{message}')
|
||||
|
||||
last_schedule = new_schedule
|
||||
else:
|
||||
logger.info('✓ Графік без змін')
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f'❌ Ошибка при проверке графика: {e}')
|
||||
|
||||
|
||||
async def check_upcoming_disconnections():
|
||||
"""Проверяет предстоящие события и отправляет предупреждения за 5 минут"""
|
||||
global last_notification_time
|
||||
|
||||
if last_schedule is None:
|
||||
return
|
||||
|
||||
now = datetime.now()
|
||||
today_blocks = [b for b in last_schedule if b['day'] == 0]
|
||||
|
||||
# Создаем список всех переходов (off -> on или on -> off)
|
||||
transitions = []
|
||||
prev_status = None
|
||||
|
||||
for block in today_blocks:
|
||||
hour = block['hour']
|
||||
|
||||
for half_idx, half in enumerate([block['first_half'], block['second_half']]):
|
||||
minute = 0 if half_idx == 0 else 30
|
||||
time_str = f'{hour:02d}:{minute:02d}'
|
||||
|
||||
current_status = half['status']
|
||||
|
||||
# Если статус изменился - это переход
|
||||
if prev_status is not None and prev_status != current_status:
|
||||
transitions.append(
|
||||
{
|
||||
'hour': hour,
|
||||
'minute': minute,
|
||||
'time_str': time_str,
|
||||
'from_status': prev_status,
|
||||
'to_status': current_status,
|
||||
'confirmed': half.get('confirmed'),
|
||||
'queue': half.get('queue'),
|
||||
}
|
||||
)
|
||||
|
||||
prev_status = current_status
|
||||
|
||||
# Проверяем переходы
|
||||
for transition in transitions:
|
||||
event_time = now.replace(hour=transition['hour'], minute=transition['minute'], second=0, microsecond=0)
|
||||
|
||||
time_until = (event_time - now).total_seconds() / 60
|
||||
notification_key = f'{transition["hour"]}:{transition["minute"]}_{transition["to_status"]}'
|
||||
|
||||
# Если за 5 минут до события (±1 минута) и еще не отправляли
|
||||
if 4 <= time_until <= 6:
|
||||
# Проверяем, не отправляли ли уже уведомление сегодня
|
||||
if notification_key in last_notification_time:
|
||||
last_notif_time = last_notification_time[notification_key]
|
||||
if last_notif_time.date() == now.date():
|
||||
continue # Уже отправляли сегодня
|
||||
|
||||
# Переход на ОТКЛЮЧЕНИЕ (on -> off)
|
||||
if transition['from_status'] == 'on' and transition['to_status'] == 'off':
|
||||
icon = '🔴' if transition['confirmed'] else '🟠'
|
||||
status = 'підтверджено' if transition['confirmed'] else 'можливе'
|
||||
queue_info = f' (Черга {transition["queue"]})' if transition['queue'] else ''
|
||||
|
||||
warning = (
|
||||
f'⚠️ <b>УВАГА! ВІДКЛЮЧЕННЯ</b>\n\n'
|
||||
f'Через ~5 хвилин\n'
|
||||
f'Час: <code>{transition["time_str"]}</code>\n'
|
||||
f'Статус: {icon} {status}{queue_info}'
|
||||
)
|
||||
|
||||
await send_to_all_users(warning)
|
||||
last_notification_time[notification_key] = now
|
||||
logger.info(f'📢 Відправлено попередження про ВІДКЛЮЧЕННЯ в {transition["time_str"]}')
|
||||
|
||||
# Переход на ВКЛЮЧЕНИЕ (off -> on)
|
||||
elif transition['from_status'] == 'off' and transition['to_status'] == 'on':
|
||||
warning = (
|
||||
f'✅ <b>УВАГА! ВКЛЮЧЕННЯ</b>\n\n'
|
||||
f'Через ~5 хвилин буде світло\n'
|
||||
f'Час: <code>{transition["time_str"]}</code>'
|
||||
)
|
||||
|
||||
await send_to_all_users(warning)
|
||||
last_notification_time[notification_key] = now
|
||||
logger.info(f'📢 Відправлено попередження про ВКЛЮЧЕННЯ в {transition["time_str"]}')
|
||||
|
||||
|
||||
async def monitoring_loop():
|
||||
"""Основной цикл мониторинга"""
|
||||
await check_schedule()
|
||||
|
||||
while True:
|
||||
try:
|
||||
await asyncio.sleep(CHECK_INTERVAL)
|
||||
await check_schedule()
|
||||
await check_upcoming_disconnections()
|
||||
except Exception as e:
|
||||
logger.error(f'Ошибка в цикле мониторинга: {e}')
|
||||
await asyncio.sleep(5)
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# ОБРАБОТЧИКИ КОМАНД
|
||||
# ============================================================================
|
||||
|
||||
|
||||
@dp.message(Command('start'))
|
||||
async def cmd_start(message: Message):
|
||||
"""Обработчик /start"""
|
||||
if message.chat.id not in ALLOWED_CHAT_IDS:
|
||||
await message.answer('❌ У вас немає доступу до цього бота.')
|
||||
return
|
||||
|
||||
await message.answer(
|
||||
'👋 <b>Ласкаво просимо!</b>\n\n'
|
||||
'🤖 <b>Бот для моніторингу графіку відключень світла</b>\n\n'
|
||||
'✨ <b>Можливості:</b>\n'
|
||||
'• 📊 Перегляд графіку на сьогодні і завтра\n'
|
||||
'• 🔔 Автоматичні сповіщення за 5 хвилин до подій\n'
|
||||
'• 🔄 Моніторинг змін графіку\n\n'
|
||||
'Використовуйте кнопки нижче 👇',
|
||||
parse_mode='HTML',
|
||||
reply_markup=get_main_keyboard(),
|
||||
)
|
||||
|
||||
|
||||
@dp.message(lambda msg: msg.text == 'ℹ️ Про бота')
|
||||
async def cmd_info(message: Message):
|
||||
"""Показывает информацию о боте"""
|
||||
if message.chat.id not in ALLOWED_CHAT_IDS:
|
||||
return
|
||||
|
||||
await message.answer(
|
||||
'<b>ℹ️ Про бота</b>\n\n'
|
||||
'🚀 <b>Версія:</b> 2.2 (Стабільна)\n\n'
|
||||
'📝 <b>Реліз-ноути:</b>\n'
|
||||
'├ 15.11.2024: Адаптація під оновлену логіку сайту VOE\n'
|
||||
'├ Виправлено парсинг half.left та half.right\n'
|
||||
'├ Покращено визначення підтвердження відключень\n'
|
||||
'└ Оптимізовано обробку статусу для всієї години\n\n'
|
||||
'⚡ <b>Функціональність:</b>\n'
|
||||
'├ Моніторинг графіку 24/7\n'
|
||||
'├ Сповіщення за 5 хвилин\n'
|
||||
'├ Детальна статистика дня\n'
|
||||
'└ Красива візуалізація\n\n'
|
||||
'🔐 <b>Безпека:</b> Використовуються .env файли\n'
|
||||
'💾 <b>Джерело:</b> voe.com.ua',
|
||||
parse_mode='HTML',
|
||||
reply_markup=get_main_keyboard(),
|
||||
)
|
||||
|
||||
|
||||
@dp.message(Command('schedule'))
|
||||
async def cmd_schedule(message: Message):
|
||||
"""Показывает полный график"""
|
||||
if message.chat.id not in ALLOWED_CHAT_IDS:
|
||||
await message.answer('❌ У вас немає доступу.')
|
||||
return
|
||||
|
||||
try:
|
||||
await message.answer('⏳ Завантаження графіку...')
|
||||
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
|
||||
schedule = parse_html(html)
|
||||
text = format_schedule_message(schedule, days_to_show=2)
|
||||
await message.answer(text, parse_mode='HTML', reply_markup=get_main_keyboard())
|
||||
except Exception as e:
|
||||
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
|
||||
|
||||
|
||||
@dp.message(Command('today'))
|
||||
async def cmd_today(message: Message):
|
||||
"""Показывает график на сегодня"""
|
||||
if message.chat.id not in ALLOWED_CHAT_IDS:
|
||||
await message.answer('❌ У вас немає доступу.')
|
||||
return
|
||||
|
||||
try:
|
||||
await message.answer('⏳ Завантаження графіку сьогодні...')
|
||||
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
|
||||
schedule = parse_html(html)
|
||||
text = format_single_day_schedule(schedule, 0)
|
||||
await message.answer(text, parse_mode='HTML', reply_markup=get_main_keyboard())
|
||||
except Exception as e:
|
||||
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
|
||||
|
||||
|
||||
@dp.message(Command('tomorrow'))
|
||||
async def cmd_tomorrow(message: Message):
|
||||
"""Показывает график на завтра"""
|
||||
if message.chat.id not in ALLOWED_CHAT_IDS:
|
||||
await message.answer('❌ У вас немає доступу.')
|
||||
return
|
||||
|
||||
try:
|
||||
await message.answer('⏳ Завантаження графіку завтра...')
|
||||
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
|
||||
schedule = parse_html(html)
|
||||
text = format_single_day_schedule(schedule, 1)
|
||||
await message.answer(text, parse_mode='HTML', reply_markup=get_main_keyboard())
|
||||
# await message.answer("❌ Функція тимчасово недоступна. Чекаємо на оновлення сайту", parse_mode="HTML", reply_markup=get_main_keyboard())
|
||||
except Exception as e:
|
||||
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
|
||||
|
||||
|
||||
@dp.message(Command('check'))
|
||||
async def cmd_check(message: Message):
|
||||
"""Принудительная проверка графика"""
|
||||
if message.chat.id not in ALLOWED_CHAT_IDS:
|
||||
await message.answer('❌ У вас немає доступу.')
|
||||
return
|
||||
|
||||
try:
|
||||
await message.answer('🔄 <b>Перевіряю графік...</b>', parse_mode='HTML')
|
||||
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
|
||||
new_schedule = parse_html(html)
|
||||
|
||||
prefix = (
|
||||
'✅ <b>Знайдено зміни!</b>\n\n'
|
||||
if schedules_differ(last_schedule, new_schedule)
|
||||
else '✓ <b>Графік без змін</b>\n\n'
|
||||
)
|
||||
result = prefix + format_schedule_message(new_schedule, days_to_show=2)
|
||||
|
||||
await message.answer(result, parse_mode='HTML', reply_markup=get_main_keyboard())
|
||||
except Exception as e:
|
||||
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
|
||||
|
||||
|
||||
@dp.message()
|
||||
async def handle_text(message: Message):
|
||||
"""Обработчик текстовых сообщений и кнопок"""
|
||||
if message.chat.id not in ALLOWED_CHAT_IDS:
|
||||
return
|
||||
|
||||
text = message.text
|
||||
|
||||
# Кнопка "Графік"
|
||||
if text == '📊 Графік':
|
||||
await cmd_schedule(message)
|
||||
|
||||
# Кнопка "Сьогодні"
|
||||
elif text == '📅 Сьогодні':
|
||||
await cmd_today(message)
|
||||
|
||||
# Кнопка "Завтра"
|
||||
elif text == '📅 Завтра':
|
||||
await cmd_tomorrow(message)
|
||||
|
||||
# Кнопка "Оновити"
|
||||
elif text == '🔄 Оновити':
|
||||
await cmd_check(message)
|
||||
|
||||
# Кнопка "Про бота"
|
||||
elif text == 'ℹ️ Про бота':
|
||||
await cmd_info(message)
|
||||
|
||||
# Неизвестная команда
|
||||
else:
|
||||
await message.answer(
|
||||
'❓ <b>Команда не розпізнана</b>\n\n'
|
||||
'Використовуйте кнопки на клавіатурі або команди:\n'
|
||||
'/start • /today • /tomorrow • /schedule • /check',
|
||||
parse_mode='HTML',
|
||||
reply_markup=get_main_keyboard(),
|
||||
)
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# ГЛАВНАЯ ФУНКЦИЯ
|
||||
# ============================================================================
|
||||
|
||||
|
||||
async def main():
|
||||
"""Главная функция"""
|
||||
logger.info('=' * 50)
|
||||
logger.info('ЗАПУСК БОТА V2.2 (stable 2.2, 15.11.2025)')
|
||||
logger.info('=' * 50)
|
||||
|
||||
if not TELEGRAM_TOKEN or os.getenv('TELEGRAM_TOKEN', 'YOUR_TOKEN_HERE') == TELEGRAM_TOKEN:
|
||||
logger.error('❌ TELEGRAM_TOKEN не конфігурований! Напишіть токен в .env файл')
|
||||
return
|
||||
|
||||
if not ALLOWED_CHAT_IDS:
|
||||
logger.error('❌ ALLOWED_CHAT_IDS не конфігуровані! Напишіть ID в .env файл')
|
||||
return
|
||||
|
||||
logger.info(f'📌 Allowed chat ids: {ALLOWED_CHAT_IDS}')
|
||||
logger.info(f'⏱ Інтервал перевірки: {CHECK_INTERVAL} сек')
|
||||
logger.info('=' * 50)
|
||||
|
||||
# Запускаем мониторинг
|
||||
monitoring_task = asyncio.create_task(monitoring_loop())
|
||||
|
||||
try:
|
||||
await dp.start_polling(bot)
|
||||
except KeyboardInterrupt:
|
||||
logger.info('⏹ Бот зупинений користувачем')
|
||||
finally:
|
||||
monitoring_task.cancel()
|
||||
await bot.session.close()
|
||||
logger.info('✓ Підключення закрито')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
asyncio.run(main())
|
||||
except KeyboardInterrupt:
|
||||
logger.info('⏹ Завершено')
|
||||
@@ -1,7 +0,0 @@
|
||||
[project]
|
||||
name = "dtek-notif"
|
||||
version = "0.1.0"
|
||||
description = "Add your description here"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.13"
|
||||
dependencies = []
|
||||
@@ -1,5 +0,0 @@
|
||||
requests>=2.31.0
|
||||
beautifulsoup4>=4.12.0
|
||||
aiogram>=3.3.0
|
||||
python-dotenv>=1.0.0
|
||||
aiohttp>=3.9.0
|
||||
@@ -17,7 +17,7 @@ services:
|
||||
|
||||
session-keeper:
|
||||
build: ./phpsessid-bot
|
||||
image: gcr.forust.xyz/forust/session-keeper:latest
|
||||
image: gcr.forust.xyz/forust/session-keeper:prod
|
||||
pull_policy: build
|
||||
env_file: .env
|
||||
restart: unless-stopped
|
||||
@@ -33,7 +33,7 @@ services:
|
||||
|
||||
webinar-checker:
|
||||
build: ./webinar-checker
|
||||
image: gcr.forust.xyz/forust/webinar-checker:latest
|
||||
image: gcr.forust.xyz/forust/webinar-checker:prod
|
||||
pull_policy: build
|
||||
env_file: .env
|
||||
restart: unless-stopped
|
||||
|
||||
@@ -11,9 +11,14 @@ spec:
|
||||
rules:
|
||||
# No successful webinar check for 5m (~2-3 missed 2-min checks).
|
||||
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
|
||||
# The last_success > 0 guard is mandatory: checker.py initialises
|
||||
# last_success to 0, so without it `time() - 0` equals the current epoch
|
||||
# and humanizeDuration renders ~20722d on every pod restart. Keep the
|
||||
# duration expression on the left so $value stays the real gap.
|
||||
- alert: WebinarCheckerNoSuccessfulCheck
|
||||
expr: |
|
||||
(time() - webinar_check_last_success_timestamp_seconds > 300)
|
||||
((time() - webinar_check_last_success_timestamp_seconds) > 300)
|
||||
and (webinar_check_last_success_timestamp_seconds > 0)
|
||||
and (webinar_check_last_run_timestamp_seconds > 0)
|
||||
for: 2m
|
||||
labels:
|
||||
@@ -22,6 +27,20 @@ spec:
|
||||
summary: "Webinar checker has no successful check for 5m"
|
||||
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
|
||||
|
||||
# Checks are running but none has ever succeeded since pod start.
|
||||
# Split out from the rule above so a zeroed gauge never feeds
|
||||
# humanizeDuration.
|
||||
- alert: WebinarCheckerNeverSucceeded
|
||||
expr: |
|
||||
(webinar_check_last_success_timestamp_seconds == 0)
|
||||
and (webinar_check_last_run_timestamp_seconds > 0)
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker has never completed a successful check"
|
||||
description: 'edu-master/webinar-checker: checks have been running for 10m but not one has ever succeeded since the pod started, so every check is failing. Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
|
||||
|
||||
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
|
||||
- alert: WebinarCheckerConsecutiveFailures
|
||||
expr: |
|
||||
|
||||
@@ -10,6 +10,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: edu-master-playwright
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -20,6 +22,15 @@ spec:
|
||||
# renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker
|
||||
image: mcr.microsoft.com/playwright:v1.56.0-jammy
|
||||
imagePullPolicy: IfNotPresent
|
||||
# p95 412M, max 478M over 7 days, no limit before. Request is set at p95
|
||||
# so the pod is not an eviction candidate; the limit stays above 2x the
|
||||
# request because browser page lifetimes are unpredictable.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "200m"
|
||||
memory: "416Mi"
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
command:
|
||||
- npx
|
||||
- -y
|
||||
|
||||
@@ -28,10 +28,10 @@ spec:
|
||||
resources:
|
||||
requests:
|
||||
cpu: 25m
|
||||
memory: 64Mi
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: 250m
|
||||
memory: 256Mi
|
||||
memory: 128Mi
|
||||
readinessProbe:
|
||||
exec:
|
||||
command: ["redis-cli", "ping"]
|
||||
|
||||
@@ -10,6 +10,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: edu-master-session-keeper
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -31,18 +33,17 @@ spec:
|
||||
echo "redis is ready"
|
||||
containers:
|
||||
- name: session-keeper
|
||||
image: gcr.forust.xyz/forust/session-keeper:latest
|
||||
imagePullPolicy: Always
|
||||
image: gcr.forust.xyz/forust/session-keeper:prod
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: edu-master-secrets
|
||||
resources:
|
||||
requests:
|
||||
cpu: 25m
|
||||
memory: 96Mi
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: 250m
|
||||
memory: 256Mi
|
||||
memory: 128Mi
|
||||
readinessProbe:
|
||||
exec:
|
||||
command: ["/bin/sh", "-ec", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"]
|
||||
|
||||
@@ -10,6 +10,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: edu-master-webinar-checker
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -45,12 +47,19 @@ spec:
|
||||
echo "playwright ok"
|
||||
containers:
|
||||
- name: webinar-checker
|
||||
image: gcr.forust.xyz/forust/webinar-checker:latest
|
||||
imagePullPolicy: Always
|
||||
image: gcr.forust.xyz/forust/webinar-checker:prod
|
||||
ports:
|
||||
- name: metrics
|
||||
containerPort: 8000
|
||||
protocol: TCP
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /health
|
||||
port: metrics
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 3
|
||||
failureThreshold: 12
|
||||
initialDelaySeconds: 10
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: edu-master-secrets
|
||||
@@ -60,7 +69,7 @@ spec:
|
||||
resources:
|
||||
requests:
|
||||
cpu: "50m"
|
||||
memory: "128Mi"
|
||||
memory: "192Mi"
|
||||
limits:
|
||||
cpu: "600m"
|
||||
memory: "512Mi"
|
||||
memory: "384Mi"
|
||||
@@ -1,7 +1,7 @@
|
||||
services:
|
||||
errorpage:
|
||||
build: .
|
||||
image: gcr.forust.xyz/forust/error-pages:latest
|
||||
image: gcr.forust.xyz/forust/error-pages:prod
|
||||
pull_policy: build
|
||||
container_name: error-pages
|
||||
restart: unless-stopped
|
||||
|
||||
@@ -20,6 +20,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: error-pages
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -27,7 +29,21 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: error-pages
|
||||
image: gcr.forust.xyz/forust/error-pages:latest
|
||||
image: gcr.forust.xyz/forust/error-pages:prod
|
||||
# p95 6M, max 10M, no limit before.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "10m"
|
||||
memory: "32Mi"
|
||||
limits:
|
||||
memory: "128Mi"
|
||||
ports:
|
||||
- containerPort: 80
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /404.html
|
||||
port: 80
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 2
|
||||
failureThreshold: 3
|
||||
---
|
||||
+6
-3
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
server:
|
||||
image: docker.gitea.com/gitea:1.27.3
|
||||
image: docker.gitea.com/gitea:28.0.0
|
||||
container_name: gitea
|
||||
restart: always
|
||||
environment:
|
||||
@@ -13,9 +13,12 @@ services:
|
||||
- GITEA__database__PASSWD=gitea
|
||||
- GITEA__database__NAME=gitea
|
||||
# Server
|
||||
- GITEA__server__ROOT_URL=https://gitea.forust.xyz
|
||||
- GITEA__server__ROOT_URL=https://git.forust.xyz
|
||||
- GITEA__server__SSH_DOMAIN=gitssh.forust.xyz
|
||||
- GITEA__server__SSH_PORT=2221
|
||||
# Pin 28.0 defaults explicitly (see k8s/config.yaml for rationale)
|
||||
- GITEA__service__DISABLE_REGISTRATION=true
|
||||
- GITEA__actions__RUN_RETENTION_DAYS=90
|
||||
# Mailer
|
||||
- GITEA__mailer__ENABLED=true
|
||||
- GITEA__mailer__FROM=${SERVICE_EMAIL}
|
||||
@@ -34,7 +37,7 @@ services:
|
||||
- "traefik.http.services.gitea.loadbalancer.server.port=3000"
|
||||
|
||||
# Prod Router
|
||||
- "traefik.http.routers.gitea.rule=Host(`gitea.forust.xyz`)"
|
||||
- "traefik.http.routers.gitea.rule=Host(`git.forust.xyz`) || Host(`gitea.forust.xyz`)"
|
||||
- "traefik.http.routers.gitea.entrypoints=websecure"
|
||||
- "traefik.http.routers.gitea.tls.certresolver"
|
||||
# Local Router
|
||||
|
||||
@@ -8,6 +8,7 @@ spec:
|
||||
dnsNames:
|
||||
- gcr.forust.xyz
|
||||
- gitea.forust.xyz
|
||||
- git.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
|
||||
+11
-2
@@ -4,11 +4,14 @@ metadata:
|
||||
name: gitea-config
|
||||
namespace: gitea
|
||||
data:
|
||||
GITEA__server__DOMAIN: "gitea.forust.xyz"
|
||||
GITEA__server__ROOT_URL: "https://gitea.forust.xyz"
|
||||
GITEA__server__ROOT_URL: "https://git.forust.xyz"
|
||||
GITEA__server__SSH_DOMAIN: "gitssh.forust.xyz"
|
||||
GITEA__server__SSH_PORT: "2221"
|
||||
|
||||
GITEA__service__DISABLE_REGISTRATION: "true"
|
||||
|
||||
GITEA__actions__RUN_RETENTION_DAYS: "90"
|
||||
|
||||
GITEA__database__DB_TYPE: "postgres"
|
||||
GITEA__database__HOST: "postgres.database.svc.cluster.local:5432"
|
||||
GITEA__database__NAME: "gitea"
|
||||
@@ -17,6 +20,12 @@ data:
|
||||
|
||||
GITEA__mailer__ENABLED: "false"
|
||||
|
||||
# No code/issue search needed: bleve reindexes the whole issue index on
|
||||
# every pod restart (cron.rebuild_issue_indexer RUN_AT_START) and hammers
|
||||
# the rotational disk for an hour. "db" serves issue search from postgres.
|
||||
GITEA__indexer__ISSUE_INDEXER_TYPE: "db"
|
||||
GITEA__indexer__REPO_INDEXER_ENABLED: "false"
|
||||
|
||||
GITEA__log__logger.access.MODE: "console, file"
|
||||
USER_UID: "1000"
|
||||
USER_GID: "1000"
|
||||
@@ -24,6 +24,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: gitea
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -31,7 +33,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: gitea
|
||||
image: gitea/gitea:1.27.3
|
||||
image: gitea/gitea:28.0.0
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: gitea-config
|
||||
@@ -47,10 +49,10 @@ spec:
|
||||
mountPath: /data
|
||||
resources:
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
cpu: "300m"
|
||||
memory: "320Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
memory: "1.5Gi"
|
||||
memory: "1Gi"
|
||||
cpu: "1300m"
|
||||
volumes:
|
||||
- name: gitea-data
|
||||
|
||||
@@ -7,19 +7,13 @@ spec:
|
||||
entryPoints:
|
||||
- websecure
|
||||
routes:
|
||||
- match: Host(`gitea.forust.xyz`)
|
||||
- match: Host(`gitea.forust.xyz`) || Host(`git.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: gitea-service
|
||||
port: 3000
|
||||
- match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: gitea-service
|
||||
port: 3000
|
||||
@@ -35,7 +29,7 @@ spec:
|
||||
entryPoints:
|
||||
- websecure
|
||||
routes:
|
||||
- match: Host(`gitea.workstation.internal`) || Host(`gitea.gigaforust.internal`)
|
||||
- match: (Host(`gitea.workstation.internal`) || Host(`gitea.gigaforust.internal`)) || (Host(`git.workstation.internal`) || Host(`git.gigaforust.internal`))
|
||||
kind: Rule
|
||||
services:
|
||||
- name: gitea-service
|
||||
|
||||
@@ -177,8 +177,8 @@ data:
|
||||
# url: https://gitssh.forust.xyz
|
||||
# - title: gcr.forust.xyz
|
||||
# url: https://gcr.forust.xyz/v2/
|
||||
- title: gitea.forust.xyz
|
||||
url: https://gitea.forust.xyz
|
||||
- title: git.forust.xyz
|
||||
url: https://git.forust.xyz
|
||||
- title: nextcloud.forust.xyz
|
||||
url: https://nextcloud.forust.xyz
|
||||
- title: mc.forust.xyz
|
||||
|
||||
@@ -20,6 +20,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: glance
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -57,10 +59,10 @@ spec:
|
||||
resources:
|
||||
requests:
|
||||
cpu: "50m"
|
||||
memory: "64Mi"
|
||||
memory: "32Mi"
|
||||
limits:
|
||||
cpu: "200m"
|
||||
memory: "256Mi"
|
||||
memory: "128Mi"
|
||||
volumes:
|
||||
- name: glance-config
|
||||
configMap:
|
||||
|
||||
@@ -23,17 +23,11 @@ spec:
|
||||
port: 8080
|
||||
- match: Host(`hs.forust.xyz`) && PathPrefix(`/admin`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: headscale-ui-external
|
||||
port: 80
|
||||
- match: Host(`hs.forust.xyz`) && PathPrefix(`/metrics`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: headscale-server-external
|
||||
port: 9090
|
||||
@@ -53,8 +47,6 @@ spec:
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: headplane-prefix
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: headplane-external
|
||||
port: 3000
|
||||
|
||||
@@ -3,7 +3,7 @@ services:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile.forust
|
||||
image: gcr.forust.xyz/forust/forust-homepage:latest
|
||||
image: gcr.forust.xyz/forust/forust-homepage:prod
|
||||
pull_policy: build
|
||||
# ports:
|
||||
# - "8085:80"
|
||||
@@ -35,7 +35,7 @@ services:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile.xdfnx
|
||||
image: gcr.forust.xyz/forust/xdfnx-homepage:latest
|
||||
image: gcr.forust.xyz/forust/xdfnx-homepage:prod
|
||||
pull_policy: build
|
||||
restart: unless-stopped
|
||||
# ports:
|
||||
|
||||
@@ -174,7 +174,7 @@
|
||||
<h2>./projects</h2>
|
||||
<ul class="repo-list">
|
||||
<li>
|
||||
<a href="https://gitea.forust.xyz/forust/gosleep" target="_blank">forust/gosleep</a>
|
||||
<a href="https://git.forust.xyz/forust/gosleep" target="_blank">forust/gosleep</a>
|
||||
<span class="comment">// linux sleep timer written in rust (originally in go)</span>
|
||||
</li>
|
||||
</ul>
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
# Gateway API PoC for homepages. Lives next to the TLS secrets so
|
||||
# certificateRefs stay same-namespace and no ReferenceGrant is needed.
|
||||
# Listener ports must match the Traefik entryPoints (80/443),
|
||||
# otherwise Traefik marks the listener Invalid.
|
||||
# Local .internal hosts are deliberately left on IngressRoute,
|
||||
# only prod is migrated here.
|
||||
apiVersion: gateway.networking.k8s.io/v1
|
||||
kind: Gateway
|
||||
metadata:
|
||||
name: homepages
|
||||
namespace: homepages
|
||||
spec:
|
||||
gatewayClassName: traefik
|
||||
listeners:
|
||||
- name: http
|
||||
protocol: HTTP
|
||||
port: 80
|
||||
allowedRoutes:
|
||||
namespaces:
|
||||
from: Same
|
||||
- name: https
|
||||
protocol: HTTPS
|
||||
port: 443
|
||||
tls:
|
||||
mode: Terminate
|
||||
certificateRefs:
|
||||
- name: forust-homepage-prod-tls
|
||||
- name: xdfnx-homepage-prod-tls
|
||||
allowedRoutes:
|
||||
namespaces:
|
||||
from: Same
|
||||
@@ -20,6 +20,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: forust-homepage
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -27,16 +29,22 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: forust-homepage
|
||||
image: gcr.forust.xyz/forust/forust-homepage:latest
|
||||
imagePullPolicy: Always
|
||||
image: gcr.forust.xyz/forust/forust-homepage:prod
|
||||
ports:
|
||||
- containerPort: 80
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /
|
||||
port: 80
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 2
|
||||
failureThreshold: 3
|
||||
resources:
|
||||
requests:
|
||||
memory: "10Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "20m"
|
||||
limits:
|
||||
memory: "100Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "50m"
|
||||
---
|
||||
apiVersion: v1
|
||||
@@ -61,6 +69,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: xdfnx-homepage
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -68,14 +78,20 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: xdfnx-homepage
|
||||
image: gcr.forust.xyz/forust/xdfnx-homepage:latest
|
||||
imagePullPolicy: Always
|
||||
image: gcr.forust.xyz/forust/xdfnx-homepage:prod
|
||||
ports:
|
||||
- containerPort: 80
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /
|
||||
port: 80
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 2
|
||||
failureThreshold: 3
|
||||
resources:
|
||||
requests:
|
||||
memory: "10Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "20m"
|
||||
limits:
|
||||
memory: "100Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "50m"
|
||||
@@ -0,0 +1,69 @@
|
||||
# PoC: homepages prod hosts via Gateway API.
|
||||
# Runs alongside k8s/ingress.yaml - delete the prod IngressRoutes only after verification.
|
||||
# There are no local .internal hosts here, they stay on the local IngressRoute.
|
||||
# No per-route security middlewares: L3 enforcement moved to the host
|
||||
# firewall bouncer, so HTTPRoutes stay clean.
|
||||
---
|
||||
apiVersion: gateway.networking.k8s.io/v1
|
||||
kind: HTTPRoute
|
||||
metadata:
|
||||
name: homepages-http-redirect
|
||||
namespace: homepages
|
||||
spec:
|
||||
parentRefs:
|
||||
- name: homepages
|
||||
kind: Gateway
|
||||
sectionName: http
|
||||
hostnames:
|
||||
- forust.xyz
|
||||
- www.forust.xyz
|
||||
- xdfnx.cfd
|
||||
rules:
|
||||
- filters:
|
||||
- type: RequestRedirect
|
||||
requestRedirect:
|
||||
scheme: https
|
||||
statusCode: 301
|
||||
---
|
||||
apiVersion: gateway.networking.k8s.io/v1
|
||||
kind: HTTPRoute
|
||||
metadata:
|
||||
name: forust-homepage-https
|
||||
namespace: homepages
|
||||
spec:
|
||||
parentRefs:
|
||||
- name: homepages
|
||||
kind: Gateway
|
||||
sectionName: https
|
||||
hostnames:
|
||||
- forust.xyz
|
||||
- www.forust.xyz
|
||||
rules:
|
||||
- matches:
|
||||
- path:
|
||||
type: PathPrefix
|
||||
value: /
|
||||
backendRefs:
|
||||
- name: forust-homepage-service
|
||||
port: 80
|
||||
---
|
||||
apiVersion: gateway.networking.k8s.io/v1
|
||||
kind: HTTPRoute
|
||||
metadata:
|
||||
name: xdfnx-homepage-https
|
||||
namespace: homepages
|
||||
spec:
|
||||
parentRefs:
|
||||
- name: homepages
|
||||
kind: Gateway
|
||||
sectionName: https
|
||||
hostnames:
|
||||
- xdfnx.cfd
|
||||
rules:
|
||||
- matches:
|
||||
- path:
|
||||
type: PathPrefix
|
||||
value: /
|
||||
backendRefs:
|
||||
- name: xdfnx-homepage-service
|
||||
port: 80
|
||||
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`forust.xyz`) || Host(`www.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
priority: 10
|
||||
services:
|
||||
- name: forust-homepage-service
|
||||
@@ -49,9 +46,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`xdfnx.cfd`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: xdfnx-homepage-service
|
||||
port: 80
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
# You can find documentation for all the supported env variables at https://docs.immich.app/install/environment-variables
|
||||
|
||||
# The location where your uploaded files are stored. The k8s manifests bind
|
||||
# mount /mnt/immich/library, which is the sdc9 partition - the same place, so
|
||||
# the two deployment paths look at one library.
|
||||
UPLOAD_LOCATION=/mnt/immich/library
|
||||
|
||||
# The location where your database files are stored. Network shares are not supported for the database
|
||||
DB_DATA_LOCATION=./postgres
|
||||
|
||||
# To set a timezone, uncomment the next line and change Etc/UTC to a TZ identifier from this list: https://en.wikipedia.org/wiki/List_of_tz_database_time_zones#List
|
||||
# TZ=Etc/UTC
|
||||
|
||||
# The Immich version to use. You can pin this to a specific version like "v2.1.0"
|
||||
IMMICH_VERSION=v3
|
||||
|
||||
# Connection secret for postgres. You should change it to a random password
|
||||
# Please use only the characters `A-Za-z0-9`, without special characters or spaces
|
||||
DB_PASSWORD=postgres
|
||||
|
||||
# The values below this line do not need to be changed
|
||||
###################################################################################
|
||||
DB_USERNAME=postgres
|
||||
DB_DATABASE_NAME=immich
|
||||
@@ -0,0 +1,63 @@
|
||||
name: immich
|
||||
|
||||
services:
|
||||
immich-server:
|
||||
container_name: immich_server
|
||||
image: ghcr.io/immich-app/immich-server:v3
|
||||
volumes:
|
||||
- ${UPLOAD_LOCATION}:/data
|
||||
- /etc/localtime:/etc/localtime:ro
|
||||
env_file:
|
||||
- .env
|
||||
ports:
|
||||
- "2283:2283"
|
||||
depends_on:
|
||||
- redis
|
||||
- database
|
||||
restart: always
|
||||
healthcheck:
|
||||
disable: false
|
||||
|
||||
immich-machine-learning:
|
||||
container_name: immich_machine_learning
|
||||
# For hardware acceleration, add one of -[armnn, cuda, rocm, openvino, rknn] to the image tag.
|
||||
# Example tag: ${IMMICH_VERSION:-release}-cuda
|
||||
image: ghcr.io/immich-app/immich-machine-learning:${IMMICH_VERSION:-release}
|
||||
# extends: # uncomment this section for hardware acceleration - see https://docs.immich.app/features/ml-hardware-acceleration
|
||||
# file: hwaccel.ml.yml
|
||||
# service: cpu # set to one of [armnn, cuda, rocm, openvino, openvino-wsl, rknn] for accelerated inference - use the `-wsl` version for WSL2 where applicable
|
||||
volumes:
|
||||
- model-cache:/cache
|
||||
env_file:
|
||||
- .env
|
||||
restart: always
|
||||
healthcheck:
|
||||
disable: false
|
||||
|
||||
redis:
|
||||
container_name: immich_redis
|
||||
image: docker.io/valkey/valkey:9@sha256:418652cfb58ef879d4978c33553735d7147016032d5aefaa14c828e611eb9dfd
|
||||
healthcheck:
|
||||
test: redis-cli ping | grep -q PONG || exit 1
|
||||
restart: always
|
||||
|
||||
database:
|
||||
container_name: immich_postgres
|
||||
image: ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0@sha256:1a078b237c1d9b420b0ee59147386b4aa60d3a07a8e6a402fc84a57e41b043a4
|
||||
environment:
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD}
|
||||
POSTGRES_USER: ${DB_USERNAME}
|
||||
POSTGRES_DB: ${DB_DATABASE_NAME}
|
||||
POSTGRES_INITDB_ARGS: "--data-checksums"
|
||||
# Uncomment the DB_STORAGE_TYPE: 'HDD' var if your database isn't stored on SSDs
|
||||
# DB_STORAGE_TYPE: 'HDD'
|
||||
volumes:
|
||||
# Do not edit the next line. If you want to change the database storage location on your system, edit the value of DB_DATA_LOCATION in the .env file
|
||||
- ${DB_DATA_LOCATION}:/var/lib/postgresql/data
|
||||
shm_size: 128mb
|
||||
restart: always
|
||||
healthcheck:
|
||||
disable: false
|
||||
|
||||
volumes:
|
||||
model-cache:
|
||||
File renamed without changes.
@@ -1,8 +1,21 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: immich-prod-tls
|
||||
namespace: immich
|
||||
spec:
|
||||
secretName: immich-prod-tls
|
||||
dnsNames:
|
||||
- immich.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: userbot
|
||||
namespace: immich
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
@@ -0,0 +1,24 @@
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: immich-config
|
||||
namespace: immich
|
||||
data:
|
||||
TZ: "Europe/Bratislava"
|
||||
|
||||
# The database in this namespace, not the shared one in the database
|
||||
# namespace: v3 needs VectorChord, and only the dedicated image carries it.
|
||||
DB_HOSTNAME: "immich-postgres"
|
||||
DB_PORT: "5432"
|
||||
DB_USERNAME: "immich"
|
||||
DB_DATABASE_NAME: "immich"
|
||||
DB_SSL_MODE: "disable"
|
||||
DB_VECTOR_EXTENSION: "vectorchord"
|
||||
|
||||
REDIS_HOSTNAME: "immich-valkey"
|
||||
REDIS_PORT: "6379"
|
||||
|
||||
# Traefik is the only client of the server, and it is a pod: the address immich
|
||||
# sees is inside the node's pod CIDR. Without this the server does not trust
|
||||
# X-Forwarded-For and every request looks like it came from Traefik itself.
|
||||
IMMICH_TRUSTED_PROXIES: "10.244.0.0/24"
|
||||
@@ -0,0 +1,95 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: immich-service
|
||||
namespace: immich
|
||||
spec:
|
||||
selector:
|
||||
app: immich
|
||||
ports:
|
||||
- name: http
|
||||
port: 2283
|
||||
targetPort: 2283
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: immich-deployment
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich
|
||||
spec:
|
||||
replicas: 2
|
||||
selector:
|
||||
matchLabels:
|
||||
app: immich
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: immich
|
||||
spec:
|
||||
containers:
|
||||
- name: immich
|
||||
image: ghcr.io/immich-app/immich-server:v3
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: immich-config
|
||||
- secretRef:
|
||||
name: immich-secrets
|
||||
ports:
|
||||
- name: http
|
||||
containerPort: 2283
|
||||
volumeMounts:
|
||||
- name: immich-data
|
||||
mountPath: /data
|
||||
# The first boot runs migrations and warms the transcoder, which can
|
||||
# take minutes, so liveness has to wait on the startup probe.
|
||||
startupProbe:
|
||||
httpGet:
|
||||
path: /api/server/ping
|
||||
port: http
|
||||
failureThreshold: 60
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /api/server/ping
|
||||
port: http
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /api/server/ping
|
||||
port: http
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 30
|
||||
timeoutSeconds: 5
|
||||
# Only the request is scheduled against, and the node is already
|
||||
# oversubscribed (5.58 of 6 cores requested) while actually running
|
||||
# at about 1.5. So the request states what this sits at while idle -
|
||||
# tens of millicores - and the limit leaves room for the burst that
|
||||
# matters: thumbnails, transcodes and metadata extraction.
|
||||
#
|
||||
# The limit used to be 2Gi, but the server OOMKilled on boot while
|
||||
# chewing through a backlog of unprocessed assets (API +Workers in
|
||||
# one container spike well past idle).
|
||||
resources:
|
||||
requests:
|
||||
cpu: "100m"
|
||||
memory: "512Mi"
|
||||
limits:
|
||||
cpu: "1500m"
|
||||
memory: "4Gi"
|
||||
volumes:
|
||||
- name: immich-data
|
||||
# The library lives on the node's own disk, not in a PVC. A PVC here
|
||||
# meant declaring a size up front for data that does not exist yet,
|
||||
# on a provisioner that cannot grow it, and the only copy of the
|
||||
# photos was one `kubectl delete namespace` away.
|
||||
#
|
||||
# Directory, not DirectoryOrCreate, on purpose: if sdc9 is not
|
||||
# mounted, this must fail loudly instead of quietly writing the
|
||||
# library onto the root filesystem.
|
||||
hostPath:
|
||||
path: /mnt/immich/library
|
||||
type: Directory
|
||||
@@ -0,0 +1,33 @@
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
metadata:
|
||||
name: immich-prod
|
||||
namespace: immich
|
||||
spec:
|
||||
entryPoints:
|
||||
- websecure
|
||||
routes:
|
||||
- match: Host(`immich.forust.xyz`)
|
||||
kind: Rule
|
||||
services:
|
||||
- name: immich-service
|
||||
port: 2283
|
||||
tls:
|
||||
secretName: immich-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
metadata:
|
||||
name: immich-local
|
||||
namespace: immich
|
||||
spec:
|
||||
entryPoints:
|
||||
- websecure
|
||||
routes:
|
||||
- match: Host(`immich.workstation.internal`) || Host(`immich.gigaforust.internal`)
|
||||
kind: Rule
|
||||
services:
|
||||
- name: immich-service
|
||||
port: 2283
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -0,0 +1,94 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: immich-machine-learning
|
||||
namespace: immich
|
||||
spec:
|
||||
selector:
|
||||
app: immich-machine-learning
|
||||
ports:
|
||||
- name: http
|
||||
port: 3003
|
||||
targetPort: 3003
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: immich-machine-learning-deployment
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-machine-learning
|
||||
spec:
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app: immich-machine-learning
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: immich-machine-learning
|
||||
spec:
|
||||
containers:
|
||||
- name: immich-machine-learning
|
||||
image: ghcr.io/immich-app/immich-machine-learning:v3
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: immich-config
|
||||
- secretRef:
|
||||
name: immich-secrets
|
||||
ports:
|
||||
- name: http
|
||||
containerPort: 3003
|
||||
volumeMounts:
|
||||
- name: model-cache
|
||||
mountPath: /cache
|
||||
# The first request pulls a model over the internet, so a cold start
|
||||
# is slower than a container start.
|
||||
startupProbe:
|
||||
httpGet:
|
||||
path: /ping
|
||||
port: http
|
||||
failureThreshold: 60
|
||||
periodSeconds: 5
|
||||
timeoutSeconds: 5
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /ping
|
||||
port: http
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /ping
|
||||
port: http
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 30
|
||||
timeoutSeconds: 5
|
||||
# Same reasoning as the server: the request covers the idle cost
|
||||
# only, because the node has no spare cores to schedule against.
|
||||
# Recognition is the burst - a busy import wants both cores.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "100m"
|
||||
memory: "1Gi"
|
||||
limits:
|
||||
cpu: "2000m"
|
||||
memory: "3Gi"
|
||||
volumes:
|
||||
- name: model-cache
|
||||
persistentVolumeClaim:
|
||||
claimName: immich-model-cache-pvc
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: immich-model-cache-pvc
|
||||
namespace: immich
|
||||
spec:
|
||||
accessModes:
|
||||
- ReadWriteOnce
|
||||
resources:
|
||||
requests:
|
||||
storage: 2Gi
|
||||
@@ -0,0 +1,5 @@
|
||||
# yaml-language-server: $schema=kubernetes
|
||||
apiVersion: v1
|
||||
kind: Namespace
|
||||
metadata:
|
||||
name: immich
|
||||
@@ -0,0 +1,145 @@
|
||||
# Immich's own database, separate from the shared postgres in the database
|
||||
# namespace. It has to be separate: v3 checks the VectorChord version at startup
|
||||
# and refuses to boot without it, VectorChord needs its .so in
|
||||
# shared_preload_libraries, and that can only be read when postmaster starts.
|
||||
# So the shared instance would have to be rebuilt on a custom image carrying
|
||||
# vchord and restarted - for every consumer of it (authentik, gitea, netbox,
|
||||
# netronome, penpot, statuspage). Not worth it for one photo library.
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: immich-postgres
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-postgres
|
||||
spec:
|
||||
selector:
|
||||
app: immich-postgres
|
||||
ports:
|
||||
- name: postgres
|
||||
port: 5432
|
||||
targetPort: postgres
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
name: immich-postgres
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-postgres
|
||||
spec:
|
||||
serviceName: immich-postgres
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app: immich-postgres
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: immich-postgres
|
||||
spec:
|
||||
containers:
|
||||
- name: postgres
|
||||
# v3.x expects vchord for its vector work and vectors (pgvecto.rs)
|
||||
# for some index types. This image ships both and preloads them, plus
|
||||
# its own shared_buffers and wal settings, through
|
||||
# /etc/postgresql/postgresql.conf - which its entrypoint reaches via
|
||||
# `postgres -c config_file=...` in the image CMD.
|
||||
#
|
||||
# So there is deliberately no `command:` here. Overriding it replaces
|
||||
# that config_file, and it also loses the step where the entrypoint
|
||||
# drops from root to the postgres user: postmaster then starts as
|
||||
# root and refuses to run.
|
||||
image: ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0@sha256:1a078b237c1d9b420b0ee59147386b4aa60d3a07a8e6a402fc84a57e41b043a4
|
||||
env:
|
||||
- name: POSTGRES_USER
|
||||
value: immich
|
||||
- name: POSTGRES_DB
|
||||
value: immich
|
||||
- name: POSTGRES_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: immich-secrets
|
||||
key: DB_PASSWORD
|
||||
# Only read when the data directory is empty, so the checksums are
|
||||
# decided here and never again.
|
||||
- name: POSTGRES_INITDB_ARGS
|
||||
value: --data-checksums
|
||||
# The postgres-data volume lives on sdc, which is rotational. The
|
||||
# SSD template is the default; HDD only changes the planner costs
|
||||
# (effective_io_concurrency, random_page_cost), nothing structural.
|
||||
- name: DB_STORAGE_TYPE
|
||||
value: HDD
|
||||
- name: TZ
|
||||
valueFrom:
|
||||
configMapKeyRef:
|
||||
name: immich-config
|
||||
key: TZ
|
||||
ports:
|
||||
- name: postgres
|
||||
containerPort: 5432
|
||||
volumeMounts:
|
||||
- name: postgres-data
|
||||
mountPath: /var/lib/postgresql/data
|
||||
# The upstream compose file asks docker for 128mb of shm. Kubernetes
|
||||
# gives every container 64mb, which is not what postmaster expects
|
||||
# for parallel query workers and the WAL writer.
|
||||
- name: shm
|
||||
mountPath: /dev/shm
|
||||
# Probes use a generous timeout on purpose: the data lives on a
|
||||
# rotational disk on a loaded single node, and pg_isready can take
|
||||
# seconds during WAL recovery. A 1s timeout kills the container
|
||||
# mid-recovery and restarts the spiral.
|
||||
#
|
||||
# Budgets are sized for HDD stalls, not for a healthy disk: fsync of
|
||||
# a single file was observed taking 70s under node IO pressure, so
|
||||
# the startup budget is 15 minutes and liveness tolerates 5 minutes
|
||||
# of unresponsiveness. Killing a stalled-but-healthy postmaster only
|
||||
# buys another full WAL replay, which is more IO, not less.
|
||||
startupProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", "pg_isready -U immich -d immich"]
|
||||
failureThreshold: 180
|
||||
periodSeconds: 5
|
||||
timeoutSeconds: 5
|
||||
readinessProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", "pg_isready -U immich -d immich"]
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
livenessProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", "pg_isready -U immich -d immich"]
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 60
|
||||
timeoutSeconds: 10
|
||||
failureThreshold: 5
|
||||
# The image template sets shared_buffers to 512MB, and the vchord and
|
||||
# vectors workers are Rust binaries with a real RSS footprint on top
|
||||
# of postmaster, checkpointer and friends. 1Gi was enough to start
|
||||
# the server but the vectors worker kept dying in it, so the limit
|
||||
# sits at 2Gi. The request stays at the idle cost.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "50m"
|
||||
memory: "256Mi"
|
||||
limits:
|
||||
cpu: "1000m"
|
||||
memory: "2Gi"
|
||||
volumes:
|
||||
- name: shm
|
||||
emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 128Mi
|
||||
volumeClaimTemplates:
|
||||
- metadata:
|
||||
name: postgres-data
|
||||
spec:
|
||||
accessModes: ["ReadWriteOnce"]
|
||||
# Retain: this is the metadata for a library that only exists in one
|
||||
# place, and local-path cannot expand a bound volume, so this size has
|
||||
# to hold until the library is rebuilt or dumped elsewhere.
|
||||
storageClassName: local-path-retain
|
||||
resources:
|
||||
requests:
|
||||
storage: 32Gi
|
||||
@@ -0,0 +1,13 @@
|
||||
apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
name: immich-secrets
|
||||
namespace: immich
|
||||
type: Opaque
|
||||
stringData:
|
||||
# Creates the immich superuser in this namespace's own postgres on first
|
||||
# boot, and is the same value the server connects with. Nothing outside the
|
||||
# immich namespace needs it. Letters and digits only: immich reads this into
|
||||
# a connection string.
|
||||
DB_PASSWORD: "changeme"
|
||||
REDIS_PASSWORD: "changeme"
|
||||
@@ -0,0 +1,84 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: immich-valkey
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-valkey
|
||||
spec:
|
||||
clusterIP: None
|
||||
selector:
|
||||
app: immich-valkey
|
||||
ports:
|
||||
- name: valkey
|
||||
port: 6379
|
||||
targetPort: valkey
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
name: immich-valkey
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-valkey
|
||||
spec:
|
||||
serviceName: immich-valkey
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app: immich-valkey
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: immich-valkey
|
||||
spec:
|
||||
containers:
|
||||
- name: valkey
|
||||
image: docker.io/valkey/valkey:9.1.2-alpine
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- valkey-server --appendonly yes --save 30 1 --loglevel warning --requirepass "$REDIS_PASSWORD"
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: immich-secrets
|
||||
ports:
|
||||
- name: valkey
|
||||
containerPort: 6379
|
||||
volumeMounts:
|
||||
- name: valkey-data
|
||||
mountPath: /data
|
||||
# Same reasoning as postgres: 1s probe timeouts flap on a loaded
|
||||
# single node with rotational storage.
|
||||
startupProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
|
||||
failureThreshold: 20
|
||||
periodSeconds: 5
|
||||
timeoutSeconds: 5
|
||||
readinessProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
livenessProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
|
||||
initialDelaySeconds: 20
|
||||
periodSeconds: 20
|
||||
timeoutSeconds: 5
|
||||
resources:
|
||||
requests:
|
||||
cpu: "25m"
|
||||
memory: "64Mi"
|
||||
limits:
|
||||
cpu: "250m"
|
||||
memory: "256Mi"
|
||||
volumeClaimTemplates:
|
||||
- metadata:
|
||||
name: valkey-data
|
||||
spec:
|
||||
accessModes: ["ReadWriteOnce"]
|
||||
resources:
|
||||
requests:
|
||||
storage: 1Gi
|
||||
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`status.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: kener-service
|
||||
port: 3000
|
||||
|
||||
@@ -20,6 +20,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: kener
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
|
||||
@@ -7,18 +7,32 @@
|
||||
|
||||
controller:
|
||||
type: daemonset
|
||||
|
||||
# config-reloader sidecar: p95 33M, max 43M. The chart keeps it at the top level,
|
||||
# not under `alloy:`.
|
||||
configReloader:
|
||||
resources:
|
||||
requests:
|
||||
memory: "128Mi"
|
||||
cpu: "50m"
|
||||
memory: "32Mi"
|
||||
cpu: "10m"
|
||||
limits:
|
||||
memory: "512Mi"
|
||||
cpu: "500m"
|
||||
memory: "128Mi"
|
||||
|
||||
image:
|
||||
tag: "v1.19.2"
|
||||
|
||||
alloy:
|
||||
# p95 275M, max 287M. Alloy tails every pod log and ships it to Loki, so it sits
|
||||
# on the same IronWolf read path the node is I/O bound on. Request is set at p95.
|
||||
# The chart key is `alloy.resources`. `controller.resources` is ignored silently,
|
||||
# which is why this pod shipped with no limits at all.
|
||||
resources:
|
||||
requests:
|
||||
memory: "288Mi"
|
||||
cpu: "50m"
|
||||
limits:
|
||||
memory: "512Mi"
|
||||
|
||||
configMap:
|
||||
create: true
|
||||
content: |
|
||||
|
||||
@@ -39,6 +39,16 @@ loki:
|
||||
local:
|
||||
directory: /var/loki/rules
|
||||
|
||||
# p95 84M, max 85M for the rules sidecar that shares the singleBinary pod.
|
||||
# The chart exposes it as `sidecar.resources`, shared with any other sidecar.
|
||||
sidecar:
|
||||
resources:
|
||||
requests:
|
||||
memory: "96Mi"
|
||||
cpu: "10m"
|
||||
limits:
|
||||
memory: "192Mi"
|
||||
|
||||
singleBinary:
|
||||
replicas: 1
|
||||
persistence:
|
||||
@@ -47,10 +57,10 @@ singleBinary:
|
||||
storageClass: local-path-retain
|
||||
resources:
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
memory: "256Mi"
|
||||
cpu: "200m"
|
||||
limits:
|
||||
memory: "2Gi"
|
||||
memory: "1Gi"
|
||||
cpu: "1000m"
|
||||
|
||||
# Zeroed: unused in SingleBinary mode (chart validation requires it).
|
||||
@@ -63,12 +73,25 @@ backend:
|
||||
|
||||
gateway:
|
||||
replicas: 1
|
||||
# Single node: chart default is required podAntiAffinity on hostname +
|
||||
# RollingUpdate 25%/25% (effective maxUnavailable=0 at replicas=1).
|
||||
# That deadlocks the rollout: the new pod stays Unschedulable while the
|
||||
# old one lives, and the old one never leaves while the new one is not
|
||||
# Ready. Null clears the default (an empty map would deep-merge with it
|
||||
# and keep the required rule); maxUnavailable=1 allows a brief gateway
|
||||
# outage during rollouts instead of a stuck deploy.
|
||||
affinity: null
|
||||
deploymentStrategy:
|
||||
type: RollingUpdate
|
||||
rollingUpdate:
|
||||
maxSurge: 1
|
||||
maxUnavailable: 1
|
||||
resources:
|
||||
requests:
|
||||
memory: "64Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "50m"
|
||||
limits:
|
||||
memory: "256Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "300m"
|
||||
|
||||
monitoring:
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
metube:
|
||||
image: ghcr.io/alexta69/metube:2026.09.25
|
||||
image: ghcr.io/alexta69/metube:2026.09.29
|
||||
container_name: metube
|
||||
restart: unless-stopped
|
||||
# ports:
|
||||
|
||||
@@ -20,6 +20,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: metube
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -27,7 +29,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: metube
|
||||
image: ghcr.io/alexta69/metube:2026.09.25
|
||||
image: ghcr.io/alexta69/metube:2026.09.29
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: metube-config
|
||||
@@ -36,12 +38,13 @@ spec:
|
||||
volumeMounts:
|
||||
- name: downloads
|
||||
mountPath: /downloads
|
||||
# p95 72M, max 80M over 7 days. Was 600Mi/2Gi.
|
||||
resources:
|
||||
requests:
|
||||
memory: "600Mi"
|
||||
memory: "96Mi"
|
||||
cpu: "400m"
|
||||
limits:
|
||||
memory: "2Gi"
|
||||
memory: "384Mi"
|
||||
cpu: "1700m"
|
||||
volumes:
|
||||
- name: downloads
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
n8n:
|
||||
image: docker.n8n.io/n8nio/n8n:2.41.3
|
||||
image: docker.n8n.io/n8nio/n8n:2.42.3
|
||||
container_name: n8n
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
|
||||
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`n8n.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: n8n-service
|
||||
port: 5678
|
||||
|
||||
+3
-1
@@ -20,6 +20,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: n8n
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -27,7 +29,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: n8n
|
||||
image: docker.n8n.io/n8nio/n8n:2.41.3
|
||||
image: docker.n8n.io/n8nio/n8n:2.42.3
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: n8n-config
|
||||
|
||||
@@ -2,7 +2,7 @@ name: netbird
|
||||
|
||||
services:
|
||||
netbird-server:
|
||||
image: netbirdio/netbird-server:0.79.0
|
||||
image: netbirdio/netbird-server:0.80.0
|
||||
container_name: netbird-server
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
@@ -87,7 +87,7 @@ services:
|
||||
- proxy
|
||||
|
||||
dashboard:
|
||||
image: netbirdio/dashboard:v2.90.10
|
||||
image: netbirdio/dashboard:v2.94.0
|
||||
container_name: netbird-dashboard
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
|
||||
File renamed without changes.
@@ -7,12 +7,18 @@ spec:
|
||||
entryPoints:
|
||||
- websecure
|
||||
routes:
|
||||
# NO crowdsec-bouncer on the API routes. These are the mesh client's own
|
||||
# endpoints: gRPC-gateway management calls plus signal/relay long-polling,
|
||||
# authenticated by NetBird's token rather than by a login form. A ban here
|
||||
# is self-defeating - the client needs the mesh to reach anything else, so
|
||||
# CrowdSec banning it locks the peer out of the network it needs to
|
||||
# function. It also backfires: a banned peer keeps retrying, every retry
|
||||
# is another 403, and LePresidente/http-generic-403-bf turns five 403s in
|
||||
# ten seconds into a 4h ban, so one 403 loop kept re-arming the ban.
|
||||
# netbird-local below has always been exempt; this makes prod match.
|
||||
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`))
|
||||
kind: Rule
|
||||
priority: 100
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: netbird-server-service
|
||||
port: 80
|
||||
@@ -20,18 +26,12 @@ spec:
|
||||
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`))
|
||||
kind: Rule
|
||||
priority: 100
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: netbird-server-service
|
||||
port: 80
|
||||
- match: Host(`nb.forust.xyz`)
|
||||
kind: Rule
|
||||
priority: 1
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: netbird-dashboard-service
|
||||
port: 80
|
||||
|
||||
@@ -39,6 +39,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: netbird-server
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -46,7 +48,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: netbird-server
|
||||
image: netbirdio/netbird-server:0.79.0
|
||||
image: netbirdio/netbird-server:0.80.0
|
||||
command: ["/bin/sh", "/opt/netbird/entrypoint.sh", "--config", "/run/netbird/config.yaml"]
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
@@ -88,12 +90,13 @@ spec:
|
||||
periodSeconds: 30
|
||||
timeoutSeconds: 5
|
||||
failureThreshold: 5
|
||||
# p95 97M, max 102M over 7 days. Was 256Mi/1Gi.
|
||||
resources:
|
||||
requests:
|
||||
memory: "256Mi"
|
||||
cpu: "250m"
|
||||
memory: "128Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
memory: "384Mi"
|
||||
cpu: "1000m"
|
||||
volumes:
|
||||
- name: netbird-data
|
||||
@@ -130,6 +133,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: netbird-dashboard
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -137,7 +142,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: dashboard
|
||||
image: netbirdio/dashboard:v2.90.10
|
||||
image: netbirdio/dashboard:v2.94.0
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: netbird-config
|
||||
@@ -162,10 +167,10 @@ spec:
|
||||
failureThreshold: 5
|
||||
resources:
|
||||
requests:
|
||||
memory: "64Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "50m"
|
||||
limits:
|
||||
memory: "256Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "300m"
|
||||
---
|
||||
apiVersion: v1
|
||||
|
||||
@@ -1,48 +1,48 @@
|
||||
import os
|
||||
|
||||
|
||||
def _csv(name, default=""):
|
||||
return [item.strip() for item in os.environ.get(name, default).split(",") if item.strip()]
|
||||
def _csv(name, default=''):
|
||||
return [item.strip() for item in os.environ.get(name, default).split(',') if item.strip()]
|
||||
|
||||
|
||||
ALLOWED_HOSTS = _csv("ALLOWED_HOSTS", "localhost,127.0.0.1,[::1]")
|
||||
CSRF_TRUSTED_ORIGINS = _csv("CSRF_TRUSTED_ORIGINS")
|
||||
ALLOWED_HOSTS = _csv('ALLOWED_HOSTS', 'localhost,127.0.0.1,[::1]')
|
||||
CSRF_TRUSTED_ORIGINS = _csv('CSRF_TRUSTED_ORIGINS')
|
||||
USE_X_FORWARDED_HOST = True
|
||||
SECURE_PROXY_SSL_HEADER = ("HTTP_X_FORWARDED_PROTO", "https")
|
||||
SECURE_PROXY_SSL_HEADER = ('HTTP_X_FORWARDED_PROTO', 'https')
|
||||
|
||||
DATABASES = {
|
||||
"default": {
|
||||
"NAME": os.environ["DB_NAME"],
|
||||
"USER": os.environ["DB_USER"],
|
||||
"PASSWORD": os.environ["DB_PASSWORD"],
|
||||
"HOST": os.environ["DB_HOST"],
|
||||
"PORT": os.environ.get("DB_PORT", "5432"),
|
||||
"OPTIONS": {"sslmode": os.environ.get("DB_SSLMODE", "disable")},
|
||||
"CONN_MAX_AGE": int(os.environ.get("DB_CONN_MAX_AGE", "300")),
|
||||
'default': {
|
||||
'NAME': os.environ['DB_NAME'],
|
||||
'USER': os.environ['DB_USER'],
|
||||
'PASSWORD': os.environ['DB_PASSWORD'],
|
||||
'HOST': os.environ['DB_HOST'],
|
||||
'PORT': os.environ.get('DB_PORT', '5432'),
|
||||
'OPTIONS': {'sslmode': os.environ.get('DB_SSLMODE', 'disable')},
|
||||
'CONN_MAX_AGE': int(os.environ.get('DB_CONN_MAX_AGE', '300')),
|
||||
}
|
||||
}
|
||||
|
||||
REDIS = {
|
||||
"tasks": {
|
||||
"HOST": os.environ["REDIS_HOST"],
|
||||
"PORT": int(os.environ.get("REDIS_PORT", "6379")),
|
||||
"PASSWORD": os.environ["REDIS_PASSWORD"],
|
||||
"DATABASE": int(os.environ.get("REDIS_DATABASE", "0")),
|
||||
"SSL": False,
|
||||
'tasks': {
|
||||
'HOST': os.environ['REDIS_HOST'],
|
||||
'PORT': int(os.environ.get('REDIS_PORT', '6379')),
|
||||
'PASSWORD': os.environ['REDIS_PASSWORD'],
|
||||
'DATABASE': int(os.environ.get('REDIS_DATABASE', '0')),
|
||||
'SSL': False,
|
||||
},
|
||||
"caching": {
|
||||
"HOST": os.environ["REDIS_CACHE_HOST"],
|
||||
"PORT": int(os.environ.get("REDIS_CACHE_PORT", "6379")),
|
||||
"PASSWORD": os.environ["REDIS_CACHE_PASSWORD"],
|
||||
"DATABASE": int(os.environ.get("REDIS_CACHE_DATABASE", "1")),
|
||||
"SSL": False,
|
||||
'caching': {
|
||||
'HOST': os.environ['REDIS_CACHE_HOST'],
|
||||
'PORT': int(os.environ.get('REDIS_CACHE_PORT', '6379')),
|
||||
'PASSWORD': os.environ['REDIS_CACHE_PASSWORD'],
|
||||
'DATABASE': int(os.environ.get('REDIS_CACHE_DATABASE', '1')),
|
||||
'SSL': False,
|
||||
},
|
||||
}
|
||||
|
||||
SECRET_KEY = os.environ["SECRET_KEY"]
|
||||
API_TOKEN_PEPPERS = {1: os.environ["API_TOKEN_PEPPER_1"]}
|
||||
TIME_ZONE = os.environ.get("TIME_ZONE", "UTC")
|
||||
MEDIA_ROOT = "/opt/netbox/netbox/media"
|
||||
REPORTS_ROOT = "/opt/netbox/netbox/reports"
|
||||
SCRIPTS_ROOT = "/opt/netbox/netbox/scripts"
|
||||
SECRET_KEY = os.environ['SECRET_KEY']
|
||||
API_TOKEN_PEPPERS = {1: os.environ['API_TOKEN_PEPPER_1']}
|
||||
TIME_ZONE = os.environ.get('TIME_ZONE', 'UTC')
|
||||
MEDIA_ROOT = '/opt/netbox/netbox/media'
|
||||
REPORTS_ROOT = '/opt/netbox/netbox/reports'
|
||||
SCRIPTS_ROOT = '/opt/netbox/netbox/scripts'
|
||||
CENSUS_REPORTING_ENABLED = False
|
||||
File renamed without changes.
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`netbox.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: netbox-service
|
||||
port: 8080
|
||||
|
||||
+28
-19
@@ -56,33 +56,42 @@ spec:
|
||||
startupProbe:
|
||||
exec:
|
||||
command:
|
||||
- /opt/netbox/venv/bin/python
|
||||
- -c
|
||||
- >-
|
||||
exec /usr/bin/curl --fail --silent --show-error --max-time 4
|
||||
--header 'Host: netbox.forust.xyz'
|
||||
http://127.0.0.1:8080/login/ >/dev/null
|
||||
- /usr/bin/curl
|
||||
- --fail
|
||||
- --silent
|
||||
- --show-error
|
||||
- --max-time
|
||||
- "4"
|
||||
- --header
|
||||
- "Host: netbox.forust.xyz"
|
||||
- http://127.0.0.1:8080/login/
|
||||
failureThreshold: 90
|
||||
periodSeconds: 10
|
||||
readinessProbe:
|
||||
exec:
|
||||
command:
|
||||
- /opt/netbox/venv/bin/python
|
||||
- -c
|
||||
- >-
|
||||
exec /usr/bin/curl --fail --silent --show-error --max-time 4
|
||||
--header 'Host: netbox.forust.xyz'
|
||||
http://127.0.0.1:8080/login/ >/dev/null
|
||||
- /usr/bin/curl
|
||||
- --fail
|
||||
- --silent
|
||||
- --show-error
|
||||
- --max-time
|
||||
- "4"
|
||||
- --header
|
||||
- "Host: netbox.forust.xyz"
|
||||
- http://127.0.0.1:8080/login/
|
||||
periodSeconds: 10
|
||||
livenessProbe:
|
||||
exec:
|
||||
command:
|
||||
- /opt/netbox/venv/bin/python
|
||||
- -c
|
||||
- >-
|
||||
exec /usr/bin/curl --fail --silent --show-error --max-time 4
|
||||
--header 'Host: netbox.forust.xyz'
|
||||
http://127.0.0.1:8080/login/ >/dev/null
|
||||
- /usr/bin/curl
|
||||
- --fail
|
||||
- --silent
|
||||
- --show-error
|
||||
- --max-time
|
||||
- "4"
|
||||
- --header
|
||||
- "Host: netbox.forust.xyz"
|
||||
- http://127.0.0.1:8080/login/
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 30
|
||||
resources:
|
||||
@@ -155,7 +164,7 @@ spec:
|
||||
memory: "256Mi"
|
||||
limits:
|
||||
cpu: "1"
|
||||
memory: "1Gi"
|
||||
memory: "512Mi"
|
||||
volumes:
|
||||
- name: netbox-config
|
||||
configMap:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
netronome:
|
||||
image: ghcr.io/autobrr/netronome:v0.14.1
|
||||
image: ghcr.io/autobrr/netronome:v0.15.0
|
||||
restart: unless-stopped
|
||||
container_name: netronome
|
||||
ports:
|
||||
|
||||
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`nm.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: netronome-service
|
||||
port: 7575
|
||||
|
||||
@@ -23,6 +23,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: netronome
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
@@ -30,7 +32,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: netronome
|
||||
image: ghcr.io/autobrr/netronome:v0.14.1
|
||||
image: ghcr.io/autobrr/netronome:v0.15.0
|
||||
ports:
|
||||
- name: netronome-port
|
||||
protocol: TCP
|
||||
@@ -51,8 +53,8 @@ spec:
|
||||
key: NETRONOME__DB_PASSWORD
|
||||
resources:
|
||||
requests:
|
||||
memory: "100Mi"
|
||||
memory: "64Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
memory: "512Mi"
|
||||
memory: "256Mi"
|
||||
cpu: "500m"
|
||||
@@ -12,8 +12,6 @@ spec:
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: nextcloud-chain@file
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: nextcloud-apache
|
||||
port: 11000
|
||||
@@ -32,8 +30,6 @@ spec:
|
||||
- match: Host(`nextcloud.workstation.internal`) || Host(`nextcloud.gigaforust.internal`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
- name: nextcloud-chain@file
|
||||
services:
|
||||
- name: nextcloud-apache
|
||||
|
||||
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`portainer.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: portainer-service
|
||||
port: 9000
|
||||
|
||||
@@ -20,6 +20,8 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: portainer
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
|
||||
@@ -60,16 +60,26 @@ spec:
|
||||
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
|
||||
initialDelaySeconds: 10
|
||||
periodSeconds: 10
|
||||
# Generous timeout: on an I/O-bound single node even exec can take
|
||||
# seconds, and a 1s default kills a healthy postgres mid-recovery.
|
||||
timeoutSeconds: 5
|
||||
startupProbe:
|
||||
exec:
|
||||
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
|
||||
failureThreshold: 30
|
||||
# Crash recovery on an I/O-starved single node can fsync for 10+
|
||||
# minutes; killing postgres mid-recovery restarts the fsync from
|
||||
# zero and loops forever. 90x10s = 15 minutes of grace.
|
||||
failureThreshold: 90
|
||||
periodSeconds: 10
|
||||
livenessProbe:
|
||||
exec:
|
||||
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 20
|
||||
# Same I/O reasoning as readiness, plus more misses before a kill:
|
||||
# restarting postgres on a loaded node only makes recovery longer.
|
||||
timeoutSeconds: 5
|
||||
failureThreshold: 5
|
||||
resources:
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
grafana:
|
||||
image: grafana/grafana:13.2.2
|
||||
image: grafana/grafana:13.2.3
|
||||
container_name: prometheus-grafana
|
||||
restart: unless-stopped
|
||||
env_file:
|
||||
|
||||
@@ -27,6 +27,33 @@ spec:
|
||||
summary: "Pod is crash looping"
|
||||
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} is in CrashLoopBackOff."
|
||||
|
||||
- alert: ContainerOOMKilled
|
||||
expr: max_over_time(kube_pod_container_status_terminated_reason{reason="OOMKilled"}[15m]) >= 1
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Container was OOMKilled"
|
||||
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} was killed for exceeding its memory limit. Raise the limit or reduce the workload."
|
||||
|
||||
- alert: ContainerRestartingTooOften
|
||||
expr: max by (namespace, pod, container) (increase(kube_pod_container_status_restarts_total[30m])) > 3
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Container restarting too often"
|
||||
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} restarted {{ $value }} times in the last 30 minutes."
|
||||
|
||||
- alert: PodEvicted
|
||||
expr: max_over_time(kube_pod_status_reason{reason="Evicted"}[15m]) >= 1
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Pod was evicted"
|
||||
description: "Pod {{ $labels.namespace }}/{{ $labels.pod }} was evicted, usually for node disk or memory pressure."
|
||||
|
||||
- alert: PersistentVolumeClaimFillingUp
|
||||
expr: kubelet_volume_stats_available_bytes / kubelet_volume_stats_capacity_bytes < 0.15
|
||||
for: 15m
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: PrometheusRule
|
||||
metadata:
|
||||
name: crowdsec
|
||||
namespace: prometheus
|
||||
labels:
|
||||
release: prometheus-stack
|
||||
spec:
|
||||
groups:
|
||||
- name: crowdsec
|
||||
rules:
|
||||
- alert: CrowdsecFirewallBouncerStale
|
||||
expr: |
|
||||
absent(cs_lapi_bouncer_requests_total{bouncer="firewall-workstation"})
|
||||
or sum(rate(cs_lapi_bouncer_requests_total{bouncer="firewall-workstation"}[10m])) == 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Firewall bouncer stopped pulling decisions"
|
||||
description: "No LAPI pulls from firewall-workstation for 15 minutes. L3 enforcement is decaying: existing bans expire, new ones never land. Check the systemd unit on the node."
|
||||
|
||||
- alert: CrowdsecDecisionsSpike
|
||||
expr: |
|
||||
sum(cs_active_decisions) - sum(cs_active_decisions offset 30m) > 20
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Spike in active CrowdSec decisions"
|
||||
description: "Active decisions grew by more than 20 in 30 minutes (current: {{ $value }}). Possible ban storm or self-ban - check cscli decisions list."
|
||||
|
||||
- alert: CrowdsecLAPIDecisionErrors
|
||||
expr: |
|
||||
sum(rate(cs_lapi_decisions_ko_total[5m])) > 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "CrowdSec LAPI decision errors"
|
||||
description: "LAPI is failing decision lookups. Bouncers may be failing open. Check LAPI logs and DB health."
|
||||
@@ -26,6 +26,25 @@ grafana:
|
||||
service:
|
||||
port: 80
|
||||
|
||||
# p95 391M, observed max 1046M with no limit at all. Request is set at p95 so the
|
||||
# scheduler sees reality; the limit is a manual exception above the 1.3x max
|
||||
# formula, because a single query burst reached 1046M.
|
||||
resources:
|
||||
requests:
|
||||
memory: "416Mi"
|
||||
cpu: 100m
|
||||
limits:
|
||||
memory: "1536Mi"
|
||||
|
||||
# One block covers both the dashboards and datasources sidecars (p95 91M / 80M).
|
||||
sidecar:
|
||||
resources:
|
||||
requests:
|
||||
memory: "96Mi"
|
||||
cpu: 10m
|
||||
limits:
|
||||
memory: "192Mi"
|
||||
|
||||
additionalDataSources:
|
||||
- name: Loki
|
||||
type: loki
|
||||
@@ -48,13 +67,21 @@ prometheus:
|
||||
storage: 40Gi
|
||||
resources:
|
||||
requests:
|
||||
memory: "700Mi"
|
||||
memory: "768Mi"
|
||||
cpu: 200m
|
||||
limits:
|
||||
memory: "2Gi"
|
||||
memory: "2560Mi"
|
||||
alertmanager:
|
||||
alertmanagerSpec:
|
||||
configSecret: alertmanager-config
|
||||
# p95 66M, max 68M. Silences and notification state live here, so the request
|
||||
# stays above p95 to keep the pod out of the eviction candidates.
|
||||
resources:
|
||||
requests:
|
||||
memory: "96Mi"
|
||||
cpu: 10m
|
||||
limits:
|
||||
memory: "192Mi"
|
||||
storage:
|
||||
volumeClaimTemplate:
|
||||
spec:
|
||||
@@ -66,6 +93,54 @@ alertmanager:
|
||||
requests:
|
||||
storage: 20Gi
|
||||
|
||||
# p95 103M, max 107M, and it grows with the number of cluster objects.
|
||||
kube-state-metrics:
|
||||
# Values key is the dependency name from Chart.yaml, not the `kubeStateMetrics`
|
||||
# condition key. Setting resources under `kubeStateMetrics:` is silently ignored.
|
||||
resources:
|
||||
requests:
|
||||
memory: "128Mi"
|
||||
cpu: 50m
|
||||
limits:
|
||||
memory: "256Mi"
|
||||
|
||||
# p95 39M, max 40M. One per node, so it scales with node count.
|
||||
prometheus-node-exporter:
|
||||
resources:
|
||||
requests:
|
||||
memory: "32Mi"
|
||||
cpu: 20m
|
||||
limits:
|
||||
memory: "128Mi"
|
||||
|
||||
# p95 75M, max 75M. Creates and reconciles every PrometheusRule in the cluster.
|
||||
prometheusOperator:
|
||||
resources:
|
||||
requests:
|
||||
memory: "96Mi"
|
||||
cpu: 50m
|
||||
limits:
|
||||
memory: "192Mi"
|
||||
|
||||
# config-reloader sidecars (p95 33M, max 43M) are not covered: the chart does not
|
||||
# template `prometheusSpec.configReloader.resources`, so there is no values key
|
||||
# for them. They keep shipping with requests only.
|
||||
|
||||
defaultRules:
|
||||
disabled:
|
||||
CPUThrottlingHigh: true
|
||||
KubeControllerManagerDown: true
|
||||
KubeSchedulerDown: true
|
||||
KubeEtcdDown: true
|
||||
KubeEtcdHighCommitDurations: true
|
||||
|
||||
# k0s runs controller-manager/scheduler/etcd internally, not as pods with
|
||||
# component=kube-controller-manager/kube-scheduler/k8s-app=kube-etcd labels.
|
||||
# Their Services get no endpoints, so the targets are permanently down.
|
||||
# kube-proxy and kubelet have endpoints on k0s, keep them enabled.
|
||||
kubeControllerManager:
|
||||
enabled: false
|
||||
kubeScheduler:
|
||||
enabled: false
|
||||
kubeEtcd:
|
||||
enabled: false
|
||||
@@ -9,9 +9,6 @@ spec:
|
||||
routes:
|
||||
- match: Host(`grafana.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: prometheus-stack-grafana
|
||||
port: 80
|
||||
|
||||
@@ -29,26 +29,34 @@ spec:
|
||||
|
||||
- alert: TraefikServiceHigh5xxRate
|
||||
expr: |
|
||||
sum(rate(traefik_service_requests_total{code=~"5.."}[5m])) by (service)
|
||||
/ sum(rate(traefik_service_requests_total[5m])) by (service) * 100 > 5
|
||||
and sum(rate(traefik_service_requests_total[5m])) by (service) > 0
|
||||
sum(rate(traefik_service_requests_total{code=~"5.."}[5m])) by (exported_service)
|
||||
/ sum(rate(traefik_service_requests_total[5m])) by (exported_service) * 100 > 5
|
||||
and sum(rate(traefik_service_requests_total[5m])) by (exported_service) > 0
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "High 5xx error rate for service {{ $labels.service }}"
|
||||
description: "Service {{ $labels.service }} is returning 5xx errors for more than 5% of requests over the last 5 minutes (current: {{ $value | humanizePercentage }})."
|
||||
summary: "High 5xx error rate for service {{ $labels.exported_service }}"
|
||||
description: "Service {{ $labels.exported_service }} is returning 5xx errors for more than 5% of requests over the last 5 minutes (current: {{ $value | humanizePercentage }})."
|
||||
|
||||
# NOTE on `exported_service`: traefik emits per-route series labelled
|
||||
# `service`, but the prometheus scrape already carries a target label
|
||||
# called `service` (the monitored Service object), and with the
|
||||
# default honorLabels=false the scraped value loses the collision and
|
||||
# is renamed to `exported_service`. Grouping by `service` therefore
|
||||
# measures one global aggregate mislabelled per backend, and any
|
||||
# exclusion written against it matches nothing. The generated names
|
||||
# below are namespace-routename-hash as traefik builds them.
|
||||
- alert: TraefikServiceHighLatency
|
||||
expr: |
|
||||
histogram_quantile(0.95,
|
||||
sum(rate(traefik_service_request_duration_seconds_bucket{service!~"xui-xui-service-.*"}[5m])) by (le, service)) > 2
|
||||
sum(rate(traefik_service_request_duration_seconds_bucket{exported_service!~"netbird-netbird-(prod|local)-.*|xui-xray-(prod|local)-.*"}[5m])) by (le, exported_service)) > 2
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "High latency for service {{ $labels.service }}"
|
||||
description: "P95 latency of {{ $labels.service }} exceeded 2 seconds over the last 5 minutes (current: {{ $value | humanizeDuration }})."
|
||||
summary: "High latency for service {{ $labels.exported_service }}"
|
||||
description: "P95 latency of {{ $labels.exported_service }} exceeded 2 seconds over the last 5 minutes (current: {{ $value | humanizeDuration }})."
|
||||
|
||||
- alert: TraefikCertExpiringSoon
|
||||
expr: min(traefik_tls_certs_not_after) - time() < 14 * 24 * 60 * 60
|
||||
|
||||
Loaded 100 of 311 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user