Compare commits

..
Author SHA1 Message Date
renovate-bot 659810afff chore(config): migrate config renovate.json
deploy / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-prettier (push) Failing after 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 0s
ci / validate (push) Successful in 1s
ci / lint-prettier (pull_request) Failing after 3s
ci / lint-ruff (pull_request) Successful in 1s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 24s
ci / build (push) Skipped
ci / lint-yaml (pull_request) Successful in 3s
ci / lint-dockerfiles (pull_request) Successful in 1s
ci / validate (pull_request) Successful in 1s
2026-09-25 22:19:25 +00:00
94 changed files with 738 additions and 3907 deletions

No files matched your search

-10
View File
@@ -1,10 +0,0 @@
# actionlint configuration. Passed explicitly from the ci workflow:
# actionlint -config-file .gitea/actionlint.yaml .gitea/workflows/*.yaml
#
# The self-hosted act_runner registers custom labels that actionlint cannot know
# about, so declare them here instead of silencing the whole runner-label check.
self-hosted-runner:
labels:
- arch
- homelab
- prod
+56 -464
View File
@@ -7,12 +7,6 @@ on:
pull_request:
workflow_dispatch:
# Every job here is checkout plus local tools. The token needs to read the tree
# and nothing else, and saying so keeps a future step that reaches for the API
# from quietly holding a token that can write to the repository.
permissions:
contents: read
concurrency:
group: ci-${{ github.ref }}
cancel-in-progress: ${{ github.ref != 'refs/heads/main' }}
@@ -21,90 +15,8 @@ env:
REGISTRY: gcr.forust.xyz
jobs:
lint-compose:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# Structure check for every committed Compose file, active or not.
# Interpolation, env-file and bind-mount resolution are all switched off,
# because inactive stacks have no .env here and would only fail on their
# ${VAR:?} guards. Active stacks get the full check with interpolation in
# the deploy workflow, where the real .env files live.
- name: Validate Compose files
shell: bash
run: |
set -euo pipefail
source .gitea/workflows/compose-lint.sh
mapfile -t safe_flags < <(compose_safe_flags)
echo "docker compose config ${safe_flags[*]-}"
mapfile -t files < <(compose_files)
if [ "${#files[@]}" -eq 0 ]; then
echo "No Compose files found."
exit 0
fi
failed=0
for f in "${files[@]}"; do
if ! out="$(validate_compose_file "$f" ${safe_flags[@]+"${safe_flags[@]}"} 2>&1)"; then
failed=1
echo "::error file=${f}::$(printf '%s' "$out" | head -1)"
fi
done
if [ "$failed" -ne 0 ]; then
echo "Compose validation failed."
exit 1
fi
echo "checked ${#files[@]} Compose file(s)"
lint-actionlint:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Lint Gitea Actions workflows with actionlint
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint)"
export PATH="$tools_dir:$PATH"
actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml
lint-shellcheck:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Lint shell scripts with ShellCheck
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck)"
export PATH="$tools_dir:$PATH"
# userbot/ is a git subtree synced from forust/userbot, so its shell
# scripts are upstream's to maintain, not ours. Linting them would let a
# routine subtree pull turn the deploy gate red on code we do not own.
mapfile -t scripts < <(
git ls-files '*.sh' ':(glob)**/*.bash' ':!userbot/**'
)
if [ "${#scripts[@]}" -eq 0 ]; then
echo "No shell scripts found."
exit 0
fi
shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}"
lint-prettier:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -112,10 +24,6 @@ jobs:
- name: Check formatting with Prettier
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)"
export PATH="$tools_dir:$PATH"
mapfile -t prettier_files < <(
git ls-files \
| grep -E '\.(md|json|ya?ml|html|css)$' \
@@ -131,23 +39,17 @@ jobs:
lint-ruff:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Lint and format-check Python with Ruff
- name: Lint Python with Ruff
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff)"
export PATH="$tools_dir:$PATH"
ruff check .
ruff format --check .
lint-yaml:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -155,10 +57,6 @@ jobs:
- name: Lint YAML syntax
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)"
export PATH="$tools_dir:$PATH"
mapfile -t yaml_files < <(
git ls-files '*.yaml' '*.yml' \
':!node_modules/**' \
@@ -174,7 +72,6 @@ jobs:
lint-dockerfiles:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -182,10 +79,6 @@ jobs:
- name: Lint Dockerfiles
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)"
export PATH="$tools_dir:$PATH"
mapfile -t dockerfiles < <(
git ls-files ':(glob)**/Dockerfile' ':(glob)**/Dockerfile.*'
)
@@ -197,169 +90,15 @@ jobs:
hadolint -c .hadolint.yaml "${dockerfiles[@]}"
# Known, accepted, and recorded. Each line is a real advisory against a
# package we build into the panel image, kept in this workflow rather than in
# the package manifest so that a subtree sync from forust/userbot cannot
# silently widen the exemption.
#
# starlette is the reason this job is not simply "fail on everything":
# fastapi 0.115.12 pins `starlette<0.47.0`, and the fixes for the last four
# below need 0.49.1 through 1.3.1, so clearing them means a jump from fastapi
# 0.115.12 to 0.141.x. That is upstream's call, not a drive-by in a lint
# commit. Of the seven, four are reachable here in principle: 1942 is a
# crafted Range header hitting FileResponse, and the panel serves its built
# SPA through exactly that; 249 is request.form() ignoring max_fields for
# x-www-form-urlencoded, which is the login form; 1941 is a large multipart
# body blocking the event loop; 161 and 248 are unvalidated Host and request
# path reaching request.url. 2280 needs HTTPEndpoint, which the panel does
# not use, and 2281 is Windows-only, and this deploys on Linux.
#
# The panel answers on userbot.workstation.internal and has no public
# forust.xyz route, which is what keeps the four reachable ones from being
# an internet-facing DoS. It still manages Telegram credentials.
#
# Deleting an entry here is how you accept a new advisory, so the diff says
# so out loud.
scan-deps:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Audit the Python dependencies that ship in the image
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh pip-audit)"
export PATH="$tools_dir:$PATH"
# requirements.txt, not requirements-dev.txt: this is what the image
# installs, and the test tooling is not a shipped attack surface.
pip-audit -r userbot/panel/backend/requirements.txt --strict \
--ignore-vuln CVE-2025-67720 \
--ignore-vuln PYSEC-2026-161 \
--ignore-vuln PYSEC-2026-1941 \
--ignore-vuln PYSEC-2026-1942 \
--ignore-vuln PYSEC-2026-2280 \
--ignore-vuln PYSEC-2026-2281 \
--ignore-vuln PYSEC-2026-248 \
--ignore-vuln PYSEC-2026-249
# devDependencies are excluded on purpose. `npm audit` on the full tree
# reports 7 findings, and every one of them is a build- or test-time
# package: the esbuild CORS advisory needs a vite dev server serving to
# the internet, and nanoid's infinite loop needs a custom generator
# called with size 0, which postcss does not do. None of them are in the
# 91 kB bundle the panel serves. The one production finding, devalue
# via svelte, is moderate, which is where --audit-level draws the line;
# this fails on the next high or critical one.
- name: Audit the production npm dependencies
shell: bash
run: |
set -euo pipefail
# The pinned node, not whatever the runner has. Its system node is a
# rolling Arch package: during this very push its npm was missing
# entirely, and an hour later it was npm 12 on node 26. Both are the
# wrong major anyway — the panel image is node:22-alpine.
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh node)"
export PATH="$tools_dir:$PATH"
cd userbot/panel/frontend
npm ci
npm audit --omit=dev --audit-level=high
test-backend:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# 25 tests over the panel's pydantic models, its auth flow, the SPA
# fallback and the Kubernetes client it shells out with. They existed and
# had never been executed by anything.
#
# Note that userbot/ is a subtree synced from forust/userbot, so a routine
# sync can turn this red on upstream's code. Unlike the shellcheck job,
# which skips that tree because style disagreements there are ours to
# lose, a failing test here is a real defect in a service we deploy.
- name: Run the panel backend test suite
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh uv)"
export PATH="$tools_dir:$PATH"
# A venv in a temp dir rather than a checked-out one: the runner is
# shared, and a leftover .venv would let a dependency the
# requirements no longer pin still satisfy an import.
#
# --python is not optional. uv otherwise takes whatever interpreter it
# finds first, and which one that is depends on the machine: this
# runner runs jobs on the host, where the only interpreter is 3.14,
# and pyrogram's sync.py calls the bare asyncio.get_event_loop() that
# 3.14 no longer auto-creates, so three tests fail at collection. The
# image is python:3.13-slim, so 3.13 is also the version worth
# testing: uv fetches a managed build of it when the host has none,
# which is what makes this job independent of the runner.
venv="$(mktemp -d)/venv"
uv venv --python 3.13 --quiet "$venv"
uv pip install --quiet --python "$venv/bin/python" \
-r userbot/panel/backend/requirements-dev.txt
# `python -m`, not bare `pytest`: the tests import `app.*` relative to
# the backend directory, which only works if the cwd is on sys.path,
# and only `python -m` puts it there.
cd userbot/panel/backend
"$venv/bin/python" -m pytest tests/ -q
test-frontend:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# One `npm ci` for both checks below: it is by far the slowest part of
# this job, and a second one would learn nothing the first did not.
#
# `npm ci`, not `npm install`, for the same reason the Dockerfile uses it:
# the lockfile is what makes the tree that gets checked the tree that
# gets shipped.
- name: Type-check and test the panel frontend
shell: bash
run: |
set -euo pipefail
# The pinned node, not whatever the runner has. Its system node is a
# rolling Arch package: during this very push its npm was missing
# entirely, and an hour later it was npm 12 on node 26. Both are the
# wrong major anyway — the panel image is node:22-alpine.
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh node)"
export PATH="$tools_dir:$PATH"
cd userbot/panel/frontend
npm ci
# svelte-check has been a devDependency all along with no script
# pointing at it, so the type errors it reports had nowhere to
# surface. It is clean today, which is the only reason it can be a
# gate: it stops at whatever upstream introduces rather than
# reporting a backlog we inherited.
npm run check
npm test
validate:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 20
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Validate Kubernetes manifests against JSON schemas
- name: Validate Kubernetes manifests
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
export PATH="$tools_dir:$PATH"
mapfile -t manifests < <(
git ls-files ':(glob)**/k8s/**/*.yaml' ':(glob)**/k8s/**/*.yml' \
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
@@ -376,115 +115,10 @@ jobs:
-summary \
"${manifests[@]}"
# kubeconform has no schemas for CRDs, so every IngressRoute, Certificate,
# PrometheusRule, Middleware, ServersTransport and ServiceMonitor is silently
# skipped above. The live API server knows the real CRD schemas (and runs the
# cert-manager / Traefik admission webhooks), so validate there too.
#
# Only services marked with a k8s/active marker are checked: server-side
# dry-run needs the target namespace to exist, and inactive services are not
# deployed. Services being enabled for the first time are still covered by
# the JSON-schema pass above.
#
# Main pushes only. `--dry-run=server` persists nothing, but it does execute
# the admission webhooks of the production API server, so anyone able to open
# a pull request would be able to run arbitrary manifest content through
# cert-manager and Traefik. A pull request has nothing to gain from it either:
# only main is ever deployed, and this job runs to completion before the
# deploy workflow is allowed to start, so a bad CRD is still caught before
# anything reaches the cluster -- just on the push rather than on the PR.
- name: Note the server-side check is not running here
if: github.event_name == 'pull_request' || github.ref != 'refs/heads/main'
shell: bash
run: |
echo "::notice::Skipping the server-side dry-run. It executes the cert-manager and" \
"Traefik admission webhooks against the production API server, so it is limited" \
"to pushes to main. CRDs are still schema-checked by kubeconform above, and the" \
"server-side pass still runs on main before the deploy."
- name: Validate active manifests against the live API server
if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main'
shell: bash
run: |
set -euo pipefail
if ! kubectl get --raw='/readyz' --request-timeout=10s >/dev/null 2>&1; then
echo "::warning::Cluster unreachable — skipped server-side validation of CRDs (IngressRoute, Certificate, PrometheusRule). Review manifest changes manually."
exit 0
fi
mapfile -t k8s_dirs < <(
git ls-files '*.yaml' '*.yml' \
| grep -E '(^|/)k8s/' \
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
| sort -u
)
manifests=()
kustomize_apps=()
for dir in "${k8s_dirs[@]}"; do
if [ ! -f "${dir}/active" ]; then
echo "skip (no k8s/active): ${dir}"
continue
fi
if [ -f "${dir}/overlays/prod/kustomization.yaml" ]; then
kustomize_apps+=("${dir}/overlays/prod")
elif [ -f "${dir}/base/kustomization.yaml" ]; then
kustomize_apps+=("${dir}/base")
else
while IFS= read -r f; do
[ -n "$f" ] && manifests+=("$f")
done < <(
git ls-files "${dir}/*.yaml" "${dir}/*.yml" \
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
)
fi
done
echo "server-side dry-run: ${#manifests[@]} manifests, ${#kustomize_apps[@]} kustomize apps"
failed=0
for m in ${manifests[@]+"${manifests[@]}"}; do
if ! out="$(kubectl apply --dry-run=server -f "$m" 2>&1)"; then
failed=1
echo "::error file=${m}::$(printf '%s' "$out" | head -1)"
fi
done
for k in ${kustomize_apps[@]+"${kustomize_apps[@]}"}; do
if ! out="$(kubectl apply -k "$k" --dry-run=server 2>&1)"; then
failed=1
echo "::error file=${k}::$(printf '%s' "$out" | head -1)"
fi
done
if [ "$failed" -ne 0 ]; then
echo "Server-side validation failed. The API server (or an admission webhook) rejected these manifests."
exit 1
fi
echo "server-side dry-run: all active manifests accepted by the API server"
build:
needs:
# scan-deps and the two test jobs were missing here, so a commit with a
# known-vulnerable dependency or a failing test still moved the :prod tag.
# The deploy was blocked either way - it requires the whole workflow to
# have succeeded - but the tag had already moved, and the next deploy to
# run resolved it. Publishing and passing the checks are the same gate.
[
lint-actionlint,
lint-shellcheck,
lint-compose,
lint-prettier,
lint-ruff,
lint-yaml,
lint-dockerfiles,
scan-deps,
test-backend,
test-frontend,
validate,
]
needs: [lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev')
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 60
outputs:
services: ${{ steps.services.outputs.services }}
steps:
@@ -497,20 +131,12 @@ jobs:
id: services
shell: bash
run: |
set -euo pipefail
base="${{ github.event.before }}"
if [ -z "$base" ] || [ "$base" = "0000000000000000000000000000000000000000" ]; then
base="$(git rev-list --max-parents=0 HEAD)"
fi
# A failed diff used to leave changed_files empty, which reads exactly
# like "nothing to build": the job went green having built nothing and
# the tag never moved. The status is checked, not assumed.
if ! changed="$(git diff --name-only "$base" "${GITHUB_SHA}")"; then
echo "::error::cannot diff ${base}..${GITHUB_SHA}"
exit 1
fi
mapfile -t changed_files <<<"$changed"
mapfile -t changed_files < <(git diff --name-only "$base" "${GITHUB_SHA}")
services=()
@@ -560,57 +186,30 @@ jobs:
- name: Log in to registry
if: steps.services.outputs.services != ''
shell: bash
# Through env, not by substitution into the script. A secret written
# into a run: block is pasted into the shell source before bash parses
# it, so a password containing a quote, a backtick or $(...) becomes
# code that runs. Masking the value in the log does not prevent that.
env:
REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }}
REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }}
run: |
set -euo pipefail
printf '%s' "$REGISTRY_PASSWORD" | docker login "${REGISTRY}" \
-u "$REGISTRY_USERNAME" \
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login "${REGISTRY}" \
-u "${{ secrets.REGISTRY_USERNAME }}" \
--password-stdin
- name: Build and push changed images
if: steps.services.outputs.services != ''
shell: bash
run: |
# This step was the one run: block in the workflow without it, and it
# is the one that cannot afford it: a docker push that failed partway
# through the loop used to be followed by more pushes, the loop's exit
# status came from the last one, and the job went green with half the
# images missing from the registry.
set -euo pipefail
IFS=, read -r -a services <<< "${{ steps.services.outputs.services }}"
# Tags for this push. The commit-pinned name is the point of this
# step: the deploy resolves it in preference to :prod, so a deploy
# that sat in the queue behind a later push still gets the build of
# the commit CI validated, instead of whatever :prod points at by the
# time it runs. See render_pinned in deploy-lib.sh.
commit_tag=""
if [ "${GITHUB_REF_NAME}" = "main" ]; then
commit_tag="sha-${GITHUB_SHA:0:12}"
fi
set_tags() {
tags=()
case "${GITHUB_REF_NAME}" in
main) tags+=("main" "prod") ;;
dev) tags+=("dev") ;;
esac
if [ -n "$commit_tag" ]; then
tags+=("$commit_tag")
fi
}
for service in "${services[@]}"; do
case "$service" in
dtek_notif)
image="${REGISTRY}/forust/dtek-notif"
set_tags
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
@@ -625,7 +224,15 @@ jobs:
;;
errorpages)
image="${REGISTRY}/forust/error-pages"
set_tags
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
@@ -639,7 +246,15 @@ jobs:
done
;;
userbot)
set_tags
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
for target in runtime panel; do
case "$target" in
runtime)
@@ -665,8 +280,8 @@ jobs:
done
;;
homepages)
for variant in forust xdfnx; do
case "$variant" in
for service in forust xdfnx; do
case "$service" in
forust)
image="${REGISTRY}/forust/forust-homepage"
;;
@@ -674,7 +289,15 @@ jobs:
image="${REGISTRY}/forust/xdfnx-homepage"
;;
esac
set_tags
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
@@ -682,15 +305,15 @@ jobs:
docker build \
--cache-from "type=registry,ref=${image}:buildcache" \
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
"${build_args[@]}" -f "homepages/Dockerfile.${variant}" homepages
"${build_args[@]}" -f "homepages/Dockerfile.${service}" homepages
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
done
;;
edu_master)
for variant in session-keeper webinar-checker; do
case "$variant" in
for service in session-keeper webinar-checker; do
case "$service" in
session-keeper)
context="edu_master/phpsessid-bot"
image="${REGISTRY}/forust/session-keeper"
@@ -700,7 +323,15 @@ jobs:
image="${REGISTRY}/forust/webinar-checker"
;;
esac
set_tags
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
@@ -716,42 +347,3 @@ jobs:
;;
esac
done
# Every image the tree names has to carry the commit-pinned name, not only
# the ones this push rebuilt. A push that touches nothing but manifests
# builds nothing, and its deploy would then find no commit-pinned tag to
# resolve and quietly fall back to the moving :prod - which is the whole
# failure the commit-pinned name exists to remove.
#
# Re-tagging copies the manifest list and transfers no layers, so pinning
# six images that already exist costs six registry writes.
#
# The list is derived from the tree rather than written out here, so an
# image added to a manifest is covered without a second place to update.
- name: Pin the commit name on the images this push did not rebuild
if: github.ref_name == 'main'
shell: bash
run: |
set -euo pipefail
commit_tag="sha-${GITHUB_SHA:0:12}"
mapfile -t repos < <(
git grep -hoE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+' -- '*.yaml' '*.yml' \
| sort -u
)
if [ "${#repos[@]}" -eq 0 ]; then
echo "No own images referenced by the tree."
exit 0
fi
echo "pinning ${#repos[@]} image(s) to $commit_tag"
for repo in "${repos[@]}"; do
if docker buildx imagetools inspect "$repo:$commit_tag" >/dev/null 2>&1; then
echo " already built by this push: ${repo##*/}"
continue
fi
if ! docker buildx imagetools inspect "$repo:prod" >/dev/null 2>&1; then
echo " WARNING: ${repo##*/} has no :prod to pin and no build produced it"
continue
fi
docker buildx imagetools create --tag "$repo:$commit_tag" "$repo:prod"
echo " pinned ${repo##*/}"
done
-46
View File
@@ -1,46 +0,0 @@
#!/usr/bin/env bash
# Shared helpers for validating Compose files. Sourced both by steps in
# .gitea/workflows/ci.yaml and by deploy-lib.sh on the workstation.
#
# Two levels of checking, matching how the repo is structured:
#
# general every committed Compose file, active or not. Pure structure check:
# no ${VAR} interpolation, no .env lookup, no bind-mount path
# resolution. Disabled stacks deliberately have no .env in the repo
# and no values on the CI runner, so a full `config` run would fail on
# their `${VAR:?}` guards for reasons that have nothing to do with the
# change under review.
#
# full active stacks only, with interpolation and env-file resolution, so
# required variables and referenced files are actually resolved. Needs
# the gitignored .env files, so this only runs in the deploy workflow
# on the workstation.
#
# This file is meant to be sourced, not executed.
# All committed Compose files, including the ones deploy never starts.
compose_files() {
git ls-files \
'*/compose.yaml' '*/compose.yml' 'compose.yaml' 'compose.yml' \
'*/docker-compose.yaml' '*/docker-compose.yml'
}
# Prints the flags that turn `docker compose config` into the general check.
# Probed rather than hardcoded so an older Compose without --no-env-resolution
# still gets the flags it does support.
compose_safe_flags() {
local help flag
help="$(docker compose config --help 2>/dev/null || true)"
for flag in --no-interpolate --no-env-resolution --no-path-resolution; do
if printf '%s' "$help" | grep -q -- "$flag"; then
printf '%s\n' "$flag"
fi
done
}
# validate_compose_file <file> [extra docker compose config flags...]
validate_compose_file() {
local file="$1"
shift
docker compose -f "$file" config --quiet "$@"
}
File diff suppressed because it is too large. Load diff
+3 -124
View File
@@ -1,26 +1,13 @@
name: deploy
on:
# Deploy only what CI already validated. workflow_run is used instead of
# workflow_dispatch so a red lint/validate run can never reach the cluster.
workflow_run:
workflows: [ci]
types: [completed]
push:
branches:
- main
workflow_dispatch:
# The deploy jobs read the tree, then reach the cluster over SSH with the
# deploy key. The Actions token itself is not part of that path, so it gets
# read-only contents and no more.
permissions:
contents: read
concurrency:
group: deploy-main
# Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and
# takes the verify job down with it, so a superseded deploy would leave the
# cluster half-applied and unchecked — the exact failure the verify job exists
# to catch. kubectl apply and docker compose up are both idempotent, so letting
# the older run finish and then deploying the newer commit costs little.
cancel-in-progress: false
env:
@@ -30,21 +17,10 @@ env:
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
# workflow_run's own GITHUB_SHA points at the branch head, not at the commit the
# finished ci run checked. Pin the exact validated commit instead, so a push
# landing mid-deploy cannot make the workstation deploy something else. Also
# what the verify job checks the snapshot against. Empty for workflow_dispatch,
# which falls back to the current origin/main.
DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }}
jobs:
preflight:
if: >-
github.event_name != 'workflow_run' ||
(github.event.workflow_run.conclusion == 'success' &&
github.event.workflow_run.head_branch == 'main')
runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -58,7 +34,6 @@ jobs:
validate:
needs: [preflight]
runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 20
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -72,36 +47,6 @@ jobs:
apply-k8s:
needs: [validate]
runs-on: [self-hosted, linux, arch, homelab, prod]
# Apply only, no verification, so this is just the work itself: snapshot,
# then sequential `helm upgrade --atomic --timeout 10m`, then the apply loop.
# Verification has its own job and its own budget.
#
# 45 is roughly four times the measured cost of the stage, which is
# deliberately not raised on a theory:
#
# helm, healthy 3 no-op upgrades ~3-5 min
# helm, one release bad --atomic spends its 10m, ~10-15 min
# then rolls that one back
# apply loop ~40 manifests, 4 of which ~1 min
# resolve an image digest
# restart_stale_images 7.6s to find 8 workloads, ~0.5 min
# 9.8s to resolve their digests
#
# The helm figure is one release, not three: `set -e` aborts
# upgrade_helm_releases on the first failure, so a broken release costs
# 10m and the other two are never attempted. Multiplying 10m by three
# overstates the worst case by 20 minutes.
#
# The 45 minutes this was last raised to 45 were still not enough, and the
# job logs for those runs no longer exist, so what actually consumed the
# budget is not known - the two measurable candidates above account for
# ~15 of it. The one unbounded thing left in this stage is
# `docker manifest inspect` at deploy-lib.sh:236, which has no timeout
# against a registry with a known hang mode. Bound it, and make the stage
# announce what it is working on, before spending any of that on a larger
# ceiling: a stage that is killed with a diagnosable last line is a bug
# report, one that vanishes is not.
timeout-minutes: 45
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -115,7 +60,6 @@ jobs:
apply-compose:
needs: [validate]
runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 30
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -125,68 +69,3 @@ jobs:
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh apply-compose
# Watches the workloads this deploy changed and rolls back the ones that never
# became healthy. Runs even when the apply jobs failed, timed out or were
# cancelled — that is the whole point of splitting it out. `always()` is what
# lets it start after a failed dependency; the needs on apply-compose are a
# barrier, so verification begins only once both applies are done.
verify-k8s:
needs: [apply-k8s, apply-compose]
if: >-
always() &&
needs.apply-k8s.result != 'skipped' &&
needs.apply-compose.result != 'skipped'
runs-on: [self-hosted, linux, arch, homelab, prod]
# Not raised, because the arithmetic does not close.
#
# 32 workloads are under management and the wave width is 8, so the verify
# itself is 4 waves of ROLLOUT_TIMEOUT (300s) = 20 minutes worst case, when
# every rollout times out rather than converging. That is already 20 of 30.
#
# The other 10 would have to absorb rollback, and rollback_workloads is a
# serial `while read` loop at 300s per failed workload. 10 minutes buys two.
# Any larger number is buying a bigger multiple of an unbounded term rather
# than covering a known cost: 60 minutes buys eight, and 60 minutes is
# therefore not a bound, it is a guess with two digits.
#
# The number becomes derivable the moment rollback uses the same wave width
# as the verify: 32 failures then cost 4 waves = 20 minutes instead of 160,
# and 45 covers verify plus rollback at full width. That change is to the
# recovery path and is not folded into a timeout edit.
timeout-minutes: 30
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Verify workloads and roll back on failure
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh verify-k8s
# Asks the public route of every active service whether it is actually
# serving, which the rollout check above structurally cannot: a pod can
# converge and still be crash-looping, or be listening on a port no Service
# points at, or answer 500.
#
# `always()` for the same reason verify-k8s has it, and it runs after that job
# specifically because a rollback is when a route most needs re-checking. The
# needs is a barrier, not a filter: whether verify-k8s passed, failed or was
# cancelled, the probes are what say whether the cluster is serving, and
# suppressing them on a rollback would hide the one run where the answer
# matters most.
smoke:
needs: [verify-k8s]
if: always() && needs.verify-k8s.result != 'skipped'
runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Probe the public route of every active service
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh smoke
-233
View File
@@ -1,233 +0,0 @@
#!/usr/bin/env bash
# Installs the pinned CI tools into "$TOOLS_DIR/bin" and echoes that directory
# on stdout, so callers can do:
#
# export PATH="$(bash .gitea/workflows/install-ci-tools.sh kubeconform shellcheck):$PATH"
#
# Versions come from tool-versions.env next to this script and are kept fresh by
# Renovate. Re-running is cheap: an already-installed tool at the pinned version
# is left alone.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# shellcheck source=tool-versions.env
. "$here/tool-versions.env"
TOOLS_DIR="${TOOLS_DIR:-${RUNNER_TEMP:-/tmp}/homelab-tools}"
BIN_DIR="$TOOLS_DIR/bin"
mkdir -p "$BIN_DIR"
arch="$(uname -m)"
# Upstream projects disagree on arch spelling: kubeconform and actionlint use
# Go names (amd64/arm64), shellcheck uses uname names (x86_64/aarch64), node
# uses neither (x64/arm64), and hadolint mixes the two in a single release
# (x86_64 but arm64).
case "$arch" in
x86_64 | amd64)
goarch=amd64
sharch=x86_64
nodearch=x64
hadolintarch=x86_64
;;
aarch64 | arm64)
goarch=arm64
sharch=aarch64
nodearch=arm64
hadolintarch=arm64
;;
*)
echo "install-ci-tools: unsupported architecture: $arch" >&2
exit 1
;;
esac
fetch() {
# fetch <url> <dest>
if command -v curl >/dev/null 2>&1; then
curl -sSLf --retry 3 -o "$2" "$1"
elif command -v wget >/dev/null 2>&1; then
wget -q -O "$2" "$1"
else
echo "install-ci-tools: neither curl nor wget is available" >&2
exit 1
fi
}
# installed_version <command>
# Prints the version of an already-installed tool, or nothing. Each tool spells
# its version flag differently, hence the case.
installed_version() {
local out
case "$1" in
kubeconform) out="$("$1" -v 2>/dev/null | head -1 || true)" ;;
*) out="$("$1" --version 2>/dev/null | head -1 || true)" ;;
esac
printf '%s' "$out"
}
# at_version <command> <expected>
at_version() {
case "$(installed_version "$1")" in
*"$2"*) return 0 ;;
*) return 1 ;;
esac
}
install_kubeconform() {
if at_version kubeconform "v${KUBECONFORM_VERSION}"; then
return 0
fi
local tmp
tmp="$(mktemp -d)"
fetch "https://github.com/yannh/kubeconform/releases/download/v${KUBECONFORM_VERSION}/kubeconform-linux-${goarch}.tar.gz" \
"$tmp/kubeconform.tar.gz"
tar -xzf "$tmp/kubeconform.tar.gz" -C "$tmp" kubeconform
install -m 0755 "$tmp/kubeconform" "$BIN_DIR/kubeconform"
rm -rf "$tmp"
}
install_shellcheck() {
if at_version shellcheck "${SHELLCHECK_VERSION}"; then
return 0
fi
local tmp
tmp="$(mktemp -d)"
fetch "https://github.com/koalaman/shellcheck/releases/download/v${SHELLCHECK_VERSION}/shellcheck-v${SHELLCHECK_VERSION}.linux.${sharch}.tar.xz" \
"$tmp/shellcheck.tar.xz"
tar -xJf "$tmp/shellcheck.tar.xz" -C "$tmp" --strip-components=1 "shellcheck-v${SHELLCHECK_VERSION}/shellcheck"
install -m 0755 "$tmp/shellcheck" "$BIN_DIR/shellcheck"
rm -rf "$tmp"
}
install_uv() {
if at_version uv "${UV_VERSION}"; then
return 0
fi
local tmp
tmp="$(mktemp -d)"
# uv release tags carry no leading v, unlike every other tool installed here.
fetch "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-${sharch}-unknown-linux-gnu.tar.gz" \
"$tmp/uv.tar.gz"
tar -xzf "$tmp/uv.tar.gz" -C "$tmp" --strip-components=1 "uv-${sharch}-unknown-linux-gnu/uv"
install -m 0755 "$tmp/uv" "$BIN_DIR/uv"
rm -rf "$tmp"
}
install_hadolint() {
if at_version hadolint "${HADOLINT_VERSION}"; then
return 0
fi
# A bare binary, no archive: hadolint ships one file per platform.
fetch "https://github.com/hadolint/hadolint/releases/download/v${HADOLINT_VERSION}/hadolint-linux-${hadolintarch}" \
"$BIN_DIR/hadolint"
chmod 0755 "$BIN_DIR/hadolint"
}
# ruff and yamllint both come from PyPI as wheels, which uv unpacks for us.
install_uv_tool() {
# <package> <pinned version>
if at_version "$1" "$2"; then
return 0
fi
install_uv
UV_TOOL_BIN_DIR="$BIN_DIR" uv tool install --force "$1==$2" >/dev/null
}
install_ruff() {
install_uv_tool ruff "${RUFF_VERSION}"
}
install_yamllint() {
install_uv_tool yamllint "${YAMLLINT_VERSION}"
}
install_pip_audit() {
install_uv_tool pip-audit "${PIP_AUDIT_VERSION}"
}
install_prettier() {
if at_version prettier "${PRETTIER_VERSION}"; then
return 0
fi
# Not a standalone binary: prettier's entry point requires ../package.json
# relative to its own real path, so the package directory has to survive
# next to it. Hence a versioned directory plus a relative symlink, rather
# than copying the one file out as the other installers do.
local dir="$BIN_DIR/prettier-${PRETTIER_VERSION}"
if [ ! -f "$dir/package/package.json" ]; then
rm -rf "$dir"
mkdir -p "$dir"
fetch "https://registry.npmjs.org/prettier/-/prettier-${PRETTIER_VERSION}.tgz" "$dir/prettier.tgz"
tar -xzf "$dir/prettier.tgz" -C "$dir"
rm -f "$dir/prettier.tgz"
# npm strips the exec bit from bin/ on the way into the tarball.
chmod 0755 "$dir/package/bin/prettier.cjs"
fi
# Relative, so the whole tree stays valid if TOOLS_DIR is relocated.
ln -sfn "prettier-${PRETTIER_VERSION}/package/bin/prettier.cjs" "$BIN_DIR/prettier"
}
install_node() {
# npm gets checked by running it, not by looking it up: what matters is that
# it answers, so a stub, a half-removed Arch package or a name that resolves
# to something broken all have to read as "not installed". The runner's npm
# is a symlink into /usr/lib/node_modules/npm, which is exactly the kind of
# thing that disappears between runs.
if at_version node "v${NODE_VERSION}" && [ -n "$(installed_version npm)" ]; then
return 0
fi
# Same shape as prettier above: the tarball's bin/npm and bin/npx are links
# into lib/node_modules, so the whole tree has to survive next to them.
local dir="$BIN_DIR/node-${NODE_VERSION}"
if [ ! -x "$dir/bin/node" ]; then
rm -rf "$dir"
mkdir -p "$dir"
fetch "https://nodejs.org/dist/v${NODE_VERSION}/node-v${NODE_VERSION}-linux-${nodearch}.tar.xz" \
"$dir/node.tar.xz"
tar -xJf "$dir/node.tar.xz" -C "$dir" --strip-components=1 "node-v${NODE_VERSION}-linux-${nodearch}"
rm -f "$dir/node.tar.xz"
fi
# Relative, so the whole tree stays valid if TOOLS_DIR is relocated.
for bin in node npm npx; do
ln -sfn "node-${NODE_VERSION}/bin/${bin}" "$BIN_DIR/${bin}"
done
}
install_actionlint() {
if at_version actionlint "${ACTIONLINT_VERSION}"; then
return 0
fi
local tmp
tmp="$(mktemp -d)"
fetch "https://github.com/rhysd/actionlint/releases/download/v${ACTIONLINT_VERSION}/actionlint_${ACTIONLINT_VERSION}_linux_${goarch}.tar.gz" \
"$tmp/actionlint.tar.gz"
tar -xzf "$tmp/actionlint.tar.gz" -C "$tmp" actionlint
install -m 0755 "$tmp/actionlint" "$BIN_DIR/actionlint"
rm -rf "$tmp"
}
wanted=("$@")
if [ "${#wanted[@]}" -eq 0 ]; then
wanted=(kubeconform shellcheck actionlint prettier ruff yamllint hadolint)
fi
for tool in "${wanted[@]}"; do
case "$tool" in
kubeconform) install_kubeconform ;;
shellcheck) install_shellcheck ;;
actionlint) install_actionlint ;;
prettier) install_prettier ;;
ruff) install_ruff ;;
yamllint) install_yamllint ;;
pip-audit) install_pip_audit ;;
hadolint) install_hadolint ;;
node) install_node ;;
uv) install_uv ;;
*)
echo "install-ci-tools: unknown tool: $tool" >&2
exit 1
;;
esac
done
printf '%s\n' "$BIN_DIR"
+18 -42
View File
@@ -7,58 +7,33 @@ on:
- main
workflow_dispatch:
permissions:
contents: read
jobs:
validate-renovate:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 20
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag,
# so the same version that runs in the cluster is the one validated here.
- name: Resolve the deployed Renovate image
id: image
- name: Validate Renovate Compose draft
shell: bash
run: |
set -euo pipefail
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
renovate/k8s/cronjob.yaml | head -1)"
if [ -z "$image" ]; then
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
exit 1
fi
echo "using $image"
echo "image=$image" >> "$GITHUB_OUTPUT"
trap 'rm -f renovate/.env' EXIT
printf '%s\n' \
'RENOVATE_ENDPOINT=https://gitea.example/api/v1' \
'RENOVATE_TOKEN=test-token' \
'RENOVATE_REPOSITORIES=forust/homelab' \
> renovate/.env
docker compose -f renovate/renovate-compose.yaml config --quiet
- name: Validate Renovate repository config
- name: Validate Kubernetes manifests
shell: bash
run: |
set -euo pipefail
docker run --rm \
-v "$PWD/renovate:/opt/renovate:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
"${{ steps.image.outputs.image }}" \
renovate-config-validator /opt/renovate/renovate.json
# The CronJob cannot read the repository, so renovate/k8s/configmap.yaml
# carries an inlined copy of the config. Fail if it no longer matches.
- name: Check the generated Renovate ConfigMap
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/sync-renovate-configmap.sh --check
- name: Validate Renovate Kubernetes manifests
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
export PATH="$tools_dir:$PATH"
kubeconform \
-v "$PWD:/work" \
-w /work \
ghcr.io/yannh/kubeconform:latest \
-strict \
-ignore-missing-schemas \
-summary \
@@ -66,11 +41,12 @@ jobs:
renovate/k8s/configmap.yaml \
renovate/k8s/cronjob.yaml
- name: Validate Renovate Compose file
- name: Validate Renovate repository config
shell: bash
run: |
set -euo pipefail
source .gitea/workflows/compose-lint.sh
mapfile -t safe_flags < <(compose_safe_flags)
validate_compose_file renovate/renovate-compose.yaml \
${safe_flags[@]+"${safe_flags[@]}"}
docker run --rm \
-v "$PWD:/work" \
-w /work \
renovate/renovate:44.103.0 \
renovate-config-validator renovate.json
+6 -29
View File
@@ -21,11 +21,6 @@ on:
default: false
type: boolean
# Renovate writes through its own bot PAT, passed in as RENOVATE_TOKEN, so the
# Actions token is only ever used to read the checkout.
permissions:
contents: read
concurrency:
group: renovate-run
cancel-in-progress: false
@@ -33,36 +28,18 @@ concurrency:
jobs:
run-renovate:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 60
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag.
# Reading it here means this workflow validates and runs the exact version
# that is deployed, instead of a copy that silently goes stale.
- name: Resolve the deployed Renovate image
id: image
shell: bash
run: |
set -euo pipefail
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
renovate/k8s/cronjob.yaml | head -1)"
if [ -z "$image" ]; then
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
exit 1
fi
echo "using $image"
echo "image=$image" >> "$GITHUB_OUTPUT"
- name: Validate Renovate config
shell: bash
run: |
set -euo pipefail
docker run --rm \
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
"${{ steps.image.outputs.image }}" \
-v "$PWD/renovate/config.js:/opt/renovate/config.js:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/config.js \
renovate/renovate:44.103.0 \
renovate-config-validator
- name: Run Renovate
@@ -79,14 +56,14 @@ jobs:
: "${RENOVATE_TOKEN:?missing RENOVATE_TOKEN secret — add a renovate-bot PAT in repo/org Actions secrets}"
docker run --rm \
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
-v "$PWD/renovate/config.js:/opt/renovate/config.js:ro" \
-e RENOVATE_PLATFORM=gitea \
-e RENOVATE_ENDPOINT=https://gitea.forust.xyz/api/v1 \
-e RENOVATE_TOKEN="$RENOVATE_TOKEN" \
-e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \
-e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \
-e RENOVATE_DRY_RUN="${RENOVATE_DRY_RUN:-}" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
-e RENOVATE_CONFIG_FILE=/opt/renovate/config.js \
-e RENOVATE_BASE_DIR=/tmp/renovate \
-e LOG_LEVEL="${LOG_LEVEL:-info}" \
"${{ steps.image.outputs.image }}"
renovate/renovate:44.103.0
+6 -40
View File
@@ -11,49 +11,15 @@ deploy_port="${DEPLOY_PORT:-22}"
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)"
# The private key is written to a per-run directory that is removed on exit, so a
# failed or cancelled job cannot leave deploy credentials in the runner's temp
# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel.
key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")"
trap 'rm -rf "$key_dir"' EXIT INT TERM
ssh_key="$key_dir/deploy_key"
ssh_key="$RUNNER_TEMP/deploy_key"
mkdir -p "$RUNNER_TEMP"
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
chmod 600 "$ssh_key"
# A connection that died silently used to hang until the job timeout, and the
# stage was never re-run: one flaky TCP session cost a whole 45-minute apply.
# ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds
# setup. Only exit 255 - ssh's own transport failures - is retried. A stage that
# fails on its own merits exits with the remote's status, so a real failure
# still surfaces its own log instead of burning three attempts. The stages are
# declarative applies, so re-running one that had already committed is harmless.
ssh_opts=(
-i "$ssh_key" -p "$deploy_port"
-o BatchMode=yes -o StrictHostKeyChecking=accept-new
-o ConnectTimeout=15
-o ServerAliveInterval=15 -o ServerAliveCountMax=4
)
rc=0
for attempt in 1 2 3; do
if [ "$attempt" -gt 1 ]; then
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
sleep $((attempt * 5))
fi
rc=0
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
"STAGE=$1" bash -se <<'EOF' || rc=$?
ssh -i "$ssh_key" -p "$deploy_port" \
-o BatchMode=yes -o StrictHostKeyChecking=accept-new \
"${DEPLOY_USER}@${DEPLOY_HOST}" \
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} STAGE=$1 bash -se" <<'EOF'
source "$REPO/.gitea/workflows/deploy-lib.sh"
run_stage "$STAGE"
EOF
[ "$rc" -eq 0 ] && break
[ "$rc" -ne 255 ] && break
done
if [ "$rc" -ne 0 ]; then
echo ":: error::stage $1 failed over ssh (exit $rc)"
fi
exit "$rc"
@@ -1,55 +0,0 @@
#!/usr/bin/env bash
# Regenerates renovate/k8s/configmap.yaml from renovate/renovate.json.
#
# renovate/renovate.json is the single source of truth: the CronJob, the Compose
# file and the renovate-run workflow all mount that exact file. A ConfigMap cannot
# read a file from the repository, so the same bytes are inlined here as a literal
# block. This script keeps the copy honest:
#
# .gitea/workflows/sync-renovate-configmap.sh # rewrite in place
# .gitea/workflows/sync-renovate-configmap.sh --check # fail if out of date
#
# renovate-ci runs the --check form on every PR and push, so a config change that
# forgets to regenerate the ConfigMap cannot be merged.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
repo="$(git -C "$here" rev-parse --show-toplevel)"
src="$repo/renovate/renovate.json"
dst="$repo/renovate/k8s/configmap.yaml"
[ -f "$src" ] || {
echo "missing $src" >&2
exit 1
}
render() {
cat <<'HEADER'
# GENERATED FILE - do not edit by hand.
# Source: renovate/renovate.json
# Regenerate: .gitea/workflows/sync-renovate-configmap.sh
# Verify: .gitea/workflows/sync-renovate-configmap.sh --check
apiVersion: v1
kind: ConfigMap
metadata:
name: renovate-config
namespace: renovate
data:
renovate.json: |
HEADER
sed 's/^/ /' "$src"
}
if [ "${1:-}" = "--check" ]; then
if ! diff -u "$dst" <(render) >/dev/null 2>&1; then
echo "ERROR: $dst is out of sync with renovate/renovate.json"
echo "Run: .gitea/workflows/sync-renovate-configmap.sh"
diff -u "$dst" <(render) || true
exit 1
fi
echo "renovate/k8s/configmap.yaml is in sync with renovate/renovate.json"
exit 0
fi
render >"$dst"
echo "wrote $dst"
-33
View File
@@ -1,33 +0,0 @@
# Pinned versions of the CI tools installed by install-ci-tools.sh.
# Renovate keeps these up to date (see customManagers in renovate/renovate.json).
#
# Every version here except NODE_VERSION matches what was already installed on
# the runner, so pinning them changes what CI does not at all. It changes what
# CI does when the runner is rebuilt with something else: today
# install-ci-tools.sh finds the pinned version already on PATH and installs
# nothing, and a runner that drifts gets the pinned one installed over it.
#
# The renovate image version is NOT pinned here: renovate/k8s/cronjob.yaml is the
# single source of truth and the workflows read the tag from it, so there is
# nothing to drift.
ACTIONLINT_VERSION="1.7.7"
SHELLCHECK_VERSION="0.11.0"
KUBECONFORM_VERSION="0.8.0"
PRETTIER_VERSION="3.8.1"
RUFF_VERSION="0.16.8"
YAMLLINT_VERSION="1.38.0"
HADOLINT_VERSION="2.14.0"
# pip-audit reads the advisory database over the network, so a floating version
# would make the same commit report different things on different days. Pin it
# like the rest: the advisories themselves are the moving part, not the tool.
PIP_AUDIT_VERSION="2.10.1"
# uv builds the throwaway venv the pytest job runs in, and unpacks the PyPI
# wheels for ruff, yamllint and pip-audit.
UV_VERSION="0.12.17"
# node runs `npm ci` for the frontend tests and the npm audit, and it is the one
# pin here that does NOT come from the runner: the runner's system node is a
# rolling Arch package (it was node 26 with no npm at all when this was pinned),
# and the panel image is node:22-alpine. Pinned to the image's major on purpose,
# so the tree that gets tested is the tree that gets built. Renovate keeps this
# in step with the Dockerfile's node: tag via the "node runtime" group.
NODE_VERSION="22.23.3"
-2
View File
@@ -96,8 +96,6 @@ replacements.txt
# Temp files
edu_master/temp/
temp/*
# Local-only tooling scratch space (pinned CI tools, verification scripts)
tmp/
# Environment
.env
+1 -1
View File
@@ -73,7 +73,7 @@ spec:
memory: "1.5Gi"
cpu: "300m"
requests:
memory: "1Gi"
memory: "500Mi"
cpu: "50m"
ports:
- containerPort: 3000
+3 -3
View File
@@ -52,7 +52,7 @@ spec:
- containerPort: 9000
resources:
requests:
memory: "768Mi"
memory: "700Mi"
cpu: "300m"
limits:
memory: "1.5Gi"
@@ -86,8 +86,8 @@ spec:
name: authentik-secrets
resources:
requests:
memory: "320Mi"
memory: "512Mi"
cpu: "300m"
limits:
memory: "768Mi"
memory: "1Gi"
cpu: "700m"
+2 -2
View File
@@ -22,10 +22,10 @@ spec:
imagePullPolicy: Always
resources:
requests:
memory: "32Mi"
memory: "20Mi"
cpu: "30m"
limits:
memory: "128Mi"
memory: "64Mi"
cpu: "50m"
envFrom:
- secretRef:
+2 -2
View File
@@ -30,8 +30,8 @@ spec:
key: TUNNEL_TOKEN
resources:
requests:
memory: "128Mi"
memory: "32Mi"
cpu: "30m"
limits:
memory: "256Mi"
memory: "128Mi"
cpu: "200m"
+1 -1
View File
@@ -54,7 +54,7 @@ services:
- "traefik.http.routers.bentopdf.tls.certresolver=letsencrypt"
- "traefik.http.routers.bentopdf.tls=true"
# Local router
- "traefik.http.routers.bentopdf-local.rule=Host(`pdf.workstation.internal`)"
- "traefik.http.routers.bentopdf-local.rule=Host(`pdf.wokstation.internal`)"
- "traefik.http.routers.bentopdf-local.entrypoints=websecure"
- "traefik.http.routers.bentopdf-local.tls=true"
# Dev router
+2 -3
View File
@@ -31,13 +31,12 @@ spec:
name: bentopdf
ports:
- containerPort: 8080
# p95 4M, max 11M over 7 days. Was 50Mi/700Mi.
resources:
requests:
memory: "32Mi"
memory: "50Mi"
cpu: "50m"
ephemeral-storage: "100Mi"
limits:
memory: "128Mi"
memory: "700Mi"
cpu: "700m"
ephemeral-storage: "5Gi"
+2 -3
View File
@@ -38,14 +38,13 @@ spec:
volumeMounts:
- mountPath: /data
name: data
# p95 85M, max 136M over 7 days, spikes while converting. Was 250Mi/1.5Gi.
resources:
requests:
memory: "128Mi"
memory: "250Mi"
cpu: "100m"
limits:
cpu: "1500m"
memory: "512Mi"
memory: "1.5Gi"
volumes:
- name: data
persistentVolumeClaim:
+1 -56
View File
@@ -1,17 +1,3 @@
# crowdsec/k8s is NOT managed by deploy.yaml - apply this by hand, and apply it
# together with a restart:
# kubectl apply -f crowdsec/k8s/crowdsec-middleware.yaml
# kubectl -n traefik rollout restart deploy/traefik
#
# The restart is not optional. In stream mode the plugin runs a package-level
# ticker goroutine (handleStreamTicker over the isCrowdsecStreamHealthy and
# updateFailure globals) that no reconfiguration stops. Applying a change
# wedges the instance: every route referencing it answers 404 and traefik logs
# 'invalid middleware crowdsec-crowdsec-bouncer@kubernetescrd' until the pod is
# replaced. Re-applying the previous config does NOT recover it, and the config
# is not the cause - a valid CIDR cannot fail NewChecker, which is a plain
# net.ParseCIDR. Only a new pod clears it. Measured cost: ~35s down for all
# 20 hosts behind this middleware.
apiVersion: traefik.io/v1alpha1
kind: Middleware
metadata:
@@ -22,48 +8,7 @@ spec:
crowdsec-bouncer:
enabled: true
LogLevel: INFO
# `live` blocked on a `GET /v1/decisions` per request, so a burst
# saturated the LAPI and the plugin 403'd IPs that were never banned.
# v1.3.3 ignores UpdateMaxFailure in `live`, so fail-open is only
# reachable in stream mode, which polls into a cache instead - no
# per-request call to saturate. 15s rather than the 60s default: the
# deploy runner shares one public IP with the house, so this bounds
# both how late a ban lands and how long a lifted one lingers.
CrowdsecMode: stream
UpdateIntervalSeconds: 15
# -1 = never block because the LAPI is unreachable. In v1.3.3
# handleStreamTicker only clears isCrowdsecStreamHealthy when
# updateMaxFailure != -1, and ServeHTTP 403s once it is false, so this
# makes a CrowdSec outage mean "no protection", not "every site 403".
UpdateMaxFailure: -1
CrowdsecMode: live
CrowdsecLapiScheme: http
CrowdsecLapiHost: crowdsec-service.crowdsec.svc.cluster.local:8080
CrowdsecLapiKeyFile: "/etc/traefik/secrets/traefik-api-key"
# Bypasses the bouncer and the decision cache, no LAPI round-trip.
# Keep in sync with forust/local-network in crowdsec-values.yaml.
ClientTrustedIPs:
- "127.0.0.0/8"
- "10.0.0.0/8"
- "172.16.0.0/12"
- "192.168.0.0/16"
- "100.64.0.0/10"
- "169.254.0.0/16"
- "fc00::/7"
- "fe80::/10"
# The mobile operator range from forust/mobile-whitelist, repeated
# deliberately rather than relying on the parser whitelist alone.
# That whitelist drops the event before it reaches a bucket, so no
# decision is ever created - but it is one config away from not
# firing, and the bouncer would then enforce a ban that was never
# justified. This is the last line: even a decision that exists for
# any reason is not served against the phone.
- "84.245.64.0/18"
# The name is HTTPTimeoutSeconds, an int in seconds (min 1) - there is
# no CrowdsecLapiTimeout, and an unrecognised key is silently dropped,
# which is how this sat at the 10s default. Nothing rides on it per
# request any more, so this only bounds the stream pull - and too low
# is the dangerous direction: the LAPI needs ~2s to answer
# /v1/decisions/stream, and a pull that times out leaves the ban cache
# frozen at its startup contents ("failed sending new decisions"),
# i.e. new bans silently never apply. Keep it above the pull latency.
HTTPTimeoutSeconds: 10
+4 -92
View File
@@ -58,37 +58,9 @@ config:
reason: "Mobile IP whitelist"
cidr:
- "84.245.64.0/18"
# CrowdSec's own guidance: CIDR allowlisting belongs at the parser stage.
# A parser whitelist discards the event before it reaches a bucket, so
# these addresses never produce an overflow and never become a decision.
# A postoverflow whitelist is checked only *after* the ban exists, and
# the bouncer answers 403 for as long as it does - which is a window we
# do not want the deploy sitting in.
local-network.yaml: |
name: forust/local-network
description: "Whitelist loopback, private and VPN networks"
whitelist:
reason: "Local network"
cidr:
- "127.0.0.0/8"
- "10.0.0.0/8"
- "172.16.0.0/12"
- "192.168.0.0/16"
# CGNAT range (RFC 6598). The workstation and the k0s node live
# here on WireGuard, and 100.64.0.0/10 is not covered by the
# RFC 1918 blocks above.
- "100.64.0.0/10"
- "169.254.0.0/16"
- "fc00::/7"
- "fe80::/10"
postoverflows:
s01-whitelist:
# The one whitelist that has to stay here: resolving a hostname is a
# network call, and the docs put expensive lookups in postoverflows on
# purpose - it runs only when a bucket actually overflows.
# ddns.forust.xyz is the public home address, not a private one, so
# forust/local-network does not cover it.
home-dynamic-ip.yaml: |
name: forust/home-dynamic-ip
description: "Whitelist home dynamic IP"
@@ -97,59 +69,6 @@ config:
expression:
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
# LAPI-only main config override, merged over config.yaml. NOTE: the
# chart's own default for this key is REPLACED, not merged, so its
# auto_registration block is repeated verbatim below - drop it and the
# agent can no longer register itself.
config.yaml.local: |
api:
server:
auto_registration: # Activate if not using TLS for authentication
enabled: true
token: "${REGISTRATION_TOKEN}" # /!\ Do not modify this variable (auto-generated and handled by the chart)
allowed_ranges: # /!\ Make sure to adapt to the pod IP ranges used by your cluster
- "127.0.0.1/32"
- "192.168.0.0/16"
- "10.0.0.0/8"
- "172.16.0.0/12"
# This homelab has no egress to console.crowdsec.cloud: DNS does
# not resolve. The LAPI kept trying anyway ("Signal push: N
# signals to push", "capi metrics: sending" every 10s) and each
# attempt sat on a resolver timeout WHILE HOLDING A WRITE
# TRANSACTION, which is what kept stalling per-request decision
# lookups even with WAL enabled. Nothing to share and nothing to
# pull - turn the Central API off instead of letting it block the
# only database writer we have.
online_client:
sharing: false
pull:
community: false
blocklists: false
disable_usage_metrics_export: true
db_config:
# SQLite without WAL serialises every reader behind the writer's
# rollback journal, and the LAPI writes constantly: the agent pushes
# Traefik alerts read from Loki, the metrics collector counts
# decisions, the bouncer touches "last pull" on every request.
# Symptom: decision lookups taking 10-30s (and a second connection
# that could not even open the database) while the LAPI sat at 28m
# CPU - the process was blocked in fsync, not computing. Every
# bouncer-protected request then blew through the plugin timeout and
# fail-closed with 403, on every site at once.
# The PVC is local-path-retain (hostPath), not a network share, so
# WAL is safe here; the crowdsec docs recommend it for exactly this
# ("allowing more concurrency in SQLite that will improve
# performances in most scenarios").
use_wal: true
# Keeps the alert table bounded. At the 5000/7d default the file
# reached 54MB in 15 days off the Traefik access log alone, and the
# metrics collector counts decisions on a timer; a smaller working
# set means fewer full scans. Crowdsec only prunes - SQLite never
# shrinks the file, so the size stays until a manual VACUUM.
flush:
max_items: 1000
max_age: 24h
lapi:
env:
- name: COLLECTIONS
@@ -171,20 +90,13 @@ lapi:
enabled: true
size: 1Gi
storageClassName: local-path-retain
# LAPI answers a blocking /v1/decisions lookup for EVERY bouncer-protected
# request (whole Traefik front door), so it is the hot path of the proxy.
# At 400m/500Mi it went CPU-throttled and idle lookups measured 1.3-7.4s,
# which pushed requests into the bouncer's fail-closed 403.
# Single replica on purpose: LAPI is stateful (BoltDB on the `data` PVC,
# credentials on the `config` PVC) - two replicas sharing those RWO
# volumes would corrupt the decision store. Scale up CPU, not replicas.
resources:
limits:
cpu: 1500m
memory: 1Gi
requests:
cpu: 250m
cpu: 400m
memory: 500Mi
requests:
cpu: 50m
memory: 150Mi
service:
type: ClusterIP
storeLAPICscliCredentialsInSecret: true
-7
View File
@@ -32,13 +32,6 @@
# on their own - same name + same password);
# 4. prune bouncer entries idle for 30d.
#
# It used to also delete LePresidente/http-generic-403-bf decisions hourly.
# That was a workaround for the bouncer failing closed on a slow LAPI and
# 403-ing the deploy runner into a 4h ban. The bouncer now polls decisions
# into a cache and never blocks on an unreachable LAPI, so it cannot
# manufacture those 403s any more, and the scenario only fires against real
# scanners - deleting their decisions hourly was undoing a working ban.
#
# Manual apply (crowdsec/k8s is NOT managed by deploy.yaml):
# kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml
# Force a run:
+2 -2
View File
@@ -17,7 +17,7 @@ services:
session-keeper:
build: ./phpsessid-bot
image: gcr.forust.xyz/forust/session-keeper:prod
image: gcr.forust.xyz/forust/session-keeper:latest
pull_policy: build
env_file: .env
restart: unless-stopped
@@ -33,7 +33,7 @@ services:
webinar-checker:
build: ./webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:prod
image: gcr.forust.xyz/forust/webinar-checker:latest
pull_policy: build
env_file: .env
restart: unless-stopped
+1 -20
View File
@@ -11,14 +11,9 @@ spec:
rules:
# No successful webinar check for 5m (~2-3 missed 2-min checks).
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
# The last_success > 0 guard is mandatory: checker.py initialises
# last_success to 0, so without it `time() - 0` equals the current epoch
# and humanizeDuration renders ~20722d on every pod restart. Keep the
# duration expression on the left so $value stays the real gap.
- alert: WebinarCheckerNoSuccessfulCheck
expr: |
((time() - webinar_check_last_success_timestamp_seconds) > 300)
and (webinar_check_last_success_timestamp_seconds > 0)
(time() - webinar_check_last_success_timestamp_seconds > 300)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 2m
labels:
@@ -27,20 +22,6 @@ spec:
summary: "Webinar checker has no successful check for 5m"
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
# Checks are running but none has ever succeeded since pod start.
# Split out from the rule above so a zeroed gauge never feeds
# humanizeDuration.
- alert: WebinarCheckerNeverSucceeded
expr: |
(webinar_check_last_success_timestamp_seconds == 0)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker has never completed a successful check"
description: 'edu-master/webinar-checker: checks have been running for 10m but not one has ever succeeded since the pod started, so every check is failing. Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
- alert: WebinarCheckerConsecutiveFailures
expr: |
-9
View File
@@ -20,15 +20,6 @@ spec:
# renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker
image: mcr.microsoft.com/playwright:v1.56.0-jammy
imagePullPolicy: IfNotPresent
# p95 412M, max 478M over 7 days, no limit before. Request is set at p95
# so the pod is not an eviction candidate; the limit stays above 2x the
# request because browser page lifetimes are unpredictable.
resources:
requests:
cpu: "200m"
memory: "416Mi"
limits:
memory: "1Gi"
command:
- npx
- -y
+2 -2
View File
@@ -28,10 +28,10 @@ spec:
resources:
requests:
cpu: 25m
memory: 32Mi
memory: 64Mi
limits:
cpu: 250m
memory: 128Mi
memory: 256Mi
readinessProbe:
exec:
command: ["redis-cli", "ping"]
+4 -3
View File
@@ -31,17 +31,18 @@ spec:
echo "redis is ready"
containers:
- name: session-keeper
image: gcr.forust.xyz/forust/session-keeper:prod
image: gcr.forust.xyz/forust/session-keeper:latest
imagePullPolicy: Always
envFrom:
- secretRef:
name: edu-master-secrets
resources:
requests:
cpu: 25m
memory: 32Mi
memory: 96Mi
limits:
cpu: 250m
memory: 128Mi
memory: 256Mi
readinessProbe:
exec:
command: ["/bin/sh", "-ec", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"]
+4 -11
View File
@@ -45,19 +45,12 @@ spec:
echo "playwright ok"
containers:
- name: webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:prod
image: gcr.forust.xyz/forust/webinar-checker:latest
imagePullPolicy: Always
ports:
- name: metrics
containerPort: 8000
protocol: TCP
readinessProbe:
httpGet:
path: /health
port: metrics
periodSeconds: 10
timeoutSeconds: 3
failureThreshold: 12
initialDelaySeconds: 10
envFrom:
- secretRef:
name: edu-master-secrets
@@ -67,7 +60,7 @@ spec:
resources:
requests:
cpu: "50m"
memory: "192Mi"
memory: "128Mi"
limits:
cpu: "600m"
memory: "384Mi"
memory: "512Mi"
+1 -1
View File
@@ -1,7 +1,7 @@
services:
errorpage:
build: .
image: gcr.forust.xyz/forust/error-pages:prod
image: gcr.forust.xyz/forust/error-pages:latest
pull_policy: build
container_name: error-pages
restart: unless-stopped
+1 -15
View File
@@ -27,21 +27,7 @@ spec:
spec:
containers:
- name: error-pages
image: gcr.forust.xyz/forust/error-pages:prod
# p95 6M, max 10M, no limit before.
resources:
requests:
cpu: "10m"
memory: "32Mi"
limits:
memory: "128Mi"
image: gcr.forust.xyz/forust/error-pages:latest
ports:
- containerPort: 80
readinessProbe:
httpGet:
path: /404.html
port: 80
periodSeconds: 10
timeoutSeconds: 2
failureThreshold: 3
---
-6
View File
@@ -17,12 +17,6 @@ data:
GITEA__mailer__ENABLED: "false"
# No code/issue search needed: bleve reindexes the whole issue index on
# every pod restart (cron.rebuild_issue_indexer RUN_AT_START) and hammers
# the rotational disk for an hour. "db" serves issue search from postgres.
GITEA__indexer__ISSUE_INDEXER_TYPE: "db"
GITEA__indexer__REPO_INDEXER_ENABLED: "false"
GITEA__log__logger.access.MODE: "console, file"
USER_UID: "1000"
USER_GID: "1000"
+2 -2
View File
@@ -47,10 +47,10 @@ spec:
mountPath: /data
resources:
requests:
memory: "320Mi"
memory: "512Mi"
cpu: "300m"
limits:
memory: "1Gi"
memory: "1.5Gi"
cpu: "1300m"
volumes:
- name: gitea-data
+3 -7
View File
@@ -15,15 +15,11 @@ spec:
services:
- name: gitea-service
port: 3000
# Registry route: NO crowdsec-bouncer.
# A deploy burst (runner Action API polls, `docker manifest inspect` per
# own image, containerd pulls, smoke probes) fires hundreds of parallel
# registry calls, and a ban on the runner breaks every later job. This
# route only serves authenticated OCI traffic - registry tokens and
# basic-auth are already handled by gitea - and scanners get nothing
# useful from /v2, so there is no bruteforce surface to protect here.
- match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`)
kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services:
- name: gitea-service
port: 3000
+2 -2
View File
@@ -57,10 +57,10 @@ spec:
resources:
requests:
cpu: "50m"
memory: "32Mi"
memory: "64Mi"
limits:
cpu: "200m"
memory: "128Mi"
memory: "256Mi"
volumes:
- name: glance-config
configMap:
+2 -2
View File
@@ -3,7 +3,7 @@ services:
build:
context: .
dockerfile: Dockerfile.forust
image: gcr.forust.xyz/forust/forust-homepage:prod
image: gcr.forust.xyz/forust/forust-homepage:latest
pull_policy: build
# ports:
# - "8085:80"
@@ -35,7 +35,7 @@ services:
build:
context: .
dockerfile: Dockerfile.xdfnx
image: gcr.forust.xyz/forust/xdfnx-homepage:prod
image: gcr.forust.xyz/forust/xdfnx-homepage:latest
pull_policy: build
restart: unless-stopped
# ports:
+8 -20
View File
@@ -27,22 +27,16 @@ spec:
spec:
containers:
- name: forust-homepage
image: gcr.forust.xyz/forust/forust-homepage:prod
image: gcr.forust.xyz/forust/forust-homepage:latest
imagePullPolicy: Always
ports:
- containerPort: 80
readinessProbe:
httpGet:
path: /
port: 80
periodSeconds: 10
timeoutSeconds: 2
failureThreshold: 3
resources:
requests:
memory: "32Mi"
memory: "10Mi"
cpu: "20m"
limits:
memory: "128Mi"
memory: "100Mi"
cpu: "50m"
---
apiVersion: v1
@@ -74,20 +68,14 @@ spec:
spec:
containers:
- name: xdfnx-homepage
image: gcr.forust.xyz/forust/xdfnx-homepage:prod
image: gcr.forust.xyz/forust/xdfnx-homepage:latest
imagePullPolicy: Always
ports:
- containerPort: 80
readinessProbe:
httpGet:
path: /
port: 80
periodSeconds: 10
timeoutSeconds: 2
failureThreshold: 3
resources:
requests:
memory: "32Mi"
memory: "10Mi"
cpu: "20m"
limits:
memory: "128Mi"
memory: "100Mi"
cpu: "50m"
-24
View File
@@ -1,24 +0,0 @@
# You can find documentation for all the supported env variables at https://docs.immich.app/install/environment-variables
# The location where your uploaded files are stored. The k8s manifests bind
# mount /mnt/immich/library, which is the sdc9 partition - the same place, so
# the two deployment paths look at one library.
UPLOAD_LOCATION=/mnt/immich/library
# The location where your database files are stored. Network shares are not supported for the database
DB_DATA_LOCATION=./postgres
# To set a timezone, uncomment the next line and change Etc/UTC to a TZ identifier from this list: https://en.wikipedia.org/wiki/List_of_tz_database_time_zones#List
# TZ=Etc/UTC
# The Immich version to use. You can pin this to a specific version like "v2.1.0"
IMMICH_VERSION=v3
# Connection secret for postgres. You should change it to a random password
# Please use only the characters `A-Za-z0-9`, without special characters or spaces
DB_PASSWORD=postgres
# The values below this line do not need to be changed
###################################################################################
DB_USERNAME=postgres
DB_DATABASE_NAME=immich
-63
View File
@@ -1,63 +0,0 @@
name: immich
services:
immich-server:
container_name: immich_server
image: ghcr.io/immich-app/immich-server:v3
volumes:
- ${UPLOAD_LOCATION}:/data
- /etc/localtime:/etc/localtime:ro
env_file:
- .env
ports:
- "2283:2283"
depends_on:
- redis
- database
restart: always
healthcheck:
disable: false
immich-machine-learning:
container_name: immich_machine_learning
# For hardware acceleration, add one of -[armnn, cuda, rocm, openvino, rknn] to the image tag.
# Example tag: ${IMMICH_VERSION:-release}-cuda
image: ghcr.io/immich-app/immich-machine-learning:${IMMICH_VERSION:-release}
# extends: # uncomment this section for hardware acceleration - see https://docs.immich.app/features/ml-hardware-acceleration
# file: hwaccel.ml.yml
# service: cpu # set to one of [armnn, cuda, rocm, openvino, openvino-wsl, rknn] for accelerated inference - use the `-wsl` version for WSL2 where applicable
volumes:
- model-cache:/cache
env_file:
- .env
restart: always
healthcheck:
disable: false
redis:
container_name: immich_redis
image: docker.io/valkey/valkey:9@sha256:418652cfb58ef879d4978c33553735d7147016032d5aefaa14c828e611eb9dfd
healthcheck:
test: redis-cli ping | grep -q PONG || exit 1
restart: always
database:
container_name: immich_postgres
image: ghcr.io/immich-app/postgres:14-vectorchord0.4.3-pgvectors0.2.0@sha256:bcf63357191b76a916ae5eb93464d65c07511da41e3bf7a8416db519b40b1c23
environment:
POSTGRES_PASSWORD: ${DB_PASSWORD}
POSTGRES_USER: ${DB_USERNAME}
POSTGRES_DB: ${DB_DATABASE_NAME}
POSTGRES_INITDB_ARGS: "--data-checksums"
# Uncomment the DB_STORAGE_TYPE: 'HDD' var if your database isn't stored on SSDs
# DB_STORAGE_TYPE: 'HDD'
volumes:
# Do not edit the next line. If you want to change the database storage location on your system, edit the value of DB_DATA_LOCATION in the .env file
- ${DB_DATA_LOCATION}:/var/lib/postgresql/data
shm_size: 128mb
restart: always
healthcheck:
disable: false
volumes:
model-cache:
-28
View File
@@ -1,28 +0,0 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: immich-prod-tls
namespace: immich
spec:
secretName: immich-prod-tls
dnsNames:
- immich.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: immich
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
-24
View File
@@ -1,24 +0,0 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: immich-config
namespace: immich
data:
TZ: "Europe/Bratislava"
# The database in this namespace, not the shared one in the database
# namespace: v3 needs VectorChord, and only the dedicated image carries it.
DB_HOSTNAME: "immich-postgres"
DB_PORT: "5432"
DB_USERNAME: "immich"
DB_DATABASE_NAME: "immich"
DB_SSL_MODE: "disable"
DB_VECTOR_EXTENSION: "vectorchord"
REDIS_HOSTNAME: "immich-valkey"
REDIS_PORT: "6379"
# Traefik is the only client of the server, and it is a pod: the address immich
# sees is inside the node's pod CIDR. Without this the server does not trust
# X-Forwarded-For and every request looks like it came from Traefik itself.
IMMICH_TRUSTED_PROXIES: "10.244.0.0/24"
-95
View File
@@ -1,95 +0,0 @@
apiVersion: v1
kind: Service
metadata:
name: immich-service
namespace: immich
spec:
selector:
app: immich
ports:
- name: http
port: 2283
targetPort: 2283
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: immich-deployment
namespace: immich
labels:
app: immich
spec:
replicas: 2
selector:
matchLabels:
app: immich
template:
metadata:
labels:
app: immich
spec:
containers:
- name: immich
image: ghcr.io/immich-app/immich-server:v3
envFrom:
- configMapRef:
name: immich-config
- secretRef:
name: immich-secrets
ports:
- name: http
containerPort: 2283
volumeMounts:
- name: immich-data
mountPath: /data
# The first boot runs migrations and warms the transcoder, which can
# take minutes, so liveness has to wait on the startup probe.
startupProbe:
httpGet:
path: /api/server/ping
port: http
failureThreshold: 60
periodSeconds: 10
timeoutSeconds: 5
readinessProbe:
httpGet:
path: /api/server/ping
port: http
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
httpGet:
path: /api/server/ping
port: http
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
# Only the request is scheduled against, and the node is already
# oversubscribed (5.58 of 6 cores requested) while actually running
# at about 1.5. So the request states what this sits at while idle -
# tens of millicores - and the limit leaves room for the burst that
# matters: thumbnails, transcodes and metadata extraction.
#
# The limit used to be 2Gi, but the server OOMKilled on boot while
# chewing through a backlog of unprocessed assets (API +Workers in
# one container spike well past idle).
resources:
requests:
cpu: "100m"
memory: "512Mi"
limits:
cpu: "1500m"
memory: "4Gi"
volumes:
- name: immich-data
# The library lives on the node's own disk, not in a PVC. A PVC here
# meant declaring a size up front for data that does not exist yet,
# on a provisioner that cannot grow it, and the only copy of the
# photos was one `kubectl delete namespace` away.
#
# Directory, not DirectoryOrCreate, on purpose: if sdc9 is not
# mounted, this must fail loudly instead of quietly writing the
# library onto the root filesystem.
hostPath:
path: /mnt/immich/library
type: Directory
-36
View File
@@ -1,36 +0,0 @@
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: immich-prod
namespace: immich
spec:
entryPoints:
- websecure
routes:
- match: Host(`immich.forust.xyz`)
kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services:
- name: immich-service
port: 2283
tls:
secretName: immich-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: immich-local
namespace: immich
spec:
entryPoints:
- websecure
routes:
- match: Host(`immich.workstation.internal`) || Host(`immich.gigaforust.internal`)
kind: Rule
services:
- name: immich-service
port: 2283
tls:
secretName: internal-wildcard-tls
-92
View File
@@ -1,92 +0,0 @@
apiVersion: v1
kind: Service
metadata:
name: immich-machine-learning
namespace: immich
spec:
selector:
app: immich-machine-learning
ports:
- name: http
port: 3003
targetPort: 3003
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: immich-machine-learning-deployment
namespace: immich
labels:
app: immich-machine-learning
spec:
replicas: 1
selector:
matchLabels:
app: immich-machine-learning
template:
metadata:
labels:
app: immich-machine-learning
spec:
containers:
- name: immich-machine-learning
image: ghcr.io/immich-app/immich-machine-learning:v3
envFrom:
- configMapRef:
name: immich-config
- secretRef:
name: immich-secrets
ports:
- name: http
containerPort: 3003
volumeMounts:
- name: model-cache
mountPath: /cache
# The first request pulls a model over the internet, so a cold start
# is slower than a container start.
startupProbe:
httpGet:
path: /ping
port: http
failureThreshold: 60
periodSeconds: 5
timeoutSeconds: 5
readinessProbe:
httpGet:
path: /ping
port: http
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
httpGet:
path: /ping
port: http
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
# Same reasoning as the server: the request covers the idle cost
# only, because the node has no spare cores to schedule against.
# Recognition is the burst - a busy import wants both cores.
resources:
requests:
cpu: "100m"
memory: "1Gi"
limits:
cpu: "2000m"
memory: "3Gi"
volumes:
- name: model-cache
persistentVolumeClaim:
claimName: immich-model-cache-pvc
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: immich-model-cache-pvc
namespace: immich
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 2Gi
-5
View File
@@ -1,5 +0,0 @@
# yaml-language-server: $schema=kubernetes
apiVersion: v1
kind: Namespace
metadata:
name: immich
-138
View File
@@ -1,138 +0,0 @@
# Immich's own database, separate from the shared postgres in the database
# namespace. It has to be separate: v3 checks the VectorChord version at startup
# and refuses to boot without it, VectorChord needs its .so in
# shared_preload_libraries, and that can only be read when postmaster starts.
# So the shared instance would have to be rebuilt on a custom image carrying
# vchord and restarted - for every consumer of it (authentik, gitea, netbox,
# netronome, penpot, statuspage). Not worth it for one photo library.
apiVersion: v1
kind: Service
metadata:
name: immich-postgres
namespace: immich
labels:
app: immich-postgres
spec:
selector:
app: immich-postgres
ports:
- name: postgres
port: 5432
targetPort: postgres
---
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: immich-postgres
namespace: immich
labels:
app: immich-postgres
spec:
serviceName: immich-postgres
replicas: 1
selector:
matchLabels:
app: immich-postgres
template:
metadata:
labels:
app: immich-postgres
spec:
containers:
- name: postgres
# v3.x expects vchord for its vector work and vectors (pgvecto.rs)
# for some index types. This image ships both and preloads them, plus
# its own shared_buffers and wal settings, through
# /etc/postgresql/postgresql.conf - which its entrypoint reaches via
# `postgres -c config_file=...` in the image CMD.
#
# So there is deliberately no `command:` here. Overriding it replaces
# that config_file, and it also loses the step where the entrypoint
# drops from root to the postgres user: postmaster then starts as
# root and refuses to run.
image: ghcr.io/immich-app/postgres:14-vectorchord0.4.3-pgvectors0.2.0@sha256:bcf63357191b76a916ae5eb93464d65c07511da41e3bf7a8416db519b40b1c23
env:
- name: POSTGRES_USER
value: immich
- name: POSTGRES_DB
value: immich
- name: POSTGRES_PASSWORD
valueFrom:
secretKeyRef:
name: immich-secrets
key: DB_PASSWORD
# Only read when the data directory is empty, so the checksums are
# decided here and never again.
- name: POSTGRES_INITDB_ARGS
value: --data-checksums
# The postgres-data volume lives on sdc, which is rotational. The
# SSD template is the default; HDD only changes the planner costs
# (effective_io_concurrency, random_page_cost), nothing structural.
- name: DB_STORAGE_TYPE
value: HDD
- name: TZ
valueFrom:
configMapKeyRef:
name: immich-config
key: TZ
ports:
- name: postgres
containerPort: 5432
volumeMounts:
- name: postgres-data
mountPath: /var/lib/postgresql/data
# The upstream compose file asks docker for 128mb of shm. Kubernetes
# gives every container 64mb, which is not what postmaster expects
# for parallel query workers and the WAL writer.
- name: shm
mountPath: /dev/shm
# Probes use a generous timeout on purpose: the data lives on a
# rotational disk on a loaded single node, and pg_isready can take
# seconds during WAL recovery. A 1s timeout kills the container
# mid-recovery and restarts the spiral.
startupProbe:
exec:
command: ["sh", "-c", "pg_isready -U immich -d immich"]
failureThreshold: 60
periodSeconds: 5
timeoutSeconds: 5
readinessProbe:
exec:
command: ["sh", "-c", "pg_isready -U immich -d immich"]
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
exec:
command: ["sh", "-c", "pg_isready -U immich -d immich"]
initialDelaySeconds: 30
periodSeconds: 20
timeoutSeconds: 5
# The image template sets shared_buffers to 512MB, and the vchord and
# vectors workers are Rust binaries with a real RSS footprint on top
# of postmaster, checkpointer and friends. 1Gi was enough to start
# the server but the vectors worker kept dying in it, so the limit
# sits at 2Gi. The request stays at the idle cost.
resources:
requests:
cpu: "50m"
memory: "256Mi"
limits:
cpu: "1000m"
memory: "2Gi"
volumes:
- name: shm
emptyDir:
medium: Memory
sizeLimit: 128Mi
volumeClaimTemplates:
- metadata:
name: postgres-data
spec:
accessModes: ["ReadWriteOnce"]
# Retain: this is the metadata for a library that only exists in one
# place, and local-path cannot expand a bound volume, so this size has
# to hold until the library is rebuilt or dumped elsewhere.
storageClassName: local-path-retain
resources:
requests:
storage: 32Gi
-13
View File
@@ -1,13 +0,0 @@
apiVersion: v1
kind: Secret
metadata:
name: immich-secrets
namespace: immich
type: Opaque
stringData:
# Creates the immich superuser in this namespace's own postgres on first
# boot, and is the same value the server connects with. Nothing outside the
# immich namespace needs it. Letters and digits only: immich reads this into
# a connection string.
DB_PASSWORD: "changeme"
REDIS_PASSWORD: "changeme"
-84
View File
@@ -1,84 +0,0 @@
apiVersion: v1
kind: Service
metadata:
name: immich-valkey
namespace: immich
labels:
app: immich-valkey
spec:
clusterIP: None
selector:
app: immich-valkey
ports:
- name: valkey
port: 6379
targetPort: valkey
---
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: immich-valkey
namespace: immich
labels:
app: immich-valkey
spec:
serviceName: immich-valkey
replicas: 1
selector:
matchLabels:
app: immich-valkey
template:
metadata:
labels:
app: immich-valkey
spec:
containers:
- name: valkey
image: docker.io/valkey/valkey:9.1.2-alpine
command:
- sh
- -c
- valkey-server --appendonly yes --save 30 1 --loglevel warning --requirepass "$REDIS_PASSWORD"
envFrom:
- secretRef:
name: immich-secrets
ports:
- name: valkey
containerPort: 6379
volumeMounts:
- name: valkey-data
mountPath: /data
# Same reasoning as postgres: 1s probe timeouts flap on a loaded
# single node with rotational storage.
startupProbe:
exec:
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
failureThreshold: 20
periodSeconds: 5
timeoutSeconds: 5
readinessProbe:
exec:
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
exec:
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
initialDelaySeconds: 20
periodSeconds: 20
timeoutSeconds: 5
resources:
requests:
cpu: "25m"
memory: "64Mi"
limits:
cpu: "250m"
memory: "256Mi"
volumeClaimTemplates:
- metadata:
name: valkey-data
spec:
accessModes: ["ReadWriteOnce"]
resources:
requests:
storage: 1Gi
+4 -18
View File
@@ -7,32 +7,18 @@
controller:
type: daemonset
# config-reloader sidecar: p95 33M, max 43M. The chart keeps it at the top level,
# not under `alloy:`.
configReloader:
resources:
requests:
memory: "32Mi"
cpu: "10m"
limits:
memory: "128Mi"
cpu: "50m"
limits:
memory: "512Mi"
cpu: "500m"
image:
tag: "v1.19.2"
alloy:
# p95 275M, max 287M. Alloy tails every pod log and ships it to Loki, so it sits
# on the same IronWolf read path the node is I/O bound on. Request is set at p95.
# The chart key is `alloy.resources`. `controller.resources` is ignored silently,
# which is why this pod shipped with no limits at all.
resources:
requests:
memory: "288Mi"
cpu: "50m"
limits:
memory: "512Mi"
configMap:
create: true
content: |
+4 -27
View File
@@ -39,16 +39,6 @@ loki:
local:
directory: /var/loki/rules
# p95 84M, max 85M for the rules sidecar that shares the singleBinary pod.
# The chart exposes it as `sidecar.resources`, shared with any other sidecar.
sidecar:
resources:
requests:
memory: "96Mi"
cpu: "10m"
limits:
memory: "192Mi"
singleBinary:
replicas: 1
persistence:
@@ -57,10 +47,10 @@ singleBinary:
storageClass: local-path-retain
resources:
requests:
memory: "256Mi"
memory: "512Mi"
cpu: "200m"
limits:
memory: "1Gi"
memory: "2Gi"
cpu: "1000m"
# Zeroed: unused in SingleBinary mode (chart validation requires it).
@@ -73,25 +63,12 @@ backend:
gateway:
replicas: 1
# Single node: chart default is required podAntiAffinity on hostname +
# RollingUpdate 25%/25% (effective maxUnavailable=0 at replicas=1).
# That deadlocks the rollout: the new pod stays Unschedulable while the
# old one lives, and the old one never leaves while the new one is not
# Ready. Null clears the default (an empty map would deep-merge with it
# and keep the required rule); maxUnavailable=1 allows a brief gateway
# outage during rollouts instead of a stuck deploy.
affinity: null
deploymentStrategy:
type: RollingUpdate
rollingUpdate:
maxSurge: 1
maxUnavailable: 1
resources:
requests:
memory: "32Mi"
memory: "64Mi"
cpu: "50m"
limits:
memory: "128Mi"
memory: "256Mi"
cpu: "300m"
monitoring:
+1 -1
View File
@@ -1,6 +1,6 @@
services:
metube:
image: ghcr.io/alexta69/metube:2026.09.27
image: ghcr.io/alexta69/metube:2026.09.25
container_name: metube
restart: unless-stopped
# ports:
+3 -4
View File
@@ -27,7 +27,7 @@ spec:
spec:
containers:
- name: metube
image: ghcr.io/alexta69/metube:2026.09.27
image: ghcr.io/alexta69/metube:2026.09.25
envFrom:
- configMapRef:
name: metube-config
@@ -36,13 +36,12 @@ spec:
volumeMounts:
- name: downloads
mountPath: /downloads
# p95 72M, max 80M over 7 days. Was 600Mi/2Gi.
resources:
requests:
memory: "96Mi"
memory: "600Mi"
cpu: "400m"
limits:
memory: "384Mi"
memory: "2Gi"
cpu: "1700m"
volumes:
- name: downloads
File renamed without changes.
+1 -1
View File
@@ -87,7 +87,7 @@ services:
- proxy
dashboard:
image: netbirdio/dashboard:v2.93.0
image: netbirdio/dashboard:v2.90.10
container_name: netbird-dashboard
restart: unless-stopped
environment:
+6 -9
View File
@@ -7,18 +7,12 @@ spec:
entryPoints:
- websecure
routes:
# NO crowdsec-bouncer on the API routes. These are the mesh client's own
# endpoints: gRPC-gateway management calls plus signal/relay long-polling,
# authenticated by NetBird's token rather than by a login form. A ban here
# is self-defeating - the client needs the mesh to reach anything else, so
# CrowdSec banning it locks the peer out of the network it needs to
# function. It also backfires: a banned peer keeps retrying, every retry
# is another 403, and LePresidente/http-generic-403-bf turns five 403s in
# ten seconds into a 4h ban, so one 403 loop kept re-arming the ban.
# netbird-local below has always been exempt; this makes prod match.
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`))
kind: Rule
priority: 100
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services:
- name: netbird-server-service
port: 80
@@ -26,6 +20,9 @@ spec:
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`))
kind: Rule
priority: 100
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services:
- name: netbird-server-service
port: 80
+5 -6
View File
@@ -88,13 +88,12 @@ spec:
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 5
# p95 97M, max 102M over 7 days. Was 256Mi/1Gi.
resources:
requests:
memory: "128Mi"
memory: "256Mi"
cpu: "250m"
limits:
memory: "384Mi"
memory: "1Gi"
cpu: "1000m"
volumes:
- name: netbird-data
@@ -138,7 +137,7 @@ spec:
spec:
containers:
- name: dashboard
image: netbirdio/dashboard:v2.93.0
image: netbirdio/dashboard:v2.90.10
envFrom:
- configMapRef:
name: netbird-config
@@ -163,10 +162,10 @@ spec:
failureThreshold: 5
resources:
requests:
memory: "32Mi"
memory: "64Mi"
cpu: "50m"
limits:
memory: "128Mi"
memory: "256Mi"
cpu: "300m"
---
apiVersion: v1
File renamed without changes.
+31 -31
View File
@@ -1,48 +1,48 @@
import os
def _csv(name, default=''):
return [item.strip() for item in os.environ.get(name, default).split(',') if item.strip()]
def _csv(name, default=""):
return [item.strip() for item in os.environ.get(name, default).split(",") if item.strip()]
ALLOWED_HOSTS = _csv('ALLOWED_HOSTS', 'localhost,127.0.0.1,[::1]')
CSRF_TRUSTED_ORIGINS = _csv('CSRF_TRUSTED_ORIGINS')
ALLOWED_HOSTS = _csv("ALLOWED_HOSTS", "localhost,127.0.0.1,[::1]")
CSRF_TRUSTED_ORIGINS = _csv("CSRF_TRUSTED_ORIGINS")
USE_X_FORWARDED_HOST = True
SECURE_PROXY_SSL_HEADER = ('HTTP_X_FORWARDED_PROTO', 'https')
SECURE_PROXY_SSL_HEADER = ("HTTP_X_FORWARDED_PROTO", "https")
DATABASES = {
'default': {
'NAME': os.environ['DB_NAME'],
'USER': os.environ['DB_USER'],
'PASSWORD': os.environ['DB_PASSWORD'],
'HOST': os.environ['DB_HOST'],
'PORT': os.environ.get('DB_PORT', '5432'),
'OPTIONS': {'sslmode': os.environ.get('DB_SSLMODE', 'disable')},
'CONN_MAX_AGE': int(os.environ.get('DB_CONN_MAX_AGE', '300')),
"default": {
"NAME": os.environ["DB_NAME"],
"USER": os.environ["DB_USER"],
"PASSWORD": os.environ["DB_PASSWORD"],
"HOST": os.environ["DB_HOST"],
"PORT": os.environ.get("DB_PORT", "5432"),
"OPTIONS": {"sslmode": os.environ.get("DB_SSLMODE", "disable")},
"CONN_MAX_AGE": int(os.environ.get("DB_CONN_MAX_AGE", "300")),
}
}
REDIS = {
'tasks': {
'HOST': os.environ['REDIS_HOST'],
'PORT': int(os.environ.get('REDIS_PORT', '6379')),
'PASSWORD': os.environ['REDIS_PASSWORD'],
'DATABASE': int(os.environ.get('REDIS_DATABASE', '0')),
'SSL': False,
"tasks": {
"HOST": os.environ["REDIS_HOST"],
"PORT": int(os.environ.get("REDIS_PORT", "6379")),
"PASSWORD": os.environ["REDIS_PASSWORD"],
"DATABASE": int(os.environ.get("REDIS_DATABASE", "0")),
"SSL": False,
},
'caching': {
'HOST': os.environ['REDIS_CACHE_HOST'],
'PORT': int(os.environ.get('REDIS_CACHE_PORT', '6379')),
'PASSWORD': os.environ['REDIS_CACHE_PASSWORD'],
'DATABASE': int(os.environ.get('REDIS_CACHE_DATABASE', '1')),
'SSL': False,
"caching": {
"HOST": os.environ["REDIS_CACHE_HOST"],
"PORT": int(os.environ.get("REDIS_CACHE_PORT", "6379")),
"PASSWORD": os.environ["REDIS_CACHE_PASSWORD"],
"DATABASE": int(os.environ.get("REDIS_CACHE_DATABASE", "1")),
"SSL": False,
},
}
SECRET_KEY = os.environ['SECRET_KEY']
API_TOKEN_PEPPERS = {1: os.environ['API_TOKEN_PEPPER_1']}
TIME_ZONE = os.environ.get('TIME_ZONE', 'UTC')
MEDIA_ROOT = '/opt/netbox/netbox/media'
REPORTS_ROOT = '/opt/netbox/netbox/reports'
SCRIPTS_ROOT = '/opt/netbox/netbox/scripts'
SECRET_KEY = os.environ["SECRET_KEY"]
API_TOKEN_PEPPERS = {1: os.environ["API_TOKEN_PEPPER_1"]}
TIME_ZONE = os.environ.get("TIME_ZONE", "UTC")
MEDIA_ROOT = "/opt/netbox/netbox/media"
REPORTS_ROOT = "/opt/netbox/netbox/reports"
SCRIPTS_ROOT = "/opt/netbox/netbox/scripts"
CENSUS_REPORTING_ENABLED = False
+20 -29
View File
@@ -56,48 +56,39 @@ spec:
startupProbe:
exec:
command:
- /usr/bin/curl
- --fail
- --silent
- --show-error
- --max-time
- "4"
- --header
- "Host: netbox.forust.xyz"
- http://127.0.0.1:8080/login/
- /opt/netbox/venv/bin/python
- -c
- >-
exec /usr/bin/curl --fail --silent --show-error --max-time 4
--header 'Host: netbox.forust.xyz'
http://127.0.0.1:8080/login/ >/dev/null
failureThreshold: 90
periodSeconds: 10
readinessProbe:
exec:
command:
- /usr/bin/curl
- --fail
- --silent
- --show-error
- --max-time
- "4"
- --header
- "Host: netbox.forust.xyz"
- http://127.0.0.1:8080/login/
- /opt/netbox/venv/bin/python
- -c
- >-
exec /usr/bin/curl --fail --silent --show-error --max-time 4
--header 'Host: netbox.forust.xyz'
http://127.0.0.1:8080/login/ >/dev/null
periodSeconds: 10
livenessProbe:
exec:
command:
- /usr/bin/curl
- --fail
- --silent
- --show-error
- --max-time
- "4"
- --header
- "Host: netbox.forust.xyz"
- http://127.0.0.1:8080/login/
- /opt/netbox/venv/bin/python
- -c
- >-
exec /usr/bin/curl --fail --silent --show-error --max-time 4
--header 'Host: netbox.forust.xyz'
http://127.0.0.1:8080/login/ >/dev/null
initialDelaySeconds: 30
periodSeconds: 30
resources:
requests:
cpu: "100m"
memory: "1Gi"
memory: "512Mi"
limits:
cpu: "2"
memory: "2Gi"
@@ -164,7 +155,7 @@ spec:
memory: "256Mi"
limits:
cpu: "1"
memory: "512Mi"
memory: "1Gi"
volumes:
- name: netbox-config
configMap:
+2 -2
View File
@@ -51,8 +51,8 @@ spec:
key: NETRONOME__DB_PASSWORD
resources:
requests:
memory: "64Mi"
memory: "100Mi"
cpu: "100m"
limits:
memory: "256Mi"
memory: "512Mi"
cpu: "500m"
File renamed without changes.
+1 -11
View File
@@ -60,26 +60,16 @@ spec:
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
initialDelaySeconds: 10
periodSeconds: 10
# Generous timeout: on an I/O-bound single node even exec can take
# seconds, and a 1s default kills a healthy postgres mid-recovery.
timeoutSeconds: 5
startupProbe:
exec:
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
# Crash recovery on an I/O-starved single node can fsync for 10+
# minutes; killing postgres mid-recovery restarts the fsync from
# zero and loops forever. 90x10s = 15 minutes of grace.
failureThreshold: 90
failureThreshold: 30
periodSeconds: 10
livenessProbe:
exec:
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
initialDelaySeconds: 30
periodSeconds: 20
# Same I/O reasoning as readiness, plus more misses before a kill:
# restarting postgres on a loaded node only makes recovery longer.
timeoutSeconds: 5
failureThreshold: 5
resources:
requests:
memory: "512Mi"
-27
View File
@@ -27,33 +27,6 @@ spec:
summary: "Pod is crash looping"
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} is in CrashLoopBackOff."
- alert: ContainerOOMKilled
expr: max_over_time(kube_pod_container_status_terminated_reason{reason="OOMKilled"}[15m]) >= 1
for: 5m
labels:
severity: warning
annotations:
summary: "Container was OOMKilled"
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} was killed for exceeding its memory limit. Raise the limit or reduce the workload."
- alert: ContainerRestartingTooOften
expr: max by (namespace, pod, container) (increase(kube_pod_container_status_restarts_total[30m])) > 3
for: 5m
labels:
severity: warning
annotations:
summary: "Container restarting too often"
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} restarted {{ $value }} times in the last 30 minutes."
- alert: PodEvicted
expr: max_over_time(kube_pod_status_reason{reason="Evicted"}[15m]) >= 1
for: 5m
labels:
severity: warning
annotations:
summary: "Pod was evicted"
description: "Pod {{ $labels.namespace }}/{{ $labels.pod }} was evicted, usually for node disk or memory pressure."
- alert: PersistentVolumeClaimFillingUp
expr: kubelet_volume_stats_available_bytes / kubelet_volume_stats_capacity_bytes < 0.15
for: 15m
+2 -77
View File
@@ -26,25 +26,6 @@ grafana:
service:
port: 80
# p95 391M, observed max 1046M with no limit at all. Request is set at p95 so the
# scheduler sees reality; the limit is a manual exception above the 1.3x max
# formula, because a single query burst reached 1046M.
resources:
requests:
memory: "416Mi"
cpu: 100m
limits:
memory: "1536Mi"
# One block covers both the dashboards and datasources sidecars (p95 91M / 80M).
sidecar:
resources:
requests:
memory: "96Mi"
cpu: 10m
limits:
memory: "192Mi"
additionalDataSources:
- name: Loki
type: loki
@@ -67,21 +48,13 @@ prometheus:
storage: 40Gi
resources:
requests:
memory: "768Mi"
memory: "700Mi"
cpu: 200m
limits:
memory: "2560Mi"
memory: "2Gi"
alertmanager:
alertmanagerSpec:
configSecret: alertmanager-config
# p95 66M, max 68M. Silences and notification state live here, so the request
# stays above p95 to keep the pod out of the eviction candidates.
resources:
requests:
memory: "96Mi"
cpu: 10m
limits:
memory: "192Mi"
storage:
volumeClaimTemplate:
spec:
@@ -93,54 +66,6 @@ alertmanager:
requests:
storage: 20Gi
# p95 103M, max 107M, and it grows with the number of cluster objects.
kube-state-metrics:
# Values key is the dependency name from Chart.yaml, not the `kubeStateMetrics`
# condition key. Setting resources under `kubeStateMetrics:` is silently ignored.
resources:
requests:
memory: "128Mi"
cpu: 50m
limits:
memory: "256Mi"
# p95 39M, max 40M. One per node, so it scales with node count.
prometheus-node-exporter:
resources:
requests:
memory: "32Mi"
cpu: 20m
limits:
memory: "128Mi"
# p95 75M, max 75M. Creates and reconciles every PrometheusRule in the cluster.
prometheusOperator:
resources:
requests:
memory: "96Mi"
cpu: 50m
limits:
memory: "192Mi"
# config-reloader sidecars (p95 33M, max 43M) are not covered: the chart does not
# template `prometheusSpec.configReloader.resources`, so there is no values key
# for them. They keep shipping with requests only.
defaultRules:
disabled:
CPUThrottlingHigh: true
KubeControllerManagerDown: true
KubeSchedulerDown: true
KubeEtcdDown: true
KubeEtcdHighCommitDurations: true
# k0s runs controller-manager/scheduler/etcd internally, not as pods with
# component=kube-controller-manager/kube-scheduler/k8s-app=kube-etcd labels.
# Their Services get no endpoints, so the targets are permanently down.
# kube-proxy and kubelet have endpoints on k0s, keep them enabled.
kubeControllerManager:
enabled: false
kubeScheduler:
enabled: false
kubeEtcd:
enabled: false
+2 -2
View File
@@ -66,10 +66,10 @@ spec:
periodSeconds: 30
resources:
requests:
memory: "160Mi"
memory: "128Mi"
cpu: "100m"
limits:
memory: "384Mi"
memory: "512Mi"
cpu: "500m"
volumes:
- name: rackpeek-config
-23
View File
@@ -1,23 +0,0 @@
# Pinned chart: stakater/reloader 2.2.17 (app v1.4.22).
# Deployed by the deploy workflow, namespace reloader.
# Restarts pods when a ConfigMap or Secret they consume changes. Opt-in per workload
# via the reloader.stakater.com/auto: "true" pod annotation; watchGlobally because
# the workloads that need it are spread across a few dozen namespaces.
reloader:
watchGlobally: true
deployment:
replicas: 1
# The chart defaults to no requests or limits, so the pod is evictable under node
# pressure and the restarts go with it.
# Memory was raised from 64Mi: measured p95 over 7 days is 73M, so the pod was
# running above its own request and sitting in the eviction candidates. This pod
# is the one that restarts every other pod, so it must not be evicted.
resources:
requests:
cpu: "10m"
memory: "96Mi"
limits:
cpu: "100m"
memory: "192Mi"
+69
View File
@@ -0,0 +1,69 @@
{
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
"extends": ["config:recommended"],
"enabledManagers": [
"dockerfile",
"docker-compose",
"kubernetes",
"helm-values",
"custom.regex"
],
"helm-values": {
"managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"]
},
"kubernetes": {
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
},
"customManagers": [
{
"customType": "regex",
"description": "singlesource: playwright npm version pinned in npx command (k8s + compose)",
"managerFilePatterns": [
"/^edu_master/k8s/playwright\\.yaml$/",
"/^edu_master/compose\\.yaml$/"
],
"matchStrings": ["playwright@(?<currentValue>\\d+\\.\\d+\\.\\d+)"],
"datasourceTemplate": "npm",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: PLAYWRIGHT_VERSION file",
"managerFilePatterns": ["/^edu_master/PLAYWRIGHT_VERSION$/"],
"matchStrings": ["^(?<currentValue>\\d+\\.\\d+\\.\\d+)$"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright"
}
],
"packageRules": [
{
"description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"groupName": "playwright singlesource",
"groupSlug": "playwright"
},
{
"description": "playwright must not automerge - version skew breaks WS handshake (checker.py:1523 vs playwright.yaml:20)",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"automerge": false
},
{
"description": "Keep private homelab images unchanged",
"matchDatasources": ["docker"],
"matchPackageNames": ["/gcr\\.forust\\.xyz\\/forust\\/.+/"],
"enabled": false
},
{
"description": "Require approval for major upgrades",
"matchUpdateTypes": ["major"],
"dependencyDashboardApproval": true,
"automerge": false
},
{
"description": "Group container patch updates",
"matchDatasources": ["docker"],
"matchUpdateTypes": ["patch"],
"groupName": "container patch updates"
}
]
}
+3 -32
View File
@@ -26,17 +26,15 @@ from Git and must be applied separately after every new cluster.
Run it immediately instead of waiting for the six-hour schedule.
Two options, both use the same `renovate/renovate.json`:
Two options, both use the same `renovate/config.js`:
```sh
kubectl create job --from=cronjob/renovate renovate-manual-$(date +%s) -n renovate
```
or the `renovate-run` Actions workflow (Actions tab → `renovate-run` →
Run workflow). It runs the same image as the CronJob on the self-hosted runner
via Docker — the tag is read out of `renovate/k8s/cronjob.yaml` at run time
rather than hardcoded, so the two cannot drift apart. Required Actions secrets
(repo or org settings):
Run workflow). It runs `renovate/renovate:44.103.0` on the self-hosted
runner via Docker. Required Actions secrets (repo or org settings):
- `RENOVATE_TOKEN` — renovate-bot PAT (repository + issue read/write).
- `RENOVATE_GITHUB_COM_TOKEN` — optional, for changelogs and GitHub rate limits.
@@ -66,33 +64,6 @@ docker compose -f renovate-compose.yaml run --rm renovate
The Compose file is intentionally named `renovate-compose.yaml`, so the
repository's automatic deployment discovery does not start it accidentally.
## Configuration
`renovate/renovate.json` is the single source of truth. The Compose file and the
`renovate-run` workflow mount that file directly.
A ConfigMap cannot read from the repository, so the CronJob needs the config
inlined. `renovate/k8s/configmap.yaml` is therefore a **generated** copy:
```sh
.gitea/workflows/sync-renovate-configmap.sh # regenerate after editing
.gitea/workflows/sync-renovate-configmap.sh --check # fail if out of date
```
The `renovate-ci` workflow runs the `--check` form on every PR and push, so a
config edit that forgets to regenerate the ConfigMap cannot be merged.
Beyond images, `customManagers` in the config track:
- Helm chart versions pinned in `.gitea/workflows/deploy-lib.sh`. The built-in
`helmv3` manager only reads `Chart.yaml` and `helm-values` only reads values
files, so neither sees a version written into a `helm upgrade` command —
these are declared as `custom.regex` managers against the `helm` datasource.
- CI linter versions in `.gitea/workflows/tool-versions.env`.
The Renovate image tag is deliberately _not_ in `tool-versions.env`:
`renovate/k8s/cronjob.yaml` owns it, and the workflows read it from there.
## How updates flow
Renovate scans both `compose.yaml` files and Kubernetes manifests, opens a
+44
View File
@@ -0,0 +1,44 @@
module.exports = {
platform: 'gitea',
endpoint: process.env.RENOVATE_ENDPOINT || 'https://gitea.forust.xyz/api/v1',
enabledManagers: ['docker-compose', 'kubernetes', 'helm-values'],
'helm-values': {
managerFilePatterns: ['/k8s/.+values\\.ya?ml$/'],
},
kubernetes: {
managerFilePatterns: ['/k8s/.+\\.ya?ml$/'],
},
repositories: (process.env.RENOVATE_REPOSITORIES || '')
.split(',')
.map((repository) => repository.trim())
.filter(Boolean),
onboarding: false,
requireConfig: 'optional',
autodiscover: false,
dependencyDashboard: true,
prCreation: 'immediate',
labels: ['dependencies', 'automated'],
extends: [
'config:recommended',
':dependencyDashboard',
],
packageRules: [
{
description: 'Do not update private homelab images',
matchDatasources: ['docker'],
matchPackageNames: ['/gcr\\.forust\\.xyz\\/forust\\/.+/'],
enabled: false,
},
{
description: 'Keep major upgrades manual',
matchUpdateTypes: ['major'],
dependencyDashboardApproval: true,
automerge: false,
},
{
description: 'Group patch updates',
matchUpdateTypes: ['patch'],
groupName: 'container patch updates',
},
],
};
+36 -196
View File
@@ -1,211 +1,51 @@
# GENERATED FILE - do not edit by hand.
# Source: renovate/renovate.json
# Regenerate: .gitea/workflows/sync-renovate-configmap.sh
# Verify: .gitea/workflows/sync-renovate-configmap.sh --check
apiVersion: v1
kind: ConfigMap
metadata:
name: renovate-config
namespace: renovate
data:
renovate.json: |
{
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
"extends": ["config:recommended", ":dependencyDashboard"],
"enabledManagers": ["dockerfile", "docker-compose", "kubernetes", "helm-values", "custom.regex"],
"onboarding": false,
"requireConfig": "optional",
"autodiscover": false,
"dependencyDashboard": true,
"prCreation": "immediate",
"labels": ["dependencies", "automated"],
"helm-values": {
"managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"]
config.js: |
module.exports = {
platform: 'gitea',
endpoint: process.env.RENOVATE_ENDPOINT || 'https://gitea.forust.xyz/api/v1',
enabledManagers: ['docker-compose', 'kubernetes', 'helm-values'],
'helm-values': {
managerFilePatterns: ['/k8s/.+values\\.ya?ml$/'],
},
"kubernetes": {
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
kubernetes: {
managerFilePatterns: ['/k8s/.+\\.ya?ml$/'],
},
"customManagers": [
{
"customType": "regex",
"description": "singlesource: playwright npm version pinned in npx command (k8s + compose)",
"managerFilePatterns": ["^edu_master/k8s/playwright\\.yaml$", "^edu_master/compose\\.yaml$"],
"matchStrings": ["playwright@(?<currentValue>\\d+\\.\\d+\\.\\d+)"],
"datasourceTemplate": "npm",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: PLAYWRIGHT_VERSION file",
"managerFilePatterns": ["^edu_master/PLAYWRIGHT_VERSION$"],
"matchStrings": ["^(?<currentValue>\\d+\\.\\d+\\.\\d+)$"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "kube-prometheus-stack chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|prometheus-community/kube-prometheus-stack\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "kube-prometheus-stack",
"registryUrlTemplate": "https://prometheus-community.github.io/helm-charts"
},
{
"customType": "regex",
"description": "grafana/loki chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|grafana/loki\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "loki",
"registryUrlTemplate": "https://grafana.github.io/helm-charts"
},
{
"customType": "regex",
"description": "grafana/alloy chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|grafana/alloy\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "alloy",
"registryUrlTemplate": "https://grafana.github.io/helm-charts"
},
{
"customType": "regex",
"description": "actionlint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)ACTIONLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "rhysd/actionlint"
},
{
"customType": "regex",
"description": "shellcheck version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)SHELLCHECK_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "koalaman/shellcheck"
},
{
"customType": "regex",
"description": "kubeconform version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)KUBECONFORM_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "yannh/kubeconform"
},
{
"customType": "regex",
"description": "uv version used to build the pytest venv",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)UV_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "astral-sh/uv"
},
{
"customType": "regex",
"description": "prettier version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)PRETTIER_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "npm",
"depNameTemplate": "prettier"
},
{
"customType": "regex",
"description": "ruff version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)RUFF_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "ruff"
},
{
"customType": "regex",
"description": "pip-audit version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)PIP_AUDIT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "pip-audit"
},
{
"customType": "regex",
"description": "yamllint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)YAMLLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "yamllint"
},
{
"customType": "regex",
"description": "hadolint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)HADOLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "hadolint/hadolint"
},
{
"customType": "regex",
"description": "node version the ci workflow runs npm with",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)NODE_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "node",
"depNameTemplate": "node"
},
{
"customType": "regex",
"description": "stakater/reloader chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|stakater/reloader\\|reloader\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "reloader",
"registryUrlTemplate": "https://stakater.github.io/stakater-charts"
}
repositories: (process.env.RENOVATE_REPOSITORIES || '')
.split(',')
.map((repository) => repository.trim())
.filter(Boolean),
onboarding: false,
requireConfig: 'optional',
autodiscover: false,
dependencyDashboard: true,
prCreation: 'immediate',
labels: ['dependencies', 'automated'],
extends: [
'config:recommended',
':dependencyDashboard',
],
"packageRules": [
packageRules: [
{
"description": "Keep private homelab images unchanged",
"matchDatasources": ["docker"],
"matchPackageNames": ["/gcr\\.forust\\.xyz\\/forust\\/.+/"],
"enabled": false
description: 'Do not update private homelab images',
matchDatasources: ['docker'],
matchPackageNames: ['/gcr\\.forust\\.xyz\\/forust\\/.+/'],
enabled: false,
},
{
"description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"groupName": "playwright singlesource",
"groupSlug": "playwright"
description: 'Keep major upgrades manual',
matchUpdateTypes: ['major'],
dependencyDashboardApproval: true,
automerge: false,
},
{
"description": "playwright must not automerge - version skew breaks the WS handshake (checker.py:1523 vs playwright.yaml:20)",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"automerge": false
description: 'Group patch updates',
matchUpdateTypes: ['patch'],
groupName: 'container patch updates',
},
{
"description": "Renovate updates itself in lockstep across the CronJob and the Compose file",
"matchPackageNames": ["renovate/renovate"],
"groupName": "renovate self-update",
"automerge": false
},
{
"description": "CI runs npm on the node the panel image is built from - the NODE_VERSION pin in tool-versions.env and node:22-alpine in the Dockerfile are the same dependency and move as one",
"matchPackageNames": ["node"],
"groupName": "node runtime",
"groupSlug": "node",
"automerge": false
},
{
"description": "Helm chart bumps change PVC fields and admission behaviour, keep them reviewable",
"matchDatasources": ["helm"],
"automerge": false
},
{
"description": "Require approval for major upgrades",
"matchUpdateTypes": ["major"],
"dependencyDashboardApproval": true,
"automerge": false
},
{
"description": "Group container patch updates",
"matchDatasources": ["docker"],
"matchUpdateTypes": ["patch"],
"groupName": "container patch updates"
}
]
}
],
};
+4 -7
View File
@@ -10,16 +10,13 @@ spec:
failedJobsHistoryLimit: 3
jobTemplate:
spec:
# Reap finished job pods (including manual `create job --from` runs,
# which history limits never delete). Keeps a day for debugging.
ttlSecondsAfterFinished: 86400
backoffLimit: 1
template:
spec:
restartPolicy: Never
containers:
- name: renovate
image: renovate/renovate:44.115.13
image: renovate/renovate:44.115.9
env:
- name: RENOVATE_PLATFORM
value: gitea
@@ -39,7 +36,7 @@ spec:
name: renovate-secrets
key: RENOVATE_REPOSITORIES
- name: RENOVATE_CONFIG_FILE
value: /opt/renovate/renovate.json
value: /opt/renovate/config.js
- name: RENOVATE_BASE_DIR
value: /tmp/renovate
- name: RENOVATE_GITHUB_COM_TOKEN
@@ -52,8 +49,8 @@ spec:
value: info
volumeMounts:
- name: config
mountPath: /opt/renovate/renovate.json
subPath: renovate.json
mountPath: /opt/renovate/config.js
subPath: config.js
readOnly: true
volumes:
- name: config
+3 -5
View File
@@ -1,8 +1,6 @@
services:
renovate:
# Kept in step with renovate/k8s/cronjob.yaml by the "renovate self-update"
# package rule in renovate/renovate.json.
image: renovate/renovate:44.115.9
image: renovate/renovate:44.103.0
container_name: renovate
restart: "no"
env_file:
@@ -12,8 +10,8 @@ services:
RENOVATE_ENDPOINT: ${RENOVATE_ENDPOINT:?set RENOVATE_ENDPOINT}
RENOVATE_TOKEN: ${RENOVATE_TOKEN:?set RENOVATE_TOKEN}
RENOVATE_REPOSITORIES: ${RENOVATE_REPOSITORIES:?set RENOVATE_REPOSITORIES}
RENOVATE_CONFIG_FILE: /opt/renovate/renovate.json
RENOVATE_CONFIG_FILE: /opt/renovate/config.js
RENOVATE_BASE_DIR: /tmp/renovate
LOG_LEVEL: ${LOG_LEVEL:-info}
volumes:
- ./renovate.json:/opt/renovate/renovate.json:ro
- ./config.js:/opt/renovate/config.js:ro
-200
View File
@@ -1,200 +0,0 @@
{
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
"extends": ["config:recommended", ":dependencyDashboard"],
"enabledManagers": ["dockerfile", "docker-compose", "kubernetes", "helm-values", "custom.regex"],
"onboarding": false,
"requireConfig": "optional",
"autodiscover": false,
"dependencyDashboard": true,
"prCreation": "immediate",
"labels": ["dependencies", "automated"],
"helm-values": {
"managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"]
},
"kubernetes": {
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
},
"customManagers": [
{
"customType": "regex",
"description": "singlesource: playwright npm version pinned in npx command (k8s + compose)",
"managerFilePatterns": ["^edu_master/k8s/playwright\\.yaml$", "^edu_master/compose\\.yaml$"],
"matchStrings": ["playwright@(?<currentValue>\\d+\\.\\d+\\.\\d+)"],
"datasourceTemplate": "npm",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: PLAYWRIGHT_VERSION file",
"managerFilePatterns": ["^edu_master/PLAYWRIGHT_VERSION$"],
"matchStrings": ["^(?<currentValue>\\d+\\.\\d+\\.\\d+)$"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "kube-prometheus-stack chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|prometheus-community/kube-prometheus-stack\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "kube-prometheus-stack",
"registryUrlTemplate": "https://prometheus-community.github.io/helm-charts"
},
{
"customType": "regex",
"description": "grafana/loki chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|grafana/loki\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "loki",
"registryUrlTemplate": "https://grafana.github.io/helm-charts"
},
{
"customType": "regex",
"description": "grafana/alloy chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|grafana/alloy\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "alloy",
"registryUrlTemplate": "https://grafana.github.io/helm-charts"
},
{
"customType": "regex",
"description": "actionlint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)ACTIONLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "rhysd/actionlint"
},
{
"customType": "regex",
"description": "shellcheck version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)SHELLCHECK_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "koalaman/shellcheck"
},
{
"customType": "regex",
"description": "kubeconform version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)KUBECONFORM_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "yannh/kubeconform"
},
{
"customType": "regex",
"description": "uv version used to build the pytest venv",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)UV_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "astral-sh/uv"
},
{
"customType": "regex",
"description": "prettier version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)PRETTIER_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "npm",
"depNameTemplate": "prettier"
},
{
"customType": "regex",
"description": "ruff version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)RUFF_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "ruff"
},
{
"customType": "regex",
"description": "pip-audit version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)PIP_AUDIT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "pip-audit"
},
{
"customType": "regex",
"description": "yamllint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)YAMLLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "yamllint"
},
{
"customType": "regex",
"description": "hadolint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)HADOLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "hadolint/hadolint"
},
{
"customType": "regex",
"description": "node version the ci workflow runs npm with",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)NODE_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "node",
"depNameTemplate": "node"
},
{
"customType": "regex",
"description": "stakater/reloader chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|stakater/reloader\\|reloader\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "reloader",
"registryUrlTemplate": "https://stakater.github.io/stakater-charts"
}
],
"packageRules": [
{
"description": "Keep private homelab images unchanged",
"matchDatasources": ["docker"],
"matchPackageNames": ["/gcr\\.forust\\.xyz\\/forust\\/.+/"],
"enabled": false
},
{
"description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"groupName": "playwright singlesource",
"groupSlug": "playwright"
},
{
"description": "playwright must not automerge - version skew breaks the WS handshake (checker.py:1523 vs playwright.yaml:20)",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"automerge": false
},
{
"description": "Renovate updates itself in lockstep across the CronJob and the Compose file",
"matchPackageNames": ["renovate/renovate"],
"groupName": "renovate self-update",
"automerge": false
},
{
"description": "CI runs npm on the node the panel image is built from - the NODE_VERSION pin in tool-versions.env and node:22-alpine in the Dockerfile are the same dependency and move as one",
"matchPackageNames": ["node"],
"groupName": "node runtime",
"groupSlug": "node",
"automerge": false
},
{
"description": "Helm chart bumps change PVC fields and admission behaviour, keep them reviewable",
"matchDatasources": ["helm"],
"automerge": false
},
{
"description": "Require approval for major upgrades",
"matchUpdateTypes": ["major"],
"dependencyDashboardApproval": true,
"automerge": false
},
{
"description": "Group container patch updates",
"matchDatasources": ["docker"],
"matchUpdateTypes": ["patch"],
"groupName": "container patch updates"
}
]
}
+1 -1
View File
@@ -3,7 +3,7 @@
services:
core:
container_name: searxng-core
image: docker.io/searxng/searxng:${SEARXNG_VERSION:-2026.9.25-12f8b6515}
image: docker.io/searxng/searxng:${SEARXNG_VERSION:-2026.09.13-d4ce87c23}
restart: unless-stopped
# ports:
# - ${SEARXNG_PORT:-8080}
+3 -4
View File
@@ -27,7 +27,7 @@ spec:
spec:
containers:
- name: searxng
image: docker.io/searxng/searxng:2026.9.25-12f8b6515
image: docker.io/searxng/searxng:2026.09.13-d4ce87c23
envFrom:
- configMapRef:
name: searxng-config
@@ -38,13 +38,12 @@ spec:
volumeMounts:
- name: cache
mountPath: /var/cache/searxng
# p95 134M, max 145M over 7 days. Was 300Mi/700Mi.
resources:
requests:
memory: "160Mi"
memory: "300Mi"
cpu: "30m"
limits:
memory: "512Mi"
memory: "700Mi"
cpu: "500m"
volumes:
- name: cache
-7
View File
@@ -30,13 +30,6 @@ spec:
containers:
- name: valkey
image: docker.io/valkey/valkey:9.1.2-alpine
# p95 15M, max 17M, no limit before. Matches the netbox valkey pod.
resources:
requests:
cpu: "50m"
memory: "64Mi"
limits:
memory: "256Mi"
command:
- valkey-server
- --save
+2 -2
View File
@@ -38,10 +38,10 @@ spec:
mountPath: /app/data
resources:
requests:
memory: "160Mi"
memory: "128Mi"
cpu: "100m"
limits:
memory: "384Mi"
memory: "256Mi"
cpu: "300m"
livenessProbe:
httpGet:
-11
View File
@@ -1,11 +0,0 @@
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: traefik-plugins
namespace: traefik
spec:
accessModes: ["ReadWriteOnce"]
storageClassName: local-path
resources:
requests:
storage: 256Mi
-7
View File
@@ -28,13 +28,6 @@ updateStrategy:
deployment:
enabled: true
additionalVolumes:
- name: plugins
persistentVolumeClaim:
claimName: traefik-plugins
additionalVolumeMounts:
- name: plugins
mountPath: /plugins-storage
providers:
kubernetesIngress:
+1 -2
View File
@@ -28,10 +28,9 @@ spec:
containers:
- name: uptime-kuma
image: louislam/uptime-kuma:2.5.5
# p95 469M, max 471M over 7 days. Was a 3Gi limit on 469M of real use.
resources:
limits:
memory: "1Gi"
memory: "3Gi"
cpu: "1"
requests:
memory: "512Mi"
+2 -2
View File
@@ -3,7 +3,7 @@ services:
build:
context: .
dockerfile: Dockerfile
image: gcr.forust.xyz/forust/userbot:prod
image: gcr.forust.xyz/forust/userbot:latest
pull_policy: build
restart: unless-stopped
env_file:
@@ -18,7 +18,7 @@ services:
- 1.1.1.1
anna:
image: gcr.forust.xyz/forust/userbot:prod
image: gcr.forust.xyz/forust/userbot:latest
pull_policy: build
depends_on:
- forust
+3 -8
View File
@@ -198,7 +198,8 @@ spec:
serviceAccountName: userbot-panel
containers:
- name: userbot-panel
image: gcr.forust.xyz/forust/userbot-panel:prod
image: gcr.forust.xyz/forust/userbot-panel:latest
imagePullPolicy: Always
ports:
- name: http
containerPort: 8080
@@ -208,13 +209,7 @@ spec:
- name: USERBOT_LEGACY_NAMESPACES
value: default
- name: USERBOT_IMAGE
# The build pushes main/prod only. The deploy resolves every gcr ref in
# this file, so a :latest here aborts the whole apply as unresolvable.
# Deliberately left on the tag: render_pinned only rewrites plain
# `image:` lines to a digest, and this ref is what the panel injects
# into the per-instance Deployments it creates. Instances track prod
# rather than the panel's own resolved digest.
value: gcr.forust.xyz/forust/userbot:prod
value: gcr.forust.xyz/forust/userbot:latest
- name: USERBOT_STORAGE_CLASS
value: local-path-retain
- name: USERBOT_DOWNLOADS_HOST_PATH
+4 -2
View File
@@ -24,7 +24,8 @@ spec:
spec:
containers:
- name: forust-userbot
image: gcr.forust.xyz/forust/userbot:prod
image: gcr.forust.xyz/forust/userbot:latest
imagePullPolicy: Always
resources:
limits:
memory: "1.5Gi"
@@ -95,7 +96,8 @@ spec:
spec:
containers:
- name: anna-userbot
image: gcr.forust.xyz/forust/userbot:prod
image: gcr.forust.xyz/forust/userbot:latest
imagePullPolicy: Always
resources:
limits:
memory: "1.5Gi"
+18 -14
View File
@@ -51,11 +51,11 @@ class TelegramAuthService:
if len(self.flows) >= self.max_flows:
raise PanelError(
429,
'Too many pending authorization flows; try again later',
"Too many pending authorization flows; try again later",
)
flow_id = secrets.token_urlsafe(24)
telegram = Client(
f'auth-{flow_id}',
f"auth-{flow_id}",
api_id=account.api_id,
api_hash=account.api_hash,
in_memory=True,
@@ -117,7 +117,7 @@ class TelegramAuthService:
account: StringSessionStart,
) -> AuthorizedAccount:
telegram = Client(
f'validate-{secrets.token_urlsafe(12)}',
f"validate-{secrets.token_urlsafe(12)}",
api_id=account.api_id,
api_hash=account.api_hash,
session_string=account.session_string,
@@ -148,7 +148,7 @@ class TelegramAuthService:
async with self._lock:
flow = self.flows.get(flow_id)
if flow is None:
raise PanelError(410, 'Authorization flow expired; start again')
raise PanelError(410, "Authorization flow expired; start again")
return flow
async def _finish(self, flow: AuthFlow) -> AuthorizedAccount:
@@ -171,7 +171,11 @@ class TelegramAuthService:
async def _cleanup_expired(self) -> None:
now = datetime.now(UTC)
async with self._lock:
expired = [self.flows.pop(flow_id) for flow_id, flow in list(self.flows.items()) if flow.expires_at <= now]
expired = [
self.flows.pop(flow_id)
for flow_id, flow in list(self.flows.items())
if flow.expires_at <= now
]
if expired:
await asyncio.gather(
*(self._disconnect(flow.client) for flow in expired),
@@ -187,19 +191,19 @@ class TelegramAuthService:
@staticmethod
def _translate(exc: Exception, *, session: bool = False) -> PanelError:
if isinstance(exc, FloodWait):
return PanelError(429, f'Telegram rate limit; retry in {exc.value} seconds')
return PanelError(429, f"Telegram rate limit; retry in {exc.value} seconds")
if isinstance(exc, ApiIdInvalid):
return PanelError(422, 'Telegram API ID or API Hash is invalid')
return PanelError(422, "Telegram API ID or API Hash is invalid")
if isinstance(exc, PhoneNumberInvalid):
return PanelError(422, 'Phone number is invalid')
return PanelError(422, "Phone number is invalid")
if isinstance(exc, PhoneCodeInvalid):
return PanelError(422, 'Telegram code is invalid')
return PanelError(422, "Telegram code is invalid")
if isinstance(exc, PhoneCodeExpired):
return PanelError(410, 'Telegram code expired; start again')
return PanelError(410, "Telegram code expired; start again")
if isinstance(exc, PasswordHashInvalid):
return PanelError(422, '2FA password is invalid')
return PanelError(422, "2FA password is invalid")
if session and isinstance(exc, (Unauthorized, RPCError)):
return PanelError(422, 'StringSession is invalid or expired')
return PanelError(422, "StringSession is invalid or expired")
if isinstance(exc, RPCError):
return PanelError(422, 'Telegram rejected the authorization request')
return PanelError(503, 'Telegram authorization is unavailable')
return PanelError(422, "Telegram rejected the authorization request")
return PanelError(503, "Telegram authorization is unavailable")
+20 -18
View File
@@ -7,37 +7,39 @@ from pathlib import Path
@dataclass(frozen=True)
class Settings:
namespace: str = os.environ.get('USERBOT_NAMESPACE', 'userbot')
namespace: str = os.environ.get("USERBOT_NAMESPACE", "userbot")
legacy_namespaces: tuple[str, ...] = tuple(
value.strip() for value in os.environ.get('USERBOT_LEGACY_NAMESPACES', 'default').split(',') if value.strip()
value.strip()
for value in os.environ.get("USERBOT_LEGACY_NAMESPACES", "default").split(",")
if value.strip()
)
image: str = os.environ.get(
'USERBOT_IMAGE',
'gcr.forust.xyz/forust/userbot:latest',
"USERBOT_IMAGE",
"gcr.forust.xyz/forust/userbot:latest",
)
common_secret: str = os.environ.get(
'USERBOT_COMMON_SECRET',
'userbot-common-secrets',
"USERBOT_COMMON_SECRET",
"userbot-common-secrets",
)
common_config: str = os.environ.get(
'USERBOT_COMMON_CONFIG',
'userbot-common-config',
"USERBOT_COMMON_CONFIG",
"userbot-common-config",
)
storage_class: str = os.environ.get(
'USERBOT_STORAGE_CLASS',
'local-path-retain',
"USERBOT_STORAGE_CLASS",
"local-path-retain",
)
downloads_host_path: str = os.environ.get(
'USERBOT_DOWNLOADS_HOST_PATH',
'/srv/homelab/userbot/Downloads',
"USERBOT_DOWNLOADS_HOST_PATH",
"/srv/homelab/userbot/Downloads",
)
static_dir: Path = Path(os.environ.get('PANEL_STATIC_DIR', '/app/static'))
auth_ttl_seconds: int = int(os.environ.get('PANEL_AUTH_TTL_SECONDS', '600'))
default_storage: str = os.environ.get('USERBOT_DEFAULT_STORAGE', '1Gi')
default_cpu_limit: str = os.environ.get('USERBOT_DEFAULT_CPU_LIMIT', '300m')
static_dir: Path = Path(os.environ.get("PANEL_STATIC_DIR", "/app/static"))
auth_ttl_seconds: int = int(os.environ.get("PANEL_AUTH_TTL_SECONDS", "600"))
default_storage: str = os.environ.get("USERBOT_DEFAULT_STORAGE", "1Gi")
default_cpu_limit: str = os.environ.get("USERBOT_DEFAULT_CPU_LIMIT", "300m")
default_memory_limit: str = os.environ.get(
'USERBOT_DEFAULT_MEMORY_LIMIT',
'1536Mi',
"USERBOT_DEFAULT_MEMORY_LIMIT",
"1536Mi",
)
+117 -109
View File
@@ -14,18 +14,18 @@ from .models import AccountBase, InstanceSummary
logger = logging.getLogger(__name__)
MANAGED_LABEL = 'app.kubernetes.io/name=userbot'
INSTANCE_LABEL = 'app.kubernetes.io/instance'
MANAGED_BY_LABEL = 'app.kubernetes.io/managed-by'
DISPLAY_ANNOTATION = 'userbot.forust.xyz/display-name'
LEGACY_ANNOTATION = 'userbot.forust.xyz/legacy'
CREDENTIALS_ANNOTATION = 'userbot.forust.xyz/credentials-secret'
PVC_ANNOTATION = 'userbot.forust.xyz/pvc'
RESTART_ANNOTATION = 'userbot.forust.xyz/restarted-at'
MANAGED_LABEL = "app.kubernetes.io/name=userbot"
INSTANCE_LABEL = "app.kubernetes.io/instance"
MANAGED_BY_LABEL = "app.kubernetes.io/managed-by"
DISPLAY_ANNOTATION = "userbot.forust.xyz/display-name"
LEGACY_ANNOTATION = "userbot.forust.xyz/legacy"
CREDENTIALS_ANNOTATION = "userbot.forust.xyz/credentials-secret"
PVC_ANNOTATION = "userbot.forust.xyz/pvc"
RESTART_ANNOTATION = "userbot.forust.xyz/restarted-at"
def _selector(labels: dict[str, str] | None) -> str:
return ','.join(f'{key}={value}' for key, value in (labels or {}).items())
return ",".join(f"{key}={value}" for key, value in (labels or {}).items())
def _as_datetime(value: Any) -> datetime | None:
@@ -33,7 +33,7 @@ def _as_datetime(value: Any) -> datetime | None:
return None
if isinstance(value, datetime):
return value
return getattr(value, 'replace', lambda **_: None)(tzinfo=UTC)
return getattr(value, "replace", lambda **_: None)(tzinfo=UTC)
class KubernetesService:
@@ -55,7 +55,7 @@ class KubernetesService:
except Exception as exc:
raise PanelError(
503,
'No in-cluster or kubeconfig configuration is available',
"No in-cluster or kubeconfig configuration is available",
) from exc
self.core = core or client.CoreV1Api()
self.apps = apps or client.AppsV1Api()
@@ -76,14 +76,14 @@ class KubernetesService:
if exc.status == 404:
raise PanelError(
503,
'Userbot common Secret or ConfigMap is missing in the userbot namespace',
"Userbot common Secret or ConfigMap is missing in the userbot namespace",
) from exc
raise self._api_error(exc, 'Could not verify userbot prerequisites') from exc
raise self._api_error(exc, "Could not verify userbot prerequisites") from exc
def list_instances(
self,
query: str = '',
status: str = '',
query: str = "",
status: str = "",
) -> list[InstanceSummary]:
instances: list[InstanceSummary] = []
for namespace in (self.settings.namespace, *self.settings.legacy_namespaces):
@@ -93,7 +93,7 @@ class KubernetesService:
label_selector=MANAGED_LABEL,
).items
except ApiException as exc:
raise self._api_error(exc, f'Could not list Deployments in {namespace}') from exc
raise self._api_error(exc, f"Could not list Deployments in {namespace}") from exc
instances.extend(self._summarize(namespace, deployment) for deployment in deployments)
query = query.strip().lower()
@@ -103,7 +103,7 @@ class KubernetesService:
for item in instances
if query in item.instance_id.lower()
or query in item.display_name.lower()
or query in (item.pod or '').lower()
or query in (item.pod or "").lower()
]
if status:
instances = [item for item in instances if item.status == status]
@@ -116,9 +116,9 @@ class KubernetesService:
def assert_available(self, instance_id: str) -> None:
names = self._resource_names(instance_id)
checks = (
(self.apps.read_namespaced_deployment, names['deployment'], 'Deployment'),
(self.core.read_namespaced_secret, names['secret'], 'Secret'),
(self.core.read_namespaced_persistent_volume_claim, names['pvc'], 'PVC'),
(self.apps.read_namespaced_deployment, names["deployment"], "Deployment"),
(self.core.read_namespaced_secret, names["secret"], "Secret"),
(self.core.read_namespaced_persistent_volume_claim, names["pvc"], "PVC"),
)
for read, name, kind in checks:
try:
@@ -126,8 +126,8 @@ class KubernetesService:
except ApiException as exc:
if exc.status == 404:
continue
raise self._api_error(exc, f'Could not check {kind} {name}') from exc
raise PanelError(409, f'{kind} {name} already exists')
raise self._api_error(exc, f"Could not check {kind} {name}") from exc
raise PanelError(409, f"{kind} {name} already exists")
def provision(self, account: AccountBase, session_string: str) -> InstanceSummary:
with self._provision_lock:
@@ -139,25 +139,25 @@ class KubernetesService:
self.settings.namespace,
self._secret(account, session_string, names),
)
created.append(('secret', names['secret']))
created.append(("secret", names["secret"]))
self.core.create_namespaced_persistent_volume_claim(
self.settings.namespace,
self._pvc(account, names),
)
created.append(('pvc', names['pvc']))
created.append(("pvc", names["pvc"]))
self.apps.create_namespaced_deployment(
self.settings.namespace,
self._deployment(account, names),
)
created.append(('deployment', names['deployment']))
created.append(("deployment", names["deployment"]))
except ApiException as exc:
self._rollback(created)
if exc.status == 409:
raise PanelError(
409,
f'Instance {account.instance_id} already exists',
f"Instance {account.instance_id} already exists",
) from exc
raise self._api_error(exc, 'Could not create userbot instance') from exc
raise self._api_error(exc, "Could not create userbot instance") from exc
return self.get_instance(account.instance_id)
def scale(self, instance_id: str, replicas: int) -> InstanceSummary:
@@ -166,40 +166,40 @@ class KubernetesService:
self.apps.patch_namespaced_deployment_scale(
deployment.metadata.name,
namespace,
{'spec': {'replicas': replicas}},
{"spec": {"replicas": replicas}},
)
except ApiException as exc:
raise self._api_error(exc, 'Could not scale userbot instance') from exc
raise self._api_error(exc, "Could not scale userbot instance") from exc
return self.get_instance(instance_id)
def restart(self, instance_id: str) -> InstanceSummary:
namespace, deployment = self._find_deployment(instance_id)
if (deployment.spec.replicas or 0) == 0:
raise PanelError(409, 'Stopped instance cannot be restarted')
raise PanelError(409, "Stopped instance cannot be restarted")
timestamp = datetime.now(UTC).isoformat()
try:
self.apps.patch_namespaced_deployment(
deployment.metadata.name,
namespace,
{
'spec': {
'template': {
'metadata': {
'annotations': {RESTART_ANNOTATION: timestamp},
"spec": {
"template": {
"metadata": {
"annotations": {RESTART_ANNOTATION: timestamp},
}
}
}
},
)
except ApiException as exc:
raise self._api_error(exc, 'Could not restart userbot instance') from exc
raise self._api_error(exc, "Could not restart userbot instance") from exc
return self.get_instance(instance_id)
def logs(self, instance_id: str, tail: int = 250) -> str:
namespace, deployment = self._find_deployment(instance_id)
pods = self._pods_for_deployment(namespace, deployment)
if not pods:
raise PanelError(409, 'Userbot Pod is not running')
raise PanelError(409, "Userbot Pod is not running")
pod = pods[0]
container = deployment.spec.template.spec.containers[0].name
try:
@@ -211,22 +211,22 @@ class KubernetesService:
timestamps=True,
)
except ApiException as exc:
raise self._api_error(exc, 'Could not read userbot logs') from exc
raise self._api_error(exc, "Could not read userbot logs") from exc
def delete(self, instance_id: str, *, delete_data: bool) -> None:
namespace, deployment = self._find_deployment(instance_id)
annotations = deployment.metadata.annotations or {}
if annotations.get(LEGACY_ANNOTATION) == 'true' or namespace != self.settings.namespace:
raise PanelError(409, 'Legacy instances cannot be deleted from the panel')
if annotations.get(LEGACY_ANNOTATION) == "true" or namespace != self.settings.namespace:
raise PanelError(409, "Legacy instances cannot be deleted from the panel")
names = self._resource_names(instance_id)
secret_name = annotations.get(CREDENTIALS_ANNOTATION, names['secret'])
pvc_name = annotations.get(PVC_ANNOTATION, names['pvc'])
secret_name = annotations.get(CREDENTIALS_ANNOTATION, names["secret"])
pvc_name = annotations.get(PVC_ANNOTATION, names["pvc"])
operations = [
(
self.apps.delete_namespaced_deployment,
(deployment.metadata.name, namespace),
{'propagation_policy': 'Foreground'},
{"propagation_policy": "Foreground"},
),
(self.core.delete_namespaced_secret, (secret_name, namespace), {}),
]
@@ -243,10 +243,10 @@ class KubernetesService:
delete_resource(*args, **kwargs)
except ApiException as exc:
if exc.status != 404:
raise self._api_error(exc, 'Could not delete userbot instance') from exc
raise self._api_error(exc, "Could not delete userbot instance") from exc
def _find_deployment(self, instance_id: str) -> tuple[str, Any]:
selector = f'{MANAGED_LABEL},{INSTANCE_LABEL}={instance_id}'
selector = f"{MANAGED_LABEL},{INSTANCE_LABEL}={instance_id}"
for namespace in (self.settings.namespace, *self.settings.legacy_namespaces):
try:
items = self.apps.list_namespaced_deployment(
@@ -254,16 +254,16 @@ class KubernetesService:
label_selector=selector,
).items
except ApiException as exc:
raise self._api_error(exc, 'Could not find userbot instance') from exc
raise self._api_error(exc, "Could not find userbot instance") from exc
if items:
return namespace, items[0]
raise PanelError(404, f'Instance {instance_id} does not exist')
raise PanelError(404, f"Instance {instance_id} does not exist")
def _summarize(self, namespace: str, deployment: Any) -> InstanceSummary:
labels = deployment.metadata.labels or {}
annotations = deployment.metadata.annotations or {}
instance_id = labels.get(INSTANCE_LABEL, deployment.metadata.name)
legacy = annotations.get(LEGACY_ANNOTATION) == 'true'
legacy = annotations.get(LEGACY_ANNOTATION) == "true"
pods = self._pods_for_deployment(namespace, deployment)
pod = pods[0] if pods else None
desired = deployment.spec.replicas or 0
@@ -287,21 +287,21 @@ class KubernetesService:
updated_at = pod.status.start_time or pod.metadata.creation_timestamp
if desired == 0:
status = 'stopped'
status = "stopped"
elif reason in {
'CrashLoopBackOff',
'Error',
'ImagePullBackOff',
'ErrImagePull',
'CreateContainerConfigError',
'RunContainerError',
} or (pod is not None and pod.status.phase == 'Failed'):
status = 'error'
"CrashLoopBackOff",
"Error",
"ImagePullBackOff",
"ErrImagePull",
"CreateContainerConfigError",
"RunContainerError",
} or (pod is not None and pod.status.phase == "Failed"):
status = "error"
elif ready and (deployment.status.available_replicas or 0) > 0:
status = 'running'
status = "running"
else:
status = 'pending'
reason = reason or (pod.status.phase if pod is not None else 'Scheduling')
status = "pending"
reason = reason or (pod.status.phase if pod is not None else "Scheduling")
container_spec = deployment.spec.template.spec.containers[0]
limits = (container_spec.resources.limits or {}) if container_spec.resources else {}
@@ -321,8 +321,8 @@ class KubernetesService:
pvc=pvc_name,
storage=storage,
image=container_spec.image,
cpu_limit=limits.get('cpu'),
memory_limit=limits.get('memory'),
cpu_limit=limits.get("cpu"),
memory_limit=limits.get("memory"),
cpu_usage=cpu_usage,
memory_usage=memory_usage,
updated_at=_as_datetime(updated_at),
@@ -338,7 +338,7 @@ class KubernetesService:
label_selector=selector,
).items
except ApiException as exc:
raise self._api_error(exc, 'Could not list userbot Pods') from exc
raise self._api_error(exc, "Could not list userbot Pods") from exc
return sorted(
pods,
key=lambda pod: pod.metadata.creation_timestamp or datetime.min.replace(tzinfo=UTC),
@@ -350,17 +350,17 @@ class KubernetesService:
return None, None
try:
metrics = self.custom.get_namespaced_custom_object(
'metrics.k8s.io',
'v1beta1',
"metrics.k8s.io",
"v1beta1",
namespace,
'pods',
"pods",
pod_name,
)
except (ApiException, AttributeError):
return None, None
containers = metrics.get('containers', [])
cpu = containers[0].get('usage', {}).get('cpu') if containers else None
memory = containers[0].get('usage', {}).get('memory') if containers else None
containers = metrics.get("containers", [])
cpu = containers[0].get("usage", {}).get("cpu") if containers else None
memory = containers[0].get("usage", {}).get("memory") if containers else None
return cpu, memory
def _pvc_storage(self, namespace: str, pvc_name: str | None) -> str | None:
@@ -371,7 +371,7 @@ class KubernetesService:
except ApiException:
return None
requests = pvc.spec.resources.requests or {}
return requests.get('storage')
return requests.get("storage")
@staticmethod
def _deployment_pvc(deployment: Any) -> str | None:
@@ -383,9 +383,9 @@ class KubernetesService:
@staticmethod
def _resource_names(instance_id: str) -> dict[str, str]:
return {
'deployment': f'userbot-{instance_id}',
'secret': f'userbot-{instance_id}-credentials',
'pvc': f'userbot-{instance_id}-data',
"deployment": f"userbot-{instance_id}",
"secret": f"userbot-{instance_id}-credentials",
"pvc": f"userbot-{instance_id}-data",
}
def _metadata(
@@ -398,14 +398,14 @@ class KubernetesService:
name=resource_name,
namespace=self.settings.namespace,
labels={
'app.kubernetes.io/name': 'userbot',
"app.kubernetes.io/name": "userbot",
INSTANCE_LABEL: account.instance_id,
MANAGED_BY_LABEL: 'userbot-panel',
MANAGED_BY_LABEL: "userbot-panel",
},
annotations={
DISPLAY_ANNOTATION: account.display_name,
CREDENTIALS_ANNOTATION: names['secret'],
PVC_ANNOTATION: names['pvc'],
CREDENTIALS_ANNOTATION: names["secret"],
PVC_ANNOTATION: names["pvc"],
},
)
@@ -416,12 +416,12 @@ class KubernetesService:
names: dict[str, str],
) -> client.V1Secret:
return client.V1Secret(
metadata=self._metadata(account, names, names['secret']),
type='Opaque',
metadata=self._metadata(account, names, names["secret"]),
type="Opaque",
string_data={
'API_ID': str(account.api_id),
'API_HASH': account.api_hash,
'STRINGSESSION': session_string,
"API_ID": str(account.api_id),
"API_HASH": account.api_hash,
"STRINGSESSION": session_string,
},
)
@@ -431,12 +431,12 @@ class KubernetesService:
names: dict[str, str],
) -> client.V1PersistentVolumeClaim:
return client.V1PersistentVolumeClaim(
metadata=self._metadata(account, names, names['pvc']),
metadata=self._metadata(account, names, names["pvc"]),
spec=client.V1PersistentVolumeClaimSpec(
access_modes=['ReadWriteOnce'],
access_modes=["ReadWriteOnce"],
storage_class_name=self.settings.storage_class,
resources=client.V1VolumeResourceRequirements(
requests={'storage': account.resources.storage},
requests={"storage": account.resources.storage},
),
),
)
@@ -447,55 +447,63 @@ class KubernetesService:
names: dict[str, str],
) -> client.V1Deployment:
pod_labels = {
'app.kubernetes.io/name': 'userbot',
"app.kubernetes.io/name": "userbot",
INSTANCE_LABEL: account.instance_id,
MANAGED_BY_LABEL: 'userbot-panel',
MANAGED_BY_LABEL: "userbot-panel",
}
container = client.V1Container(
name='userbot',
name="userbot",
image=self.settings.image,
image_pull_policy='Always',
image_pull_policy="Always",
env_from=[
client.V1EnvFromSource(secret_ref=client.V1SecretEnvSource(name=self.settings.common_secret)),
client.V1EnvFromSource(config_map_ref=client.V1ConfigMapEnvSource(name=self.settings.common_config)),
client.V1EnvFromSource(secret_ref=client.V1SecretEnvSource(name=names['secret'])),
client.V1EnvFromSource(
secret_ref=client.V1SecretEnvSource(name=self.settings.common_secret)
),
client.V1EnvFromSource(
config_map_ref=client.V1ConfigMapEnvSource(name=self.settings.common_config)
),
client.V1EnvFromSource(
secret_ref=client.V1SecretEnvSource(name=names["secret"])
),
],
resources=client.V1ResourceRequirements(
requests={'cpu': '80m', 'memory': '512Mi'},
requests={"cpu": "80m", "memory": "512Mi"},
limits={
'cpu': account.resources.cpu_limit,
'memory': account.resources.memory_limit,
"cpu": account.resources.cpu_limit,
"memory": account.resources.memory_limit,
},
),
volume_mounts=[
client.V1VolumeMount(name='data', mount_path='/app/data'),
client.V1VolumeMount(name='downloads', mount_path='/app/downloads'),
client.V1VolumeMount(name="data", mount_path="/app/data"),
client.V1VolumeMount(name="downloads", mount_path="/app/downloads"),
],
)
pod_spec = client.V1PodSpec(
service_account_name='userbot-runtime',
service_account_name="userbot-runtime",
automount_service_account_token=False,
containers=[container],
termination_grace_period_seconds=30,
volumes=[
client.V1Volume(
name='data',
persistent_volume_claim=client.V1PersistentVolumeClaimVolumeSource(claim_name=names['pvc']),
name="data",
persistent_volume_claim=client.V1PersistentVolumeClaimVolumeSource(
claim_name=names["pvc"]
),
),
client.V1Volume(
name='downloads',
name="downloads",
host_path=client.V1HostPathVolumeSource(
path=self.settings.downloads_host_path,
type='DirectoryOrCreate',
type="DirectoryOrCreate",
),
),
],
)
return client.V1Deployment(
metadata=self._metadata(account, names, names['deployment']),
metadata=self._metadata(account, names, names["deployment"]),
spec=client.V1DeploymentSpec(
replicas=1,
strategy=client.V1DeploymentStrategy(type='Recreate'),
strategy=client.V1DeploymentStrategy(type="Recreate"),
selector=client.V1LabelSelector(match_labels=pod_labels),
template=client.V1PodTemplateSpec(
metadata=client.V1ObjectMeta(labels=pod_labels),
@@ -507,9 +515,9 @@ class KubernetesService:
def _rollback(self, created: list[tuple[str, str]]) -> None:
for kind, name in reversed(created):
try:
if kind == 'deployment':
if kind == "deployment":
self.apps.delete_namespaced_deployment(name, self.settings.namespace)
elif kind == 'pvc':
elif kind == "pvc":
self.core.delete_namespaced_persistent_volume_claim(
name,
self.settings.namespace,
@@ -518,7 +526,7 @@ class KubernetesService:
self.core.delete_namespaced_secret(name, self.settings.namespace)
except ApiException as exc:
logger.warning(
'Rollback of %s %s in %s failed: %s',
"Rollback of %s %s in %s failed: %s",
kind,
name,
self.settings.namespace,
@@ -528,9 +536,9 @@ class KubernetesService:
@staticmethod
def _api_error(exc: ApiException, detail: str) -> PanelError:
if exc.status == 403:
return PanelError(503, f'{detail}: Kubernetes RBAC denied the operation')
return PanelError(503, f"{detail}: Kubernetes RBAC denied the operation")
if exc.status == 409:
return PanelError(409, f'{detail}: resource conflict')
return PanelError(409, f"{detail}: resource conflict")
if exc.status == 404:
return PanelError(404, f'{detail}: resource not found')
return PanelError(404, f"{detail}: resource not found")
return PanelError(503, detail)
+38 -38
View File
@@ -30,21 +30,21 @@ async def lifespan(app: FastAPI):
try:
app.state.kubernetes.ensure_prerequisites()
except PanelError as exc:
print(f'WARNING: userbot prerequisites check failed at startup: {exc.detail}')
print(f"WARNING: userbot prerequisites check failed at startup: {exc.detail}")
yield
await app.state.telegram.close()
app = FastAPI(
title='Userbot Kubernetes Control',
version='1.0.0',
title="Userbot Kubernetes Control",
version="1.0.0",
lifespan=lifespan,
)
@app.exception_handler(PanelError)
async def panel_error_handler(_request: Request, exc: PanelError) -> JSONResponse:
return JSONResponse(status_code=exc.status_code, content={'detail': exc.detail})
return JSONResponse(status_code=exc.status_code, content={"detail": exc.detail})
@app.exception_handler(RequestValidationError)
@@ -54,13 +54,13 @@ async def validation_error_handler(
) -> JSONResponse:
errors = [
{
'loc': error.get('loc', ()),
'msg': error.get('msg', 'Invalid value'),
'type': error.get('type', 'value_error'),
"loc": error.get("loc", ()),
"msg": error.get("msg", "Invalid value"),
"type": error.get("type", "value_error"),
}
for error in exc.errors()
]
return JSONResponse(status_code=422, content={'detail': errors})
return JSONResponse(status_code=422, content={"detail": errors})
def kube(request: Request) -> KubernetesService:
@@ -71,7 +71,7 @@ def telegram(request: Request) -> TelegramAuthService:
return request.app.state.telegram
@app.get('/api/health')
@app.get("/api/health")
def health(request: Request) -> dict[str, object]:
service = kube(request)
try:
@@ -80,61 +80,61 @@ def health(request: Request) -> dict[str, object]:
limit=1,
)
except Exception:
return {'ok': False, 'kubernetes': False}
return {'ok': True, 'kubernetes': True}
return {"ok": False, "kubernetes": False}
return {"ok": True, "kubernetes": True}
@app.get('/api/instances', response_model=list[InstanceSummary])
@app.get("/api/instances", response_model=list[InstanceSummary])
def list_instances(
request: Request,
query: str = '',
status: str = '',
query: str = "",
status: str = "",
) -> list[InstanceSummary]:
return kube(request).list_instances(query=query, status=status)
@app.get('/api/instances/{instance_id}', response_model=InstanceSummary)
@app.get("/api/instances/{instance_id}", response_model=InstanceSummary)
def get_instance(instance_id: str, request: Request) -> InstanceSummary:
return kube(request).get_instance(instance_id)
@app.get('/api/instances/{instance_id}/logs')
@app.get("/api/instances/{instance_id}/logs")
def get_logs(
instance_id: str,
request: Request,
tail: Annotated[int, Query(ge=1, le=1000)] = 250,
) -> dict[str, str]:
return {'logs': kube(request).logs(instance_id, tail)}
return {"logs": kube(request).logs(instance_id, tail)}
@app.post('/api/instances/{instance_id}/start', response_model=InstanceSummary)
@app.post("/api/instances/{instance_id}/start", response_model=InstanceSummary)
def start_instance(instance_id: str, request: Request) -> InstanceSummary:
return kube(request).scale(instance_id, 1)
@app.post('/api/instances/{instance_id}/stop', response_model=InstanceSummary)
@app.post("/api/instances/{instance_id}/stop", response_model=InstanceSummary)
def stop_instance(instance_id: str, request: Request) -> InstanceSummary:
return kube(request).scale(instance_id, 0)
@app.post('/api/instances/{instance_id}/restart', response_model=InstanceSummary)
@app.post("/api/instances/{instance_id}/restart", response_model=InstanceSummary)
def restart_instance(instance_id: str, request: Request) -> InstanceSummary:
return kube(request).restart(instance_id)
@app.post('/api/instances/{instance_id}/delete', status_code=204)
@app.post("/api/instances/{instance_id}/delete", status_code=204)
def delete_instance(
instance_id: str,
payload: DeleteRequest,
request: Request,
) -> Response:
if payload.confirmation != instance_id:
raise PanelError(422, 'Type the instance id exactly to confirm deletion')
raise PanelError(422, "Type the instance id exactly to confirm deletion")
kube(request).delete(instance_id, delete_data=payload.delete_data)
return Response(status_code=204)
@app.post('/api/auth/phone/start', response_model=AuthResult)
@app.post("/api/auth/phone/start", response_model=AuthResult)
async def auth_phone_start(
payload: PhoneStart,
request: Request,
@@ -142,10 +142,10 @@ async def auth_phone_start(
service = kube(request)
service.assert_available(payload.instance_id)
flow_id = await telegram(request).start_phone(payload)
return AuthResult(status='code_required', flow_id=flow_id)
return AuthResult(status="code_required", flow_id=flow_id)
@app.post('/api/auth/phone/{flow_id}/code', response_model=AuthResult)
@app.post("/api/auth/phone/{flow_id}/code", response_model=AuthResult)
async def auth_phone_code(
flow_id: str,
payload: CodeSubmit,
@@ -153,12 +153,12 @@ async def auth_phone_code(
) -> AuthResult:
authorized = await telegram(request).submit_code(flow_id, payload.code)
if authorized is None:
return AuthResult(status='password_required', flow_id=flow_id)
return AuthResult(status="password_required", flow_id=flow_id)
instance = kube(request).provision(authorized.account, authorized.session_string)
return AuthResult(status='ready', instance=instance)
return AuthResult(status="ready", instance=instance)
@app.post('/api/auth/phone/{flow_id}/password', response_model=AuthResult)
@app.post("/api/auth/phone/{flow_id}/password", response_model=AuthResult)
async def auth_phone_password(
flow_id: str,
payload: PasswordSubmit,
@@ -166,10 +166,10 @@ async def auth_phone_password(
) -> AuthResult:
authorized = await telegram(request).submit_password(flow_id, payload.password)
instance = kube(request).provision(authorized.account, authorized.session_string)
return AuthResult(status='ready', instance=instance)
return AuthResult(status="ready", instance=instance)
@app.post('/api/auth/string-session', response_model=AuthResult)
@app.post("/api/auth/string-session", response_model=AuthResult)
async def auth_string_session(
payload: StringSessionStart,
request: Request,
@@ -178,28 +178,28 @@ async def auth_string_session(
service.assert_available(payload.instance_id)
authorized = await telegram(request).validate_string_session(payload)
instance = service.provision(authorized.account, authorized.session_string)
return AuthResult(status='ready', instance=instance)
return AuthResult(status="ready", instance=instance)
static_dir = settings.static_dir
assets_dir = static_dir / 'assets'
assets_dir = static_dir / "assets"
if assets_dir.exists():
app.mount('/assets', StaticFiles(directory=assets_dir), name='assets')
app.mount("/assets", StaticFiles(directory=assets_dir), name="assets")
@app.get('/', include_in_schema=False)
@app.get("/", include_in_schema=False)
def index() -> FileResponse:
return FileResponse(static_dir / 'index.html')
return FileResponse(static_dir / "index.html")
@app.get('/{path:path}', include_in_schema=False)
@app.get("/{path:path}", include_in_schema=False)
def spa_fallback(path: str) -> FileResponse:
root = static_dir.resolve()
candidate = (static_dir / path).resolve()
try:
candidate.relative_to(root)
except ValueError:
return FileResponse(static_dir / 'index.html')
return FileResponse(static_dir / "index.html")
if candidate.is_file():
return FileResponse(candidate)
return FileResponse(static_dir / 'index.html')
return FileResponse(static_dir / "index.html")
+12 -12
View File
@@ -5,13 +5,13 @@ from typing import Literal
from pydantic import BaseModel, Field, field_validator
INSTANCE_PATTERN = r'^[a-z0-9](?:[a-z0-9-]{0,38}[a-z0-9])?$'
INSTANCE_PATTERN = r"^[a-z0-9](?:[a-z0-9-]{0,38}[a-z0-9])?$"
class Resources(BaseModel):
storage: str = '1Gi'
cpu_limit: str = '300m'
memory_limit: str = '1536Mi'
storage: str = "1Gi"
cpu_limit: str = "300m"
memory_limit: str = "1536Mi"
class AccountBase(BaseModel):
@@ -21,7 +21,7 @@ class AccountBase(BaseModel):
api_hash: str = Field(min_length=16, max_length=128)
resources: Resources = Field(default_factory=Resources)
@field_validator('display_name', 'api_hash')
@field_validator("display_name", "api_hash")
@classmethod
def strip_text(cls, value: str) -> str:
return value.strip()
@@ -30,12 +30,12 @@ class AccountBase(BaseModel):
class PhoneStart(AccountBase):
phone: str = Field(min_length=7, max_length=32)
@field_validator('phone')
@field_validator("phone")
@classmethod
def normalize_phone(cls, value: str) -> str:
value = value.strip()
if not value.startswith('+'):
raise ValueError('phone must use international format')
if not value.startswith("+"):
raise ValueError("phone must use international format")
return value
@@ -46,10 +46,10 @@ class StringSessionStart(AccountBase):
class CodeSubmit(BaseModel):
code: str = Field(min_length=3, max_length=12)
@field_validator('code')
@field_validator("code")
@classmethod
def normalize_code(cls, value: str) -> str:
return ''.join(value.split())
return "".join(value.split())
class PasswordSubmit(BaseModel):
@@ -67,7 +67,7 @@ class InstanceSummary(BaseModel):
namespace: str
deployment: str
pod: str | None = None
status: Literal['running', 'stopped', 'pending', 'error']
status: Literal["running", "stopped", "pending", "error"]
reason: str | None = None
ready: bool = False
restarts: int = 0
@@ -84,6 +84,6 @@ class InstanceSummary(BaseModel):
class AuthResult(BaseModel):
status: Literal['code_required', 'password_required', 'ready']
status: Literal["code_required", "password_required", "ready"]
flow_id: str | None = None
instance: InstanceSummary | None = None
@@ -23,7 +23,7 @@ class FakeClient:
self.disconnected = True
async def send_code(self, _phone):
return SimpleNamespace(phone_code_hash='hash')
return SimpleNamespace(phone_code_hash="hash")
async def sign_in(self, *_args):
if self.requires_password:
@@ -38,49 +38,49 @@ class FakeClient:
return SimpleNamespace(id=1)
async def export_session_string(self):
return 'exported-session'
return "exported-session"
def phone_payload() -> PhoneStart:
return PhoneStart(
instance_id='test',
display_name='Test',
instance_id="test",
display_name="Test",
api_id=123,
api_hash='0123456789abcdef',
phone='+421900000000',
api_hash="0123456789abcdef",
phone="+421900000000",
)
@pytest.mark.asyncio
async def test_phone_code_success_closes_and_forgets_client(monkeypatch) -> None:
monkeypatch.setattr('app.auth_service.Client', FakeClient)
monkeypatch.setattr("app.auth_service.Client", FakeClient)
auth = TelegramAuthService()
flow_id = await auth.start_phone(phone_payload())
result = await auth.submit_code(flow_id, '12345')
result = await auth.submit_code(flow_id, "12345")
assert result.session_string == 'exported-session'
assert result.session_string == "exported-session"
assert flow_id not in auth.flows
assert FakeClient.instances[-1].disconnected is True
assert FakeClient.instances[-1].kwargs['in_memory'] is True
assert FakeClient.instances[-1].kwargs['no_updates'] is True
assert FakeClient.instances[-1].kwargs["in_memory"] is True
assert FakeClient.instances[-1].kwargs["no_updates"] is True
@pytest.mark.asyncio
async def test_string_session_is_validated_and_closed(monkeypatch) -> None:
monkeypatch.setattr('app.auth_service.Client', FakeClient)
monkeypatch.setattr("app.auth_service.Client", FakeClient)
auth = TelegramAuthService()
payload = StringSessionStart(
instance_id='test',
display_name='Test',
instance_id="test",
display_name="Test",
api_id=123,
api_hash='0123456789abcdef',
session_string='x' * 64,
api_hash="0123456789abcdef",
session_string="x" * 64,
)
result = await auth.validate_string_session(payload)
assert result.session_string == 'exported-session'
assert result.session_string == "exported-session"
assert FakeClient.instances[-1].disconnected is True
@@ -88,7 +88,7 @@ async def test_string_session_is_validated_and_closed(monkeypatch) -> None:
async def test_start_phone_rejects_overflow(monkeypatch) -> None:
from app.errors import PanelError
monkeypatch.setattr('app.auth_service.Client', FakeClient)
monkeypatch.setattr("app.auth_service.Client", FakeClient)
auth = TelegramAuthService(ttl_seconds=3600, max_flows=2)
await auth.start_phone(phone_payload())
@@ -98,4 +98,4 @@ async def test_start_phone_rejects_overflow(monkeypatch) -> None:
await auth.start_phone(phone_payload())
assert error.value.status_code == 429
assert 'Too many pending' in error.value.detail
assert "Too many pending" in error.value.detail
@@ -20,35 +20,35 @@ def service() -> KubernetesService:
def account() -> AccountBase:
return AccountBase(
instance_id='test-account',
display_name='Test Account',
instance_id="test-account",
display_name="Test Account",
api_id=12345,
api_hash='0123456789abcdef0123456789abcdef',
api_hash="0123456789abcdef0123456789abcdef",
)
def test_renders_managed_resources_without_leaking_credentials() -> None:
kube = service()
names = kube._resource_names('test-account')
secret = kube._secret(account(), 'SESSION', names)
names = kube._resource_names("test-account")
secret = kube._secret(account(), "SESSION", names)
pvc = kube._pvc(account(), names)
deployment = kube._deployment(account(), names)
assert secret.string_data == {
'API_ID': '12345',
'API_HASH': '0123456789abcdef0123456789abcdef',
'STRINGSESSION': 'SESSION',
"API_ID": "12345",
"API_HASH": "0123456789abcdef0123456789abcdef",
"STRINGSESSION": "SESSION",
}
assert pvc.spec.storage_class_name == 'local-path-retain'
assert pvc.spec.resources.requests['storage'] == '1Gi'
assert deployment.spec.strategy.type == 'Recreate'
assert pvc.spec.storage_class_name == "local-path-retain"
assert pvc.spec.resources.requests["storage"] == "1Gi"
assert deployment.spec.strategy.type == "Recreate"
assert deployment.spec.replicas == 1
assert deployment.spec.template.spec.automount_service_account_token is False
assert deployment.spec.template.spec.service_account_name == 'userbot-runtime'
assert deployment.spec.template.spec.volumes[1].host_path.path.endswith('/Downloads')
assert deployment.spec.template.spec.service_account_name == "userbot-runtime"
assert deployment.spec.template.spec.volumes[1].host_path.path.endswith("/Downloads")
assert deployment.spec.template.spec.containers[0].resources.requests == {
'cpu': '80m',
'memory': '512Mi',
"cpu": "80m",
"memory": "512Mi",
}
@@ -57,10 +57,10 @@ def test_collision_is_reported_before_auth() -> None:
kube.apps.read_namespaced_deployment.return_value = object()
with pytest.raises(PanelError) as error:
kube.assert_available('test-account')
kube.assert_available("test-account")
assert error.value.status_code == 409
assert 'Deployment' in error.value.detail
assert "Deployment" in error.value.detail
def test_partial_provision_rolls_back_only_created_resources() -> None:
@@ -70,11 +70,11 @@ def test_partial_provision_rolls_back_only_created_resources() -> None:
kube.core.create_namespaced_persistent_volume_claim.side_effect = ApiException(status=500)
with pytest.raises(PanelError):
kube.provision(account(), 'SESSION')
kube.provision(account(), "SESSION")
kube.core.delete_namespaced_secret.assert_called_once_with(
'userbot-test-account-credentials',
'userbot',
"userbot-test-account-credentials",
"userbot",
)
kube.core.delete_namespaced_persistent_volume_claim.assert_not_called()
kube.apps.delete_namespaced_deployment.assert_not_called()
@@ -86,18 +86,18 @@ def test_provision_conflict_reports_existing_instance() -> None:
kube.apps.create_namespaced_deployment.side_effect = ApiException(status=409)
with pytest.raises(PanelError) as error:
kube.provision(account(), 'SESSION')
kube.provision(account(), "SESSION")
assert error.value.status_code == 409
assert 'already exists' in error.value.detail
assert "already exists" in error.value.detail
# Partial resources created before the 409 must be rolled back.
kube.core.delete_namespaced_secret.assert_called_once_with(
'userbot-test-account-credentials',
'userbot',
"userbot-test-account-credentials",
"userbot",
)
kube.core.delete_namespaced_persistent_volume_claim.assert_called_once_with(
'userbot-test-account-data',
'userbot',
"userbot-test-account-data",
"userbot",
)
@@ -105,25 +105,25 @@ def test_delete_retains_pvc_unless_explicitly_requested() -> None:
kube = service()
deployment = SimpleNamespace(
metadata=SimpleNamespace(
name='userbot-test-account',
name="userbot-test-account",
annotations={
'userbot.forust.xyz/credentials-secret': 'credentials',
'userbot.forust.xyz/pvc': 'data',
"userbot.forust.xyz/credentials-secret": "credentials",
"userbot.forust.xyz/pvc": "data",
},
)
)
kube._find_deployment = Mock(return_value=('userbot', deployment))
kube._find_deployment = Mock(return_value=("userbot", deployment))
kube.delete('test-account', delete_data=False)
kube.delete("test-account", delete_data=False)
kube.apps.delete_namespaced_deployment.assert_called_once()
kube.core.delete_namespaced_secret.assert_called_once_with('credentials', 'userbot')
kube.core.delete_namespaced_secret.assert_called_once_with("credentials", "userbot")
kube.core.delete_namespaced_persistent_volume_claim.assert_not_called()
kube.delete('test-account', delete_data=True)
kube.delete("test-account", delete_data=True)
kube.core.delete_namespaced_persistent_volume_claim.assert_called_once_with(
'data',
'userbot',
"data",
"userbot",
)
@@ -131,19 +131,19 @@ def test_legacy_delete_is_blocked() -> None:
kube = service()
deployment = SimpleNamespace(
metadata=SimpleNamespace(
name='forust-userbot-deployment',
annotations={'userbot.forust.xyz/legacy': 'true'},
name="forust-userbot-deployment",
annotations={"userbot.forust.xyz/legacy": "true"},
)
)
kube._find_deployment = Mock(return_value=('default', deployment))
kube._find_deployment = Mock(return_value=("default", deployment))
with pytest.raises(PanelError) as error:
kube.delete('forust', delete_data=False)
kube.delete("forust", delete_data=False)
assert error.value.status_code == 409
def pod(*, ready: bool, phase: str = 'Running', reason: str | None = None):
def pod(*, ready: bool, phase: str = "Running", reason: str | None = None):
waiting = SimpleNamespace(reason=reason) if reason else None
state = SimpleNamespace(waiting=waiting, terminated=None)
status = SimpleNamespace(
@@ -154,25 +154,25 @@ def pod(*, ready: bool, phase: str = 'Running', reason: str | None = None):
start_time=datetime.now(UTC),
)
return SimpleNamespace(
metadata=SimpleNamespace(name='pod-1', creation_timestamp=datetime.now(UTC)),
metadata=SimpleNamespace(name="pod-1", creation_timestamp=datetime.now(UTC)),
status=status,
)
def deployment(replicas: int = 1):
resources = SimpleNamespace(limits={'cpu': '300m', 'memory': '1536Mi'})
container = SimpleNamespace(image='userbot:latest', resources=resources)
resources = SimpleNamespace(limits={"cpu": "300m", "memory": "1536Mi"})
container = SimpleNamespace(image="userbot:latest", resources=resources)
template_spec = SimpleNamespace(containers=[container], volumes=[])
return SimpleNamespace(
metadata=SimpleNamespace(
name='userbot-test-account',
labels={'app.kubernetes.io/instance': 'test-account'},
name="userbot-test-account",
labels={"app.kubernetes.io/instance": "test-account"},
annotations={},
creation_timestamp=datetime.now(UTC),
),
spec=SimpleNamespace(
replicas=replicas,
selector=SimpleNamespace(match_labels={'app': 'test'}),
selector=SimpleNamespace(match_labels={"app": "test"}),
template=SimpleNamespace(spec=template_spec),
),
status=SimpleNamespace(available_replicas=1 if replicas else 0),
@@ -180,13 +180,13 @@ def deployment(replicas: int = 1):
@pytest.mark.parametrize(
('replicas', 'pod_value', 'expected'),
("replicas", "pod_value", "expected"),
[
(0, None, 'stopped'),
(1, pod(ready=True), 'running'),
(1, pod(ready=False, phase='Pending'), 'pending'),
(1, pod(ready=False, reason='CrashLoopBackOff'), 'error'),
(1, pod(ready=False, phase='Failed'), 'error'),
(0, None, "stopped"),
(1, pod(ready=True), "running"),
(1, pod(ready=False, phase="Pending"), "pending"),
(1, pod(ready=False, reason="CrashLoopBackOff"), "error"),
(1, pod(ready=False, phase="Failed"), "error"),
],
)
def test_status_classification(replicas, pod_value, expected) -> None:
@@ -195,6 +195,6 @@ def test_status_classification(replicas, pod_value, expected) -> None:
kube._pvc_storage = Mock(return_value=None)
kube._pod_metrics = Mock(return_value=(None, None))
result = kube._summarize('userbot', deployment(replicas))
result = kube._summarize("userbot", deployment(replicas))
assert result.status == expected
+10 -10
View File
@@ -6,34 +6,34 @@ from pydantic import ValidationError
@pytest.mark.parametrize(
'instance_id',
['Upper', 'has_space', '-leading', 'trailing-', 'x' * 41],
"instance_id",
["Upper", "has_space", "-leading", "trailing-", "x" * 41],
)
def test_invalid_instance_ids(instance_id: str) -> None:
with pytest.raises(ValidationError):
AccountBase(
instance_id=instance_id,
display_name='Test',
display_name="Test",
api_id=1,
api_hash='0123456789abcdef',
api_hash="0123456789abcdef",
)
def test_delete_data_defaults_to_false() -> None:
request = DeleteRequest(confirmation='test')
request = DeleteRequest(confirmation="test")
assert request.delete_data is False
@pytest.mark.asyncio
async def test_validation_response_does_not_echo_secret_input() -> None:
sensitive_value = '-'.join(('very', 'private', 'string', 'session'))
sensitive_value = "-".join(("very", "private", "string", "session"))
error = RequestValidationError(
[
{
'type': 'string_too_short',
'loc': ('body', 'session_string'),
'msg': 'String should have at least 32 characters',
'input': sensitive_value,
"type": "string_too_short",
"loc": ("body", "session_string"),
"msg": "String should have at least 32 characters",
"input": sensitive_value,
}
]
)
+16 -16
View File
@@ -6,43 +6,43 @@ from app.main import spa_fallback
@pytest.fixture
def static_dir(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path:
static = tmp_path / 'static'
static = tmp_path / "static"
static.mkdir()
(static / 'index.html').write_text('<html>index</html>', encoding='utf-8')
(static / 'app.js').write_text("console.log('app')", encoding='utf-8')
secret = tmp_path / 'secret.txt'
secret.write_text('TOP SECRET', encoding='utf-8')
monkeypatch.setattr('app.main.static_dir', static)
(static / "index.html").write_text("<html>index</html>", encoding="utf-8")
(static / "app.js").write_text("console.log('app')", encoding="utf-8")
secret = tmp_path / "secret.txt"
secret.write_text("TOP SECRET", encoding="utf-8")
monkeypatch.setattr("app.main.static_dir", static)
return static
def static_index(static_dir: Path) -> Path:
return static_dir / 'index.html'
return static_dir / "index.html"
def test_returns_existing_file(static_dir: Path) -> None:
response = spa_fallback('app.js')
assert response.path == static_dir / 'app.js'
response = spa_fallback("app.js")
assert response.path == static_dir / "app.js"
def test_unknown_path_falls_back_to_index(static_dir: Path) -> None:
response = spa_fallback('does/not/exist.js')
response = spa_fallback("does/not/exist.js")
assert response.path == static_index(static_dir)
def test_traversal_does_not_leak_outside_static(static_dir: Path) -> None:
response = spa_fallback('../secret.txt')
response = spa_fallback("../secret.txt")
assert response.path == static_index(static_dir)
response = spa_fallback('%2e%2e/secret.txt')
response = spa_fallback("%2e%2e/secret.txt")
assert response.path == static_index(static_dir)
def test_symlink_outside_static_is_blocked(static_dir: Path, tmp_path: Path) -> None:
target = tmp_path / 'outside.txt'
target.write_text('secret', encoding='utf-8')
link = static_dir / 'leak.txt'
target = tmp_path / "outside.txt"
target.write_text("secret", encoding="utf-8")
link = static_dir / "leak.txt"
link.symlink_to(target)
response = spa_fallback('leak.txt')
response = spa_fallback("leak.txt")
assert response.path == static_index(static_dir)
+1 -2
View File
@@ -6,8 +6,7 @@
"scripts": {
"build": "vite build",
"dev": "vite --host 0.0.0.0",
"test": "vitest run",
"check": "svelte-check --tsconfig ./tsconfig.json"
"test": "vitest run"
},
"dependencies": {
"lucide-svelte": "^0.468.0",
+2 -2
View File
@@ -36,10 +36,10 @@ spec:
resources:
requests:
cpu: "100m"
memory: "80Mi"
memory: "128Mi"
limits:
cpu: "500m"
memory: "256Mi"
memory: "512Mi"
volumeMounts:
- name: vaultwarden-data
mountPath: /data
+2 -2
View File
@@ -51,10 +51,10 @@ spec:
mountPath: /etc/x-ui
resources:
requests:
memory: "192Mi"
memory: "128Mi"
cpu: "100m"
limits:
memory: "512Mi"
memory: "1Gi"
cpu: "1000m"
volumes:
- name: x-ui-db