Compare commits
23
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e451c97dfc | ||
|
|
33c54ac830 | ||
|
|
b09d718310 | ||
|
|
b4f76373bb | ||
|
|
74adf38d63 | ||
|
|
f5b2f89f38 | ||
|
|
8e63284240 | ||
|
|
1d81410cd8 | ||
|
|
2f891a5d31 | ||
|
|
cda0022d81 | ||
|
|
dde6b1c743 | ||
|
|
7c4843c88c | ||
|
|
f9e4623ade | ||
|
|
2ad4fa1b82 | ||
|
|
2a4f215546 | ||
|
|
16aaeb60c1 | ||
|
|
a5409edbf2 | ||
|
|
c70d2db3a1 | ||
|
|
24dd82e801 | ||
|
|
11e92fdf4e | ||
|
|
b1f98fc148 | ||
|
|
9b91b5847e | ||
|
|
892790822d |
No files matched your search
+108
-49
@@ -464,7 +464,24 @@ jobs:
|
||||
|
||||
build:
|
||||
needs:
|
||||
[lint-actionlint, lint-shellcheck, lint-compose, lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
|
||||
# scan-deps and the two test jobs were missing here, so a commit with a
|
||||
# known-vulnerable dependency or a failing test still moved the :prod tag.
|
||||
# The deploy was blocked either way - it requires the whole workflow to
|
||||
# have succeeded - but the tag had already moved, and the next deploy to
|
||||
# run resolved it. Publishing and passing the checks are the same gate.
|
||||
[
|
||||
lint-actionlint,
|
||||
lint-shellcheck,
|
||||
lint-compose,
|
||||
lint-prettier,
|
||||
lint-ruff,
|
||||
lint-yaml,
|
||||
lint-dockerfiles,
|
||||
scan-deps,
|
||||
test-backend,
|
||||
test-frontend,
|
||||
validate,
|
||||
]
|
||||
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev')
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 60
|
||||
@@ -480,12 +497,20 @@ jobs:
|
||||
id: services
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
base="${{ github.event.before }}"
|
||||
if [ -z "$base" ] || [ "$base" = "0000000000000000000000000000000000000000" ]; then
|
||||
base="$(git rev-list --max-parents=0 HEAD)"
|
||||
fi
|
||||
|
||||
mapfile -t changed_files < <(git diff --name-only "$base" "${GITHUB_SHA}")
|
||||
# A failed diff used to leave changed_files empty, which reads exactly
|
||||
# like "nothing to build": the job went green having built nothing and
|
||||
# the tag never moved. The status is checked, not assumed.
|
||||
if ! changed="$(git diff --name-only "$base" "${GITHUB_SHA}")"; then
|
||||
echo "::error::cannot diff ${base}..${GITHUB_SHA}"
|
||||
exit 1
|
||||
fi
|
||||
mapfile -t changed_files <<<"$changed"
|
||||
|
||||
services=()
|
||||
|
||||
@@ -535,30 +560,57 @@ jobs:
|
||||
- name: Log in to registry
|
||||
if: steps.services.outputs.services != ''
|
||||
shell: bash
|
||||
# Through env, not by substitution into the script. A secret written
|
||||
# into a run: block is pasted into the shell source before bash parses
|
||||
# it, so a password containing a quote, a backtick or $(...) becomes
|
||||
# code that runs. Masking the value in the log does not prevent that.
|
||||
env:
|
||||
REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }}
|
||||
REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }}
|
||||
run: |
|
||||
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login "${REGISTRY}" \
|
||||
-u "${{ secrets.REGISTRY_USERNAME }}" \
|
||||
set -euo pipefail
|
||||
printf '%s' "$REGISTRY_PASSWORD" | docker login "${REGISTRY}" \
|
||||
-u "$REGISTRY_USERNAME" \
|
||||
--password-stdin
|
||||
|
||||
- name: Build and push changed images
|
||||
if: steps.services.outputs.services != ''
|
||||
shell: bash
|
||||
run: |
|
||||
# This step was the one run: block in the workflow without it, and it
|
||||
# is the one that cannot afford it: a docker push that failed partway
|
||||
# through the loop used to be followed by more pushes, the loop's exit
|
||||
# status came from the last one, and the job went green with half the
|
||||
# images missing from the registry.
|
||||
set -euo pipefail
|
||||
IFS=, read -r -a services <<< "${{ steps.services.outputs.services }}"
|
||||
|
||||
# Tags for this push. The commit-pinned name is the point of this
|
||||
# step: the deploy resolves it in preference to :prod, so a deploy
|
||||
# that sat in the queue behind a later push still gets the build of
|
||||
# the commit CI validated, instead of whatever :prod points at by the
|
||||
# time it runs. See render_pinned in deploy-lib.sh.
|
||||
commit_tag=""
|
||||
if [ "${GITHUB_REF_NAME}" = "main" ]; then
|
||||
commit_tag="sha-${GITHUB_SHA:0:12}"
|
||||
fi
|
||||
|
||||
set_tags() {
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main) tags+=("main" "prod") ;;
|
||||
dev) tags+=("dev") ;;
|
||||
esac
|
||||
if [ -n "$commit_tag" ]; then
|
||||
tags+=("$commit_tag")
|
||||
fi
|
||||
}
|
||||
|
||||
for service in "${services[@]}"; do
|
||||
case "$service" in
|
||||
dtek_notif)
|
||||
image="${REGISTRY}/forust/dtek-notif"
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
set_tags
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
@@ -573,15 +625,7 @@ jobs:
|
||||
;;
|
||||
errorpages)
|
||||
image="${REGISTRY}/forust/error-pages"
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
set_tags
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
@@ -595,15 +639,7 @@ jobs:
|
||||
done
|
||||
;;
|
||||
userbot)
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
set_tags
|
||||
for target in runtime panel; do
|
||||
case "$target" in
|
||||
runtime)
|
||||
@@ -638,15 +674,7 @@ jobs:
|
||||
image="${REGISTRY}/forust/xdfnx-homepage"
|
||||
;;
|
||||
esac
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
set_tags
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
@@ -672,15 +700,7 @@ jobs:
|
||||
image="${REGISTRY}/forust/webinar-checker"
|
||||
;;
|
||||
esac
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
set_tags
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
@@ -696,3 +716,42 @@ jobs:
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
# Every image the tree names has to carry the commit-pinned name, not only
|
||||
# the ones this push rebuilt. A push that touches nothing but manifests
|
||||
# builds nothing, and its deploy would then find no commit-pinned tag to
|
||||
# resolve and quietly fall back to the moving :prod - which is the whole
|
||||
# failure the commit-pinned name exists to remove.
|
||||
#
|
||||
# Re-tagging copies the manifest list and transfers no layers, so pinning
|
||||
# six images that already exist costs six registry writes.
|
||||
#
|
||||
# The list is derived from the tree rather than written out here, so an
|
||||
# image added to a manifest is covered without a second place to update.
|
||||
- name: Pin the commit name on the images this push did not rebuild
|
||||
if: github.ref_name == 'main'
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
commit_tag="sha-${GITHUB_SHA:0:12}"
|
||||
mapfile -t repos < <(
|
||||
git grep -hoE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+' -- '*.yaml' '*.yml' \
|
||||
| sort -u
|
||||
)
|
||||
if [ "${#repos[@]}" -eq 0 ]; then
|
||||
echo "No own images referenced by the tree."
|
||||
exit 0
|
||||
fi
|
||||
echo "pinning ${#repos[@]} image(s) to $commit_tag"
|
||||
for repo in "${repos[@]}"; do
|
||||
if docker buildx imagetools inspect "$repo:$commit_tag" >/dev/null 2>&1; then
|
||||
echo " already built by this push: ${repo##*/}"
|
||||
continue
|
||||
fi
|
||||
if ! docker buildx imagetools inspect "$repo:prod" >/dev/null 2>&1; then
|
||||
echo " WARNING: ${repo##*/} has no :prod to pin and no build produced it"
|
||||
continue
|
||||
fi
|
||||
docker buildx imagetools create --tag "$repo:$commit_tag" "$repo:prod"
|
||||
echo " pinned ${repo##*/}"
|
||||
done
|
||||
@@ -242,11 +242,49 @@ registry_digest() {
|
||||
| head -1 || true
|
||||
}
|
||||
|
||||
# The commit this deploy is for: what CI validated, or - on a manual dispatch,
|
||||
# whatever stage_preflight just checked out.
|
||||
deploy_commit() {
|
||||
local c="${DEPLOY_SHA:-}"
|
||||
[ -n "$c" ] || c="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
|
||||
printf '%.12s' "${c:-}"
|
||||
}
|
||||
|
||||
# Resolves one of our image refs to the digest THIS commit's build produced.
|
||||
#
|
||||
# A manifest naming `:prod` names a pointer, not a version, and the deploy
|
||||
# resolves it when the apply runs - which is not when CI ran it. Deploy runs are
|
||||
# queued rather than cancelled (see deploy.yaml), so two pushes in a row leave
|
||||
# the first deploy resolving the second push's build: the right manifests with
|
||||
# the wrong code, and nothing anywhere reports it. ci therefore publishes every
|
||||
# image it ships under `sha-<commit12>`, a name that cannot move, and that is
|
||||
# the name resolved here.
|
||||
#
|
||||
# The fallback to the plain tag is for an image this pipeline never built. It
|
||||
# reports itself, because a fallback nobody sees is the failure this removes.
|
||||
pinned_digest() {
|
||||
local ref="$1" commit pinned
|
||||
commit="$(deploy_commit)"
|
||||
if [ -n "$commit" ]; then
|
||||
pinned="$(registry_digest "${ref%:*}:sha-$commit")"
|
||||
if [ -n "$pinned" ]; then
|
||||
printf '%s' "$pinned"
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
pinned="$(registry_digest "$ref")"
|
||||
if [ -n "$pinned" ]; then
|
||||
echo "WARNING: ${ref} carries no sha-${commit:-<unknown>} tag; resolved the moving tag instead" >&2
|
||||
fi
|
||||
printf '%s' "$pinned"
|
||||
}
|
||||
|
||||
# Rewrites our own images to immutable digests on the way into the cluster.
|
||||
# Reads a manifest stream on stdin, writes the pinned stream to stdout.
|
||||
#
|
||||
# A digest is not knowable when a manifest is written, so it is resolved here, at
|
||||
# apply time, and never committed: git keeps a readable `:prod` tag. That is what
|
||||
# A digest is not knowable when a manifest is written, so it is never committed:
|
||||
# git keeps a readable `:prod` tag and the exact bytes are chosen here, at apply
|
||||
# time, from the tag ci published for the commit being deployed. That is what
|
||||
# makes rollback mean something. `kubectl rollout undo` restores the previous
|
||||
# ReplicaSet's pod template verbatim, and a template naming a digest restores the
|
||||
# exact bytes that were serving before. A template naming a moving tag does not —
|
||||
@@ -271,7 +309,7 @@ render_pinned() {
|
||||
|
||||
while read -r ref; do
|
||||
[ -n "$ref" ] || continue
|
||||
digest="$(registry_digest "$ref")"
|
||||
digest="$(pinned_digest "$ref")"
|
||||
if [ -z "$digest" ]; then
|
||||
echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2
|
||||
echo " The build job has to push that tag before the deploy resolves it." >&2
|
||||
@@ -339,7 +377,7 @@ restart_stale_images() {
|
||||
while read -r ns target image; do
|
||||
[ -n "${target:-}" ] || continue
|
||||
if [ -z "${digests[$image]:-}" ]; then
|
||||
digests[$image]="$(registry_digest "$image")"
|
||||
digests[$image]="$(pinned_digest "$image")"
|
||||
fi
|
||||
want="${digests[$image]}"
|
||||
if [ -z "$want" ]; then
|
||||
@@ -451,6 +489,14 @@ rollback_workloads() {
|
||||
local -a recovered=()
|
||||
while read -r kind ns name; do
|
||||
[ -n "${kind:-}" ] || continue
|
||||
# Helm-owned workloads are already rolled back by the release's --atomic
|
||||
# upgrade. `rollout undo` here would step back to the revision Helm just
|
||||
# escaped (the failed one), so leave them for the operator instead.
|
||||
if kubectl get "${kind}/${name}" -n "$ns" -o jsonpath='{.metadata.annotations}' 2>/dev/null | grep -q 'meta.helm.sh/release-name'; then
|
||||
echo " skip (helm-managed, needs manual check): ${kind}/${ns}/${name}"
|
||||
unrecovered+=("${kind}/${ns}/${name} (helm-managed)")
|
||||
continue
|
||||
fi
|
||||
if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \
|
||||
&& kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
|
||||
echo " rolled back: ${kind}/${ns}/${name}"
|
||||
@@ -489,6 +535,44 @@ helm_repo_for() {
|
||||
esac
|
||||
}
|
||||
|
||||
# helm_release_status <release> <namespace>
|
||||
# Prints the release status in lowercase (deployed, failed, pending-rollback,
|
||||
# ...) or "not-found" when the release does not exist yet.
|
||||
helm_release_status() {
|
||||
local out
|
||||
if ! out="$(helm status "$1" -n "$2" 2>&1)"; then
|
||||
echo "not-found"
|
||||
return 0
|
||||
fi
|
||||
awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]'
|
||||
}
|
||||
|
||||
# recover_pending_release <release> <namespace>
|
||||
# Rolls a release out of a pending-* state left by a failed --atomic upgrade
|
||||
# whose own rollback never completed. Without this every future upgrade errors
|
||||
# out until a human runs `helm rollback`. Passes through releases that are not
|
||||
# pending (deployed, failed, not-found). Returns non-zero when the release is
|
||||
# still not recoverable, so the pipeline fails loud instead of wedging.
|
||||
recover_pending_release() {
|
||||
local release="$1" namespace="$2" status
|
||||
status="$(helm_release_status "$release" "$namespace")"
|
||||
case "$status" in
|
||||
pending-upgrade|pending-rollback|pending-install)
|
||||
log "Release $release is $status, rolling back to the last deployed revision"
|
||||
if ! helm rollback "$release" -n "$namespace" --wait --timeout 10m >/dev/null 2>&1; then
|
||||
echo "WARN: helm rollback of $release did not complete"
|
||||
return 1
|
||||
fi
|
||||
status="$(helm_release_status "$release" "$namespace")"
|
||||
if [ "$status" != "deployed" ]; then
|
||||
echo "WARN: $release is $status after rollback"
|
||||
return 1
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
return 0
|
||||
}
|
||||
|
||||
upgrade_helm_releases() {
|
||||
local entry release chart namespace version values marker repo
|
||||
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
|
||||
@@ -509,13 +593,31 @@ upgrade_helm_releases() {
|
||||
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
|
||||
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
|
||||
log "Upgrading $release ($chart $version)"
|
||||
# A previous --atomic run whose own rollback never finished leaves the
|
||||
# release in pending-*, which blocks every future upgrade. Recover first
|
||||
# so one wedged revision cannot wedge the pipeline forever.
|
||||
if ! recover_pending_release "$release" "$namespace"; then
|
||||
echo "ERROR: $release is stuck and automatic rollback did not recover it, run 'helm rollback $release -n $namespace' by hand."
|
||||
return 1
|
||||
fi
|
||||
# --atomic rolls the release back when the upgrade times out or the workloads
|
||||
# it touches never become ready, so a bad chart bump is not left half applied.
|
||||
helm upgrade --install "$release" "$chart" \
|
||||
if ! helm upgrade --install "$release" "$chart" \
|
||||
--namespace "$namespace" \
|
||||
--version "$version" \
|
||||
--values "$REPO/$values" \
|
||||
--atomic --cleanup-on-fail --timeout 10m
|
||||
--atomic --cleanup-on-fail --timeout 10m; then
|
||||
echo "WARN: upgrade of $release failed, checking release state"
|
||||
# --atomic already attempted its own rollback; finish the job when that
|
||||
# rollback never completed, otherwise the release stays pending-* and
|
||||
# blocks every future run.
|
||||
if ! recover_pending_release "$release" "$namespace"; then
|
||||
echo "ERROR: upgrade of $release failed and the release did not recover, run 'helm rollback $release -n $namespace' by hand."
|
||||
else
|
||||
echo "ERROR: upgrade of $release failed (release is back on its previous revision)."
|
||||
fi
|
||||
return 1
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
@@ -825,6 +927,32 @@ smoke_hosts() {
|
||||
| sort -u
|
||||
}
|
||||
|
||||
# Traefik's own list of the routes it actually built. The Kubernetes CRs are the
|
||||
# wrong source for this: when a middleware fails to load, Traefik drops the
|
||||
# router that referenced it and leaves the CR behind looking perfectly healthy.
|
||||
#
|
||||
# api.insecure is already on for the internal `traefik` entrypoint, but the pod
|
||||
# IP is not routable from the node, so read it through kubectl exec rather than
|
||||
# standing up a port-forward. HTTP only: the TCP routers match on HostSNI(`*`)
|
||||
# and the UDP ones carry no rule at all, both selected by entrypoint and port,
|
||||
# so neither can answer whether a given host has a route.
|
||||
traefik_http_routes() {
|
||||
kubectl -n traefik exec deploy/traefik -- \
|
||||
wget -qO- --timeout=10 http://127.0.0.1:8080/api/http/routers 2>/dev/null \
|
||||
| jq -c '[.[] | {status, rule: (.rule // "")}]'
|
||||
}
|
||||
|
||||
# The hosts Traefik currently routes to, one per line. Every backticked token of
|
||||
# an enabled rule counts, which is a superset of the hosts -- PathPrefix values
|
||||
# land here too, harmlessly -- but it keeps the host syntax in one place instead
|
||||
# of a matcher per host. The scan keeps the delimiters, so strip them: what
|
||||
# belongs in a comparison against a hostname is the bare name.
|
||||
traefik_routed_hosts() {
|
||||
jq -r '[.[] | select(.status == "enabled") | (.rule // "")
|
||||
| scan("`[^`]+`") | ltrimstr("`") | rtrimstr("`")]
|
||||
| unique | .[]' <<<"$1"
|
||||
}
|
||||
|
||||
# stage_verify_k8s watches the rollout, which reports that the pods converged.
|
||||
# It cannot tell a converged pod from a serving one: a route pointing at the
|
||||
# wrong port, a Service selector that matches nothing the app listens on, a 500
|
||||
@@ -836,6 +964,13 @@ smoke_hosts() {
|
||||
# or a 404 from a path the service does not serve still means the chain is
|
||||
# intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means
|
||||
# the service is not serving, and only those fail the run.
|
||||
#
|
||||
# Except that a 404 is not evidence on its own. A router Traefik refused to
|
||||
# build answers with the same 404 and nothing behind it, so a middleware that
|
||||
# fails to load -- the crowdsec bouncer, which Traefik disables silently when
|
||||
# it cannot fetch the plugin -- takes down every route that referenced it while
|
||||
# this stage reports `ok` for all of them. No status code separates those two
|
||||
# cases, so ask Traefik which routes it built and fail on the difference.
|
||||
stage_smoke() {
|
||||
cd "$REPO"
|
||||
select_manifests >/dev/null
|
||||
@@ -882,11 +1017,52 @@ stage_smoke() {
|
||||
esac
|
||||
done
|
||||
|
||||
# Second gate. The probe above only means something if a router matched the
|
||||
# host in the first place, so compare the hosts we expect against the hosts
|
||||
# Traefik reports and fail on the difference.
|
||||
local routes routed
|
||||
if ! routes="$(traefik_http_routes)"; then
|
||||
echo "ERROR: could not read Traefik's router list, refusing to report success"
|
||||
return 1
|
||||
fi
|
||||
routed="$(traefik_routed_hosts "$routes")"
|
||||
|
||||
local -a unrouted=()
|
||||
local tries=3
|
||||
while :; do
|
||||
unrouted=()
|
||||
for h in "${hosts[@]}"; do
|
||||
grep -qxF "$h" <<<"$routed" || unrouted+=("$h")
|
||||
done
|
||||
if [ "${#unrouted[@]}" -eq 0 ]; then
|
||||
break
|
||||
fi
|
||||
# A router mid-rollout is legitimately absent for a moment. A middleware
|
||||
# that failed to load stays absent, so waiting cannot paper over it.
|
||||
if [ "$tries" -le 1 ]; then
|
||||
break
|
||||
fi
|
||||
tries=$((tries - 1))
|
||||
warn "${#unrouted[@]} host(s) have no enabled route yet, re-checking in 10s"
|
||||
sleep 10
|
||||
if ! routes="$(traefik_http_routes)"; then
|
||||
break
|
||||
fi
|
||||
routed="$(traefik_routed_hosts "$routes")"
|
||||
done
|
||||
|
||||
if [ "${#unrouted[@]}" -ne 0 ]; then
|
||||
for h in "${unrouted[@]}"; do
|
||||
echo " NO ROUTE $h (Traefik has no enabled router for this host)"
|
||||
done
|
||||
bad=1
|
||||
fi
|
||||
|
||||
if [ "$bad" -ne 0 ]; then
|
||||
echo "ERROR: at least one active service is not serving over its public route"
|
||||
return 1
|
||||
fi
|
||||
echo "all ${#hosts[@]} route(s) answered"
|
||||
echo "all ${#hosts[@]} route(s) answered and have a router"
|
||||
}
|
||||
|
||||
stage_apply_compose() {
|
||||
|
||||
@@ -73,8 +73,34 @@ jobs:
|
||||
needs: [validate]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
# Apply only, no verification, so this is just the work itself: snapshot,
|
||||
# then up to three sequential `helm upgrade --atomic --timeout 10m`, then the
|
||||
# apply loop. Verification has its own job and its own budget.
|
||||
# then sequential `helm upgrade --atomic --timeout 10m`, then the apply loop.
|
||||
# Verification has its own job and its own budget.
|
||||
#
|
||||
# 45 is roughly four times the measured cost of the stage, which is
|
||||
# deliberately not raised on a theory:
|
||||
#
|
||||
# helm, healthy 3 no-op upgrades ~3-5 min
|
||||
# helm, one release bad --atomic spends its 10m, ~10-15 min
|
||||
# then rolls that one back
|
||||
# apply loop ~40 manifests, 4 of which ~1 min
|
||||
# resolve an image digest
|
||||
# restart_stale_images 7.6s to find 8 workloads, ~0.5 min
|
||||
# 9.8s to resolve their digests
|
||||
#
|
||||
# The helm figure is one release, not three: `set -e` aborts
|
||||
# upgrade_helm_releases on the first failure, so a broken release costs
|
||||
# 10m and the other two are never attempted. Multiplying 10m by three
|
||||
# overstates the worst case by 20 minutes.
|
||||
#
|
||||
# The 45 minutes this was last raised to 45 were still not enough, and the
|
||||
# job logs for those runs no longer exist, so what actually consumed the
|
||||
# budget is not known - the two measurable candidates above account for
|
||||
# ~15 of it. The one unbounded thing left in this stage is
|
||||
# `docker manifest inspect` at deploy-lib.sh:236, which has no timeout
|
||||
# against a registry with a known hang mode. Bound it, and make the stage
|
||||
# announce what it is working on, before spending any of that on a larger
|
||||
# ceiling: a stage that is killed with a diagnosable last line is a bug
|
||||
# report, one that vanishes is not.
|
||||
timeout-minutes: 45
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
@@ -112,7 +138,22 @@ jobs:
|
||||
needs.apply-k8s.result != 'skipped' &&
|
||||
needs.apply-compose.result != 'skipped'
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
# ceil(changed_workloads / 8) waves of ROLLOUT_TIMEOUT each, plus rollback.
|
||||
# Not raised, because the arithmetic does not close.
|
||||
#
|
||||
# 32 workloads are under management and the wave width is 8, so the verify
|
||||
# itself is 4 waves of ROLLOUT_TIMEOUT (300s) = 20 minutes worst case, when
|
||||
# every rollout times out rather than converging. That is already 20 of 30.
|
||||
#
|
||||
# The other 10 would have to absorb rollback, and rollback_workloads is a
|
||||
# serial `while read` loop at 300s per failed workload. 10 minutes buys two.
|
||||
# Any larger number is buying a bigger multiple of an unbounded term rather
|
||||
# than covering a known cost: 60 minutes buys eight, and 60 minutes is
|
||||
# therefore not a bound, it is a guess with two digits.
|
||||
#
|
||||
# The number becomes derivable the moment rollback uses the same wave width
|
||||
# as the verify: 32 failures then cost 4 waves = 20 minutes instead of 160,
|
||||
# and 45 covers verify plus rollback at full width. That change is to the
|
||||
# recovery path and is not folded into a timeout edit.
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
|
||||
@@ -21,10 +21,39 @@ ssh_key="$key_dir/deploy_key"
|
||||
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
|
||||
chmod 600 "$ssh_key"
|
||||
|
||||
ssh -i "$ssh_key" -p "$deploy_port" \
|
||||
-o BatchMode=yes -o StrictHostKeyChecking=accept-new \
|
||||
"${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} DEPLOY_SHA=${DEPLOY_SHA:-} DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-} STAGE=$1 bash -se" <<'EOF'
|
||||
# A connection that died silently used to hang until the job timeout, and the
|
||||
# stage was never re-run: one flaky TCP session cost a whole 45-minute apply.
|
||||
# ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds
|
||||
# setup. Only exit 255 - ssh's own transport failures - is retried. A stage that
|
||||
# fails on its own merits exits with the remote's status, so a real failure
|
||||
# still surfaces its own log instead of burning three attempts. The stages are
|
||||
# declarative applies, so re-running one that had already committed is harmless.
|
||||
ssh_opts=(
|
||||
-i "$ssh_key" -p "$deploy_port"
|
||||
-o BatchMode=yes -o StrictHostKeyChecking=accept-new
|
||||
-o ConnectTimeout=15
|
||||
-o ServerAliveInterval=15 -o ServerAliveCountMax=4
|
||||
)
|
||||
|
||||
rc=0
|
||||
for attempt in 1 2 3; do
|
||||
if [ "$attempt" -gt 1 ]; then
|
||||
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
|
||||
sleep $((attempt * 5))
|
||||
fi
|
||||
rc=0
|
||||
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
|
||||
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
|
||||
"STAGE=$1" bash -se <<'EOF' || rc=$?
|
||||
source "$REPO/.gitea/workflows/deploy-lib.sh"
|
||||
run_stage "$STAGE"
|
||||
EOF
|
||||
[ "$rc" -eq 0 ] && break
|
||||
[ "$rc" -ne 255 ] && break
|
||||
done
|
||||
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo ":: error::stage $1 failed over ssh (exit $rc)"
|
||||
fi
|
||||
exit "$rc"
|
||||
@@ -73,7 +73,7 @@ spec:
|
||||
memory: "1.5Gi"
|
||||
cpu: "300m"
|
||||
requests:
|
||||
memory: "500Mi"
|
||||
memory: "1Gi"
|
||||
cpu: "50m"
|
||||
ports:
|
||||
- containerPort: 3000
|
||||
|
||||
@@ -52,7 +52,7 @@ spec:
|
||||
- containerPort: 9000
|
||||
resources:
|
||||
requests:
|
||||
memory: "700Mi"
|
||||
memory: "768Mi"
|
||||
cpu: "300m"
|
||||
limits:
|
||||
memory: "1.5Gi"
|
||||
@@ -86,8 +86,8 @@ spec:
|
||||
name: authentik-secrets
|
||||
resources:
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
memory: "320Mi"
|
||||
cpu: "300m"
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
memory: "768Mi"
|
||||
cpu: "700m"
|
||||
@@ -22,10 +22,10 @@ spec:
|
||||
imagePullPolicy: Always
|
||||
resources:
|
||||
requests:
|
||||
memory: "20Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "30m"
|
||||
limits:
|
||||
memory: "64Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "50m"
|
||||
envFrom:
|
||||
- secretRef:
|
||||
|
||||
@@ -30,8 +30,8 @@ spec:
|
||||
key: TUNNEL_TOKEN
|
||||
resources:
|
||||
requests:
|
||||
memory: "32Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "30m"
|
||||
limits:
|
||||
memory: "128Mi"
|
||||
memory: "256Mi"
|
||||
cpu: "200m"
|
||||
@@ -31,12 +31,13 @@ spec:
|
||||
name: bentopdf
|
||||
ports:
|
||||
- containerPort: 8080
|
||||
# p95 4M, max 11M over 7 days. Was 50Mi/700Mi.
|
||||
resources:
|
||||
requests:
|
||||
memory: "50Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "50m"
|
||||
ephemeral-storage: "100Mi"
|
||||
limits:
|
||||
memory: "700Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "700m"
|
||||
ephemeral-storage: "5Gi"
|
||||
@@ -38,13 +38,14 @@ spec:
|
||||
volumeMounts:
|
||||
- mountPath: /data
|
||||
name: data
|
||||
# p95 85M, max 136M over 7 days, spikes while converting. Was 250Mi/1.5Gi.
|
||||
resources:
|
||||
requests:
|
||||
memory: "250Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
cpu: "1500m"
|
||||
memory: "1.5Gi"
|
||||
memory: "512Mi"
|
||||
volumes:
|
||||
- name: data
|
||||
persistentVolumeClaim:
|
||||
|
||||
@@ -1,3 +1,17 @@
|
||||
# crowdsec/k8s is NOT managed by deploy.yaml - apply this by hand, and apply it
|
||||
# together with a restart:
|
||||
# kubectl apply -f crowdsec/k8s/crowdsec-middleware.yaml
|
||||
# kubectl -n traefik rollout restart deploy/traefik
|
||||
#
|
||||
# The restart is not optional. In stream mode the plugin runs a package-level
|
||||
# ticker goroutine (handleStreamTicker over the isCrowdsecStreamHealthy and
|
||||
# updateFailure globals) that no reconfiguration stops. Applying a change
|
||||
# wedges the instance: every route referencing it answers 404 and traefik logs
|
||||
# 'invalid middleware crowdsec-crowdsec-bouncer@kubernetescrd' until the pod is
|
||||
# replaced. Re-applying the previous config does NOT recover it, and the config
|
||||
# is not the cause - a valid CIDR cannot fail NewChecker, which is a plain
|
||||
# net.ParseCIDR. Only a new pod clears it. Measured cost: ~35s down for all
|
||||
# 20 hosts behind this middleware.
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: Middleware
|
||||
metadata:
|
||||
@@ -8,17 +22,48 @@ spec:
|
||||
crowdsec-bouncer:
|
||||
enabled: true
|
||||
LogLevel: INFO
|
||||
CrowdsecMode: live
|
||||
# `live` blocked on a `GET /v1/decisions` per request, so a burst
|
||||
# saturated the LAPI and the plugin 403'd IPs that were never banned.
|
||||
# v1.3.3 ignores UpdateMaxFailure in `live`, so fail-open is only
|
||||
# reachable in stream mode, which polls into a cache instead - no
|
||||
# per-request call to saturate. 15s rather than the 60s default: the
|
||||
# deploy runner shares one public IP with the house, so this bounds
|
||||
# both how late a ban lands and how long a lifted one lingers.
|
||||
CrowdsecMode: stream
|
||||
UpdateIntervalSeconds: 15
|
||||
# -1 = never block because the LAPI is unreachable. In v1.3.3
|
||||
# handleStreamTicker only clears isCrowdsecStreamHealthy when
|
||||
# updateMaxFailure != -1, and ServeHTTP 403s once it is false, so this
|
||||
# makes a CrowdSec outage mean "no protection", not "every site 403".
|
||||
UpdateMaxFailure: -1
|
||||
CrowdsecLapiScheme: http
|
||||
CrowdsecLapiHost: crowdsec-service.crowdsec.svc.cluster.local:8080
|
||||
CrowdsecLapiKeyFile: "/etc/traefik/secrets/traefik-api-key"
|
||||
# LAPI lookup is SYNCHRONOUS and per-request: the plugin blocks on
|
||||
# `GET /v1/decisions?ip=...&banned=true` before the request reaches
|
||||
# the backend, and fails CLOSED (403) if the lookup exceeds the
|
||||
# timeout. Unset, the fork defaults to 10s, which is an eternity for
|
||||
# a request path: a single slow LAPI (idle 1.3-7.4s here) turned
|
||||
# every request into a 10s hang and then a self-inflicted 403.
|
||||
# 2s keeps the fail-closed path fast and bounded; with the LAPI
|
||||
# resourced properly (see crowdsec-values.yaml) the lookup is
|
||||
# sub-100ms and this budget is never hit.
|
||||
CrowdsecLapiTimeout: "2s"
|
||||
# Bypasses the bouncer and the decision cache, no LAPI round-trip.
|
||||
# Keep in sync with forust/local-network in crowdsec-values.yaml.
|
||||
ClientTrustedIPs:
|
||||
- "127.0.0.0/8"
|
||||
- "10.0.0.0/8"
|
||||
- "172.16.0.0/12"
|
||||
- "192.168.0.0/16"
|
||||
- "100.64.0.0/10"
|
||||
- "169.254.0.0/16"
|
||||
- "fc00::/7"
|
||||
- "fe80::/10"
|
||||
# The mobile operator range from forust/mobile-whitelist, repeated
|
||||
# deliberately rather than relying on the parser whitelist alone.
|
||||
# That whitelist drops the event before it reaches a bucket, so no
|
||||
# decision is ever created - but it is one config away from not
|
||||
# firing, and the bouncer would then enforce a ban that was never
|
||||
# justified. This is the last line: even a decision that exists for
|
||||
# any reason is not served against the phone.
|
||||
- "84.245.64.0/18"
|
||||
# The name is HTTPTimeoutSeconds, an int in seconds (min 1) - there is
|
||||
# no CrowdsecLapiTimeout, and an unrecognised key is silently dropped,
|
||||
# which is how this sat at the 10s default. Nothing rides on it per
|
||||
# request any more, so this only bounds the stream pull - and too low
|
||||
# is the dangerous direction: the LAPI needs ~2s to answer
|
||||
# /v1/decisions/stream, and a pull that times out leaves the ban cache
|
||||
# frozen at its startup contents ("failed sending new decisions"),
|
||||
# i.e. new bans silently never apply. Keep it above the pull latency.
|
||||
HTTPTimeoutSeconds: 10
|
||||
@@ -58,26 +58,15 @@ config:
|
||||
reason: "Mobile IP whitelist"
|
||||
cidr:
|
||||
- "84.245.64.0/18"
|
||||
|
||||
postoverflows:
|
||||
s01-whitelist:
|
||||
home-dynamic-ip.yaml: |
|
||||
name: forust/home-dynamic-ip
|
||||
description: "Whitelist home dynamic IP"
|
||||
whitelist:
|
||||
reason: "Home dynamic IP"
|
||||
expression:
|
||||
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
|
||||
# The hairpin-NAT address of the router (192.168.88.1) is what the
|
||||
# Gitea Actions runner presents to Traefik - it is NOT the home
|
||||
# dynamic IP, so the whitelist above did not cover it. During a
|
||||
# deploy the runner POSTs to the Actions API many times a second;
|
||||
# a single 403 storm was enough to earn it a 4h ban and break every
|
||||
# later job. Whitelisting the whole LAN also covers phones and
|
||||
# tablets browsing over 192.168.88.0/24.
|
||||
lan.yaml: |
|
||||
name: forust/lan
|
||||
description: "Whitelist local network"
|
||||
# CrowdSec's own guidance: CIDR allowlisting belongs at the parser stage.
|
||||
# A parser whitelist discards the event before it reaches a bucket, so
|
||||
# these addresses never produce an overflow and never become a decision.
|
||||
# A postoverflow whitelist is checked only *after* the ban exists, and
|
||||
# the bouncer answers 403 for as long as it does - which is a window we
|
||||
# do not want the deploy sitting in.
|
||||
local-network.yaml: |
|
||||
name: forust/local-network
|
||||
description: "Whitelist loopback, private and VPN networks"
|
||||
whitelist:
|
||||
reason: "Local network"
|
||||
cidr:
|
||||
@@ -85,6 +74,81 @@ config:
|
||||
- "10.0.0.0/8"
|
||||
- "172.16.0.0/12"
|
||||
- "192.168.0.0/16"
|
||||
# CGNAT range (RFC 6598). The workstation and the k0s node live
|
||||
# here on WireGuard, and 100.64.0.0/10 is not covered by the
|
||||
# RFC 1918 blocks above.
|
||||
- "100.64.0.0/10"
|
||||
- "169.254.0.0/16"
|
||||
- "fc00::/7"
|
||||
- "fe80::/10"
|
||||
|
||||
postoverflows:
|
||||
s01-whitelist:
|
||||
# The one whitelist that has to stay here: resolving a hostname is a
|
||||
# network call, and the docs put expensive lookups in postoverflows on
|
||||
# purpose - it runs only when a bucket actually overflows.
|
||||
# ddns.forust.xyz is the public home address, not a private one, so
|
||||
# forust/local-network does not cover it.
|
||||
home-dynamic-ip.yaml: |
|
||||
name: forust/home-dynamic-ip
|
||||
description: "Whitelist home dynamic IP"
|
||||
whitelist:
|
||||
reason: "Home dynamic IP"
|
||||
expression:
|
||||
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
|
||||
|
||||
# LAPI-only main config override, merged over config.yaml. NOTE: the
|
||||
# chart's own default for this key is REPLACED, not merged, so its
|
||||
# auto_registration block is repeated verbatim below - drop it and the
|
||||
# agent can no longer register itself.
|
||||
config.yaml.local: |
|
||||
api:
|
||||
server:
|
||||
auto_registration: # Activate if not using TLS for authentication
|
||||
enabled: true
|
||||
token: "${REGISTRATION_TOKEN}" # /!\ Do not modify this variable (auto-generated and handled by the chart)
|
||||
allowed_ranges: # /!\ Make sure to adapt to the pod IP ranges used by your cluster
|
||||
- "127.0.0.1/32"
|
||||
- "192.168.0.0/16"
|
||||
- "10.0.0.0/8"
|
||||
- "172.16.0.0/12"
|
||||
# This homelab has no egress to console.crowdsec.cloud: DNS does
|
||||
# not resolve. The LAPI kept trying anyway ("Signal push: N
|
||||
# signals to push", "capi metrics: sending" every 10s) and each
|
||||
# attempt sat on a resolver timeout WHILE HOLDING A WRITE
|
||||
# TRANSACTION, which is what kept stalling per-request decision
|
||||
# lookups even with WAL enabled. Nothing to share and nothing to
|
||||
# pull - turn the Central API off instead of letting it block the
|
||||
# only database writer we have.
|
||||
online_client:
|
||||
sharing: false
|
||||
pull:
|
||||
community: false
|
||||
blocklists: false
|
||||
disable_usage_metrics_export: true
|
||||
db_config:
|
||||
# SQLite without WAL serialises every reader behind the writer's
|
||||
# rollback journal, and the LAPI writes constantly: the agent pushes
|
||||
# Traefik alerts read from Loki, the metrics collector counts
|
||||
# decisions, the bouncer touches "last pull" on every request.
|
||||
# Symptom: decision lookups taking 10-30s (and a second connection
|
||||
# that could not even open the database) while the LAPI sat at 28m
|
||||
# CPU - the process was blocked in fsync, not computing. Every
|
||||
# bouncer-protected request then blew through the plugin timeout and
|
||||
# fail-closed with 403, on every site at once.
|
||||
# The PVC is local-path-retain (hostPath), not a network share, so
|
||||
# WAL is safe here; the crowdsec docs recommend it for exactly this
|
||||
# ("allowing more concurrency in SQLite that will improve
|
||||
# performances in most scenarios").
|
||||
use_wal: true
|
||||
# Keeps the alert table bounded. At the 5000/7d default the file
|
||||
# reached 54MB in 15 days off the Traefik access log alone, and the
|
||||
# metrics collector counts decisions on a timer; a smaller working
|
||||
# set means fewer full scans. Crowdsec only prunes - SQLite never
|
||||
# shrinks the file, so the size stays until a manual VACUUM.
|
||||
flush:
|
||||
max_items: 1000
|
||||
max_age: 24h
|
||||
|
||||
lapi:
|
||||
env:
|
||||
|
||||
@@ -30,10 +30,14 @@
|
||||
# 3. ensure the static machine exists, recreating it with the
|
||||
# Secret password if missing (agent retry loops reconnect
|
||||
# on their own - same name + same password);
|
||||
# 4. prune bouncer entries idle for 30d;
|
||||
# 5. delete decisions from LePresidente/http-generic-403-bf, a hub
|
||||
# scenario that bans an IP for 4h after 5 POST-403s in 10s and
|
||||
# therefore bans us for our own bouncer's fail-closed 403s.
|
||||
# 4. prune bouncer entries idle for 30d.
|
||||
#
|
||||
# It used to also delete LePresidente/http-generic-403-bf decisions hourly.
|
||||
# That was a workaround for the bouncer failing closed on a slow LAPI and
|
||||
# 403-ing the deploy runner into a 4h ban. The bouncer now polls decisions
|
||||
# into a cache and never blocks on an unreachable LAPI, so it cannot
|
||||
# manufacture those 403s any more, and the scenario only fires against real
|
||||
# scanners - deleting their decisions hourly was undoing a working ban.
|
||||
#
|
||||
# Manual apply (crowdsec/k8s is NOT managed by deploy.yaml):
|
||||
# kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml
|
||||
@@ -196,19 +200,3 @@ spec:
|
||||
fi
|
||||
echo "== 4. prune stale bouncers (no pull for 30d) =="
|
||||
$LAPI_EXEC cscli bouncers prune -d 720h --force
|
||||
echo "== 5. drop http-403-bf decisions (4h self-bans) =="
|
||||
# `LePresidente/http-generic-403-bf` (hub item
|
||||
# crowdsecurity/http-generic-bf v0.9) bans any source IP
|
||||
# after 5 POSTs answered 403 within 10s, for 4h. That
|
||||
# includes 403s this homelab generates ITSELF (any
|
||||
# bouncer fail-closed, any app CSRF/rate-limit 403), and a
|
||||
# 4h ban on the runner/home IP silently breaks deploys and
|
||||
# browsing. The scenario cannot be removed per-scenario -
|
||||
# it is baked into a hub item, and disabling the whole
|
||||
# base-http-scenarios collection would drop ~40 useful
|
||||
# detections. Instead we keep the detection and drop its
|
||||
# decisions hourly; the LAN/home whitelists in
|
||||
# crowdsec-values.yaml handle the legit sources, so this
|
||||
# only ever hits real scanners (who are re-banned anyway).
|
||||
$LAPI_EXEC cscli decisions delete \
|
||||
--scenario LePresidente/http-generic-403-bf --all || true
|
||||
@@ -17,7 +17,7 @@ services:
|
||||
|
||||
session-keeper:
|
||||
build: ./phpsessid-bot
|
||||
image: gcr.forust.xyz/forust/session-keeper:latest
|
||||
image: gcr.forust.xyz/forust/session-keeper:prod
|
||||
pull_policy: build
|
||||
env_file: .env
|
||||
restart: unless-stopped
|
||||
@@ -33,7 +33,7 @@ services:
|
||||
|
||||
webinar-checker:
|
||||
build: ./webinar-checker
|
||||
image: gcr.forust.xyz/forust/webinar-checker:latest
|
||||
image: gcr.forust.xyz/forust/webinar-checker:prod
|
||||
pull_policy: build
|
||||
env_file: .env
|
||||
restart: unless-stopped
|
||||
|
||||
@@ -20,6 +20,15 @@ spec:
|
||||
# renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker
|
||||
image: mcr.microsoft.com/playwright:v1.56.0-jammy
|
||||
imagePullPolicy: IfNotPresent
|
||||
# p95 412M, max 478M over 7 days, no limit before. Request is set at p95
|
||||
# so the pod is not an eviction candidate; the limit stays above 2x the
|
||||
# request because browser page lifetimes are unpredictable.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "200m"
|
||||
memory: "416Mi"
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
command:
|
||||
- npx
|
||||
- -y
|
||||
|
||||
@@ -28,10 +28,10 @@ spec:
|
||||
resources:
|
||||
requests:
|
||||
cpu: 25m
|
||||
memory: 64Mi
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: 250m
|
||||
memory: 256Mi
|
||||
memory: 128Mi
|
||||
readinessProbe:
|
||||
exec:
|
||||
command: ["redis-cli", "ping"]
|
||||
|
||||
@@ -38,10 +38,10 @@ spec:
|
||||
resources:
|
||||
requests:
|
||||
cpu: 25m
|
||||
memory: 96Mi
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: 250m
|
||||
memory: 256Mi
|
||||
memory: 128Mi
|
||||
readinessProbe:
|
||||
exec:
|
||||
command: ["/bin/sh", "-ec", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"]
|
||||
|
||||
@@ -67,7 +67,7 @@ spec:
|
||||
resources:
|
||||
requests:
|
||||
cpu: "50m"
|
||||
memory: "128Mi"
|
||||
memory: "192Mi"
|
||||
limits:
|
||||
cpu: "600m"
|
||||
memory: "512Mi"
|
||||
memory: "384Mi"
|
||||
@@ -1,7 +1,7 @@
|
||||
services:
|
||||
errorpage:
|
||||
build: .
|
||||
image: gcr.forust.xyz/forust/error-pages:latest
|
||||
image: gcr.forust.xyz/forust/error-pages:prod
|
||||
pull_policy: build
|
||||
container_name: error-pages
|
||||
restart: unless-stopped
|
||||
|
||||
@@ -28,6 +28,13 @@ spec:
|
||||
containers:
|
||||
- name: error-pages
|
||||
image: gcr.forust.xyz/forust/error-pages:prod
|
||||
# p95 6M, max 10M, no limit before.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "10m"
|
||||
memory: "32Mi"
|
||||
limits:
|
||||
memory: "128Mi"
|
||||
ports:
|
||||
- containerPort: 80
|
||||
readinessProbe:
|
||||
|
||||
@@ -17,6 +17,12 @@ data:
|
||||
|
||||
GITEA__mailer__ENABLED: "false"
|
||||
|
||||
# No code/issue search needed: bleve reindexes the whole issue index on
|
||||
# every pod restart (cron.rebuild_issue_indexer RUN_AT_START) and hammers
|
||||
# the rotational disk for an hour. "db" serves issue search from postgres.
|
||||
GITEA__indexer__ISSUE_INDEXER_TYPE: "db"
|
||||
GITEA__indexer__REPO_INDEXER_ENABLED: "false"
|
||||
|
||||
GITEA__log__logger.access.MODE: "console, file"
|
||||
USER_UID: "1000"
|
||||
USER_GID: "1000"
|
||||
@@ -47,10 +47,10 @@ spec:
|
||||
mountPath: /data
|
||||
resources:
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
memory: "320Mi"
|
||||
cpu: "300m"
|
||||
limits:
|
||||
memory: "1.5Gi"
|
||||
memory: "1Gi"
|
||||
cpu: "1300m"
|
||||
volumes:
|
||||
- name: gitea-data
|
||||
|
||||
@@ -16,15 +16,12 @@ spec:
|
||||
- name: gitea-service
|
||||
port: 3000
|
||||
# Registry route: NO crowdsec-bouncer.
|
||||
# The bouncer plugin does a blocking `GET /v1/decisions` to the LAPI on
|
||||
# *every* request. A deploy burst (runner Action API polls, `docker
|
||||
# manifest inspect` per own image, containerd pulls, smoke probes) fires
|
||||
# hundreds of parallel registry calls; LAPI saturation pushed the lookup
|
||||
# past the plugin timeout, and the bouncer fail-closed with 403 - which
|
||||
# containerd surfaces as ErrImagePull/ImagePullBackOff on the next pod.
|
||||
# This route only serves authenticated OCI traffic (registry tokens,
|
||||
# basic-auth already handled by gitea) and scanners get nothing useful
|
||||
# from /v2, so there is no bruteforce surface to protect here.
|
||||
# A deploy burst (runner Action API polls, `docker manifest inspect` per
|
||||
# own image, containerd pulls, smoke probes) fires hundreds of parallel
|
||||
# registry calls, and a ban on the runner breaks every later job. This
|
||||
# route only serves authenticated OCI traffic - registry tokens and
|
||||
# basic-auth are already handled by gitea - and scanners get nothing
|
||||
# useful from /v2, so there is no bruteforce surface to protect here.
|
||||
- match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`)
|
||||
kind: Rule
|
||||
services:
|
||||
|
||||
@@ -57,10 +57,10 @@ spec:
|
||||
resources:
|
||||
requests:
|
||||
cpu: "50m"
|
||||
memory: "64Mi"
|
||||
memory: "32Mi"
|
||||
limits:
|
||||
cpu: "200m"
|
||||
memory: "256Mi"
|
||||
memory: "128Mi"
|
||||
volumes:
|
||||
- name: glance-config
|
||||
configMap:
|
||||
|
||||
@@ -3,7 +3,7 @@ services:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile.forust
|
||||
image: gcr.forust.xyz/forust/forust-homepage:latest
|
||||
image: gcr.forust.xyz/forust/forust-homepage:prod
|
||||
pull_policy: build
|
||||
# ports:
|
||||
# - "8085:80"
|
||||
@@ -35,7 +35,7 @@ services:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile.xdfnx
|
||||
image: gcr.forust.xyz/forust/xdfnx-homepage:latest
|
||||
image: gcr.forust.xyz/forust/xdfnx-homepage:prod
|
||||
pull_policy: build
|
||||
restart: unless-stopped
|
||||
# ports:
|
||||
|
||||
@@ -39,10 +39,10 @@ spec:
|
||||
failureThreshold: 3
|
||||
resources:
|
||||
requests:
|
||||
memory: "10Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "20m"
|
||||
limits:
|
||||
memory: "100Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "50m"
|
||||
---
|
||||
apiVersion: v1
|
||||
@@ -86,8 +86,8 @@ spec:
|
||||
failureThreshold: 3
|
||||
resources:
|
||||
requests:
|
||||
memory: "10Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "20m"
|
||||
limits:
|
||||
memory: "100Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "50m"
|
||||
@@ -0,0 +1,24 @@
|
||||
# You can find documentation for all the supported env variables at https://docs.immich.app/install/environment-variables
|
||||
|
||||
# The location where your uploaded files are stored. The k8s manifests bind
|
||||
# mount /mnt/immich/library, which is the sdc9 partition - the same place, so
|
||||
# the two deployment paths look at one library.
|
||||
UPLOAD_LOCATION=/mnt/immich/library
|
||||
|
||||
# The location where your database files are stored. Network shares are not supported for the database
|
||||
DB_DATA_LOCATION=./postgres
|
||||
|
||||
# To set a timezone, uncomment the next line and change Etc/UTC to a TZ identifier from this list: https://en.wikipedia.org/wiki/List_of_tz_database_time_zones#List
|
||||
# TZ=Etc/UTC
|
||||
|
||||
# The Immich version to use. You can pin this to a specific version like "v2.1.0"
|
||||
IMMICH_VERSION=v3
|
||||
|
||||
# Connection secret for postgres. You should change it to a random password
|
||||
# Please use only the characters `A-Za-z0-9`, without special characters or spaces
|
||||
DB_PASSWORD=postgres
|
||||
|
||||
# The values below this line do not need to be changed
|
||||
###################################################################################
|
||||
DB_USERNAME=postgres
|
||||
DB_DATABASE_NAME=immich
|
||||
@@ -0,0 +1,63 @@
|
||||
name: immich
|
||||
|
||||
services:
|
||||
immich-server:
|
||||
container_name: immich_server
|
||||
image: ghcr.io/immich-app/immich-server:v3
|
||||
volumes:
|
||||
- ${UPLOAD_LOCATION}:/data
|
||||
- /etc/localtime:/etc/localtime:ro
|
||||
env_file:
|
||||
- .env
|
||||
ports:
|
||||
- "2283:2283"
|
||||
depends_on:
|
||||
- redis
|
||||
- database
|
||||
restart: always
|
||||
healthcheck:
|
||||
disable: false
|
||||
|
||||
immich-machine-learning:
|
||||
container_name: immich_machine_learning
|
||||
# For hardware acceleration, add one of -[armnn, cuda, rocm, openvino, rknn] to the image tag.
|
||||
# Example tag: ${IMMICH_VERSION:-release}-cuda
|
||||
image: ghcr.io/immich-app/immich-machine-learning:${IMMICH_VERSION:-release}
|
||||
# extends: # uncomment this section for hardware acceleration - see https://docs.immich.app/features/ml-hardware-acceleration
|
||||
# file: hwaccel.ml.yml
|
||||
# service: cpu # set to one of [armnn, cuda, rocm, openvino, openvino-wsl, rknn] for accelerated inference - use the `-wsl` version for WSL2 where applicable
|
||||
volumes:
|
||||
- model-cache:/cache
|
||||
env_file:
|
||||
- .env
|
||||
restart: always
|
||||
healthcheck:
|
||||
disable: false
|
||||
|
||||
redis:
|
||||
container_name: immich_redis
|
||||
image: docker.io/valkey/valkey:9@sha256:418652cfb58ef879d4978c33553735d7147016032d5aefaa14c828e611eb9dfd
|
||||
healthcheck:
|
||||
test: redis-cli ping | grep -q PONG || exit 1
|
||||
restart: always
|
||||
|
||||
database:
|
||||
container_name: immich_postgres
|
||||
image: ghcr.io/immich-app/postgres:14-vectorchord0.4.3-pgvectors0.2.0@sha256:bcf63357191b76a916ae5eb93464d65c07511da41e3bf7a8416db519b40b1c23
|
||||
environment:
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD}
|
||||
POSTGRES_USER: ${DB_USERNAME}
|
||||
POSTGRES_DB: ${DB_DATABASE_NAME}
|
||||
POSTGRES_INITDB_ARGS: "--data-checksums"
|
||||
# Uncomment the DB_STORAGE_TYPE: 'HDD' var if your database isn't stored on SSDs
|
||||
# DB_STORAGE_TYPE: 'HDD'
|
||||
volumes:
|
||||
# Do not edit the next line. If you want to change the database storage location on your system, edit the value of DB_DATA_LOCATION in the .env file
|
||||
- ${DB_DATA_LOCATION}:/var/lib/postgresql/data
|
||||
shm_size: 128mb
|
||||
restart: always
|
||||
healthcheck:
|
||||
disable: false
|
||||
|
||||
volumes:
|
||||
model-cache:
|
||||
File renamed without changes.
@@ -0,0 +1,28 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: immich-prod-tls
|
||||
namespace: immich
|
||||
spec:
|
||||
secretName: immich-prod-tls
|
||||
dnsNames:
|
||||
- immich.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: immich
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -0,0 +1,24 @@
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: immich-config
|
||||
namespace: immich
|
||||
data:
|
||||
TZ: "Europe/Bratislava"
|
||||
|
||||
# The database in this namespace, not the shared one in the database
|
||||
# namespace: v3 needs VectorChord, and only the dedicated image carries it.
|
||||
DB_HOSTNAME: "immich-postgres"
|
||||
DB_PORT: "5432"
|
||||
DB_USERNAME: "immich"
|
||||
DB_DATABASE_NAME: "immich"
|
||||
DB_SSL_MODE: "disable"
|
||||
DB_VECTOR_EXTENSION: "vectorchord"
|
||||
|
||||
REDIS_HOSTNAME: "immich-valkey"
|
||||
REDIS_PORT: "6379"
|
||||
|
||||
# Traefik is the only client of the server, and it is a pod: the address immich
|
||||
# sees is inside the node's pod CIDR. Without this the server does not trust
|
||||
# X-Forwarded-For and every request looks like it came from Traefik itself.
|
||||
IMMICH_TRUSTED_PROXIES: "10.244.0.0/24"
|
||||
@@ -0,0 +1,95 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: immich-service
|
||||
namespace: immich
|
||||
spec:
|
||||
selector:
|
||||
app: immich
|
||||
ports:
|
||||
- name: http
|
||||
port: 2283
|
||||
targetPort: 2283
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: immich-deployment
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich
|
||||
spec:
|
||||
replicas: 2
|
||||
selector:
|
||||
matchLabels:
|
||||
app: immich
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: immich
|
||||
spec:
|
||||
containers:
|
||||
- name: immich
|
||||
image: ghcr.io/immich-app/immich-server:v3
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: immich-config
|
||||
- secretRef:
|
||||
name: immich-secrets
|
||||
ports:
|
||||
- name: http
|
||||
containerPort: 2283
|
||||
volumeMounts:
|
||||
- name: immich-data
|
||||
mountPath: /data
|
||||
# The first boot runs migrations and warms the transcoder, which can
|
||||
# take minutes, so liveness has to wait on the startup probe.
|
||||
startupProbe:
|
||||
httpGet:
|
||||
path: /api/server/ping
|
||||
port: http
|
||||
failureThreshold: 60
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /api/server/ping
|
||||
port: http
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /api/server/ping
|
||||
port: http
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 30
|
||||
timeoutSeconds: 5
|
||||
# Only the request is scheduled against, and the node is already
|
||||
# oversubscribed (5.58 of 6 cores requested) while actually running
|
||||
# at about 1.5. So the request states what this sits at while idle -
|
||||
# tens of millicores - and the limit leaves room for the burst that
|
||||
# matters: thumbnails, transcodes and metadata extraction.
|
||||
#
|
||||
# The limit used to be 2Gi, but the server OOMKilled on boot while
|
||||
# chewing through a backlog of unprocessed assets (API +Workers in
|
||||
# one container spike well past idle).
|
||||
resources:
|
||||
requests:
|
||||
cpu: "100m"
|
||||
memory: "512Mi"
|
||||
limits:
|
||||
cpu: "1500m"
|
||||
memory: "4Gi"
|
||||
volumes:
|
||||
- name: immich-data
|
||||
# The library lives on the node's own disk, not in a PVC. A PVC here
|
||||
# meant declaring a size up front for data that does not exist yet,
|
||||
# on a provisioner that cannot grow it, and the only copy of the
|
||||
# photos was one `kubectl delete namespace` away.
|
||||
#
|
||||
# Directory, not DirectoryOrCreate, on purpose: if sdc9 is not
|
||||
# mounted, this must fail loudly instead of quietly writing the
|
||||
# library onto the root filesystem.
|
||||
hostPath:
|
||||
path: /mnt/immich/library
|
||||
type: Directory
|
||||
@@ -0,0 +1,36 @@
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
metadata:
|
||||
name: immich-prod
|
||||
namespace: immich
|
||||
spec:
|
||||
entryPoints:
|
||||
- websecure
|
||||
routes:
|
||||
- match: Host(`immich.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: immich-service
|
||||
port: 2283
|
||||
tls:
|
||||
secretName: immich-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
metadata:
|
||||
name: immich-local
|
||||
namespace: immich
|
||||
spec:
|
||||
entryPoints:
|
||||
- websecure
|
||||
routes:
|
||||
- match: Host(`immich.workstation.internal`) || Host(`immich.gigaforust.internal`)
|
||||
kind: Rule
|
||||
services:
|
||||
- name: immich-service
|
||||
port: 2283
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -0,0 +1,92 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: immich-machine-learning
|
||||
namespace: immich
|
||||
spec:
|
||||
selector:
|
||||
app: immich-machine-learning
|
||||
ports:
|
||||
- name: http
|
||||
port: 3003
|
||||
targetPort: 3003
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: immich-machine-learning-deployment
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-machine-learning
|
||||
spec:
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app: immich-machine-learning
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: immich-machine-learning
|
||||
spec:
|
||||
containers:
|
||||
- name: immich-machine-learning
|
||||
image: ghcr.io/immich-app/immich-machine-learning:v3
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: immich-config
|
||||
- secretRef:
|
||||
name: immich-secrets
|
||||
ports:
|
||||
- name: http
|
||||
containerPort: 3003
|
||||
volumeMounts:
|
||||
- name: model-cache
|
||||
mountPath: /cache
|
||||
# The first request pulls a model over the internet, so a cold start
|
||||
# is slower than a container start.
|
||||
startupProbe:
|
||||
httpGet:
|
||||
path: /ping
|
||||
port: http
|
||||
failureThreshold: 60
|
||||
periodSeconds: 5
|
||||
timeoutSeconds: 5
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /ping
|
||||
port: http
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /ping
|
||||
port: http
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 30
|
||||
timeoutSeconds: 5
|
||||
# Same reasoning as the server: the request covers the idle cost
|
||||
# only, because the node has no spare cores to schedule against.
|
||||
# Recognition is the burst - a busy import wants both cores.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "100m"
|
||||
memory: "1Gi"
|
||||
limits:
|
||||
cpu: "2000m"
|
||||
memory: "3Gi"
|
||||
volumes:
|
||||
- name: model-cache
|
||||
persistentVolumeClaim:
|
||||
claimName: immich-model-cache-pvc
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: immich-model-cache-pvc
|
||||
namespace: immich
|
||||
spec:
|
||||
accessModes:
|
||||
- ReadWriteOnce
|
||||
resources:
|
||||
requests:
|
||||
storage: 2Gi
|
||||
@@ -0,0 +1,5 @@
|
||||
# yaml-language-server: $schema=kubernetes
|
||||
apiVersion: v1
|
||||
kind: Namespace
|
||||
metadata:
|
||||
name: immich
|
||||
@@ -0,0 +1,138 @@
|
||||
# Immich's own database, separate from the shared postgres in the database
|
||||
# namespace. It has to be separate: v3 checks the VectorChord version at startup
|
||||
# and refuses to boot without it, VectorChord needs its .so in
|
||||
# shared_preload_libraries, and that can only be read when postmaster starts.
|
||||
# So the shared instance would have to be rebuilt on a custom image carrying
|
||||
# vchord and restarted - for every consumer of it (authentik, gitea, netbox,
|
||||
# netronome, penpot, statuspage). Not worth it for one photo library.
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: immich-postgres
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-postgres
|
||||
spec:
|
||||
selector:
|
||||
app: immich-postgres
|
||||
ports:
|
||||
- name: postgres
|
||||
port: 5432
|
||||
targetPort: postgres
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
name: immich-postgres
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-postgres
|
||||
spec:
|
||||
serviceName: immich-postgres
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app: immich-postgres
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: immich-postgres
|
||||
spec:
|
||||
containers:
|
||||
- name: postgres
|
||||
# v3.x expects vchord for its vector work and vectors (pgvecto.rs)
|
||||
# for some index types. This image ships both and preloads them, plus
|
||||
# its own shared_buffers and wal settings, through
|
||||
# /etc/postgresql/postgresql.conf - which its entrypoint reaches via
|
||||
# `postgres -c config_file=...` in the image CMD.
|
||||
#
|
||||
# So there is deliberately no `command:` here. Overriding it replaces
|
||||
# that config_file, and it also loses the step where the entrypoint
|
||||
# drops from root to the postgres user: postmaster then starts as
|
||||
# root and refuses to run.
|
||||
image: ghcr.io/immich-app/postgres:14-vectorchord0.4.3-pgvectors0.2.0@sha256:bcf63357191b76a916ae5eb93464d65c07511da41e3bf7a8416db519b40b1c23
|
||||
env:
|
||||
- name: POSTGRES_USER
|
||||
value: immich
|
||||
- name: POSTGRES_DB
|
||||
value: immich
|
||||
- name: POSTGRES_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: immich-secrets
|
||||
key: DB_PASSWORD
|
||||
# Only read when the data directory is empty, so the checksums are
|
||||
# decided here and never again.
|
||||
- name: POSTGRES_INITDB_ARGS
|
||||
value: --data-checksums
|
||||
# The postgres-data volume lives on sdc, which is rotational. The
|
||||
# SSD template is the default; HDD only changes the planner costs
|
||||
# (effective_io_concurrency, random_page_cost), nothing structural.
|
||||
- name: DB_STORAGE_TYPE
|
||||
value: HDD
|
||||
- name: TZ
|
||||
valueFrom:
|
||||
configMapKeyRef:
|
||||
name: immich-config
|
||||
key: TZ
|
||||
ports:
|
||||
- name: postgres
|
||||
containerPort: 5432
|
||||
volumeMounts:
|
||||
- name: postgres-data
|
||||
mountPath: /var/lib/postgresql/data
|
||||
# The upstream compose file asks docker for 128mb of shm. Kubernetes
|
||||
# gives every container 64mb, which is not what postmaster expects
|
||||
# for parallel query workers and the WAL writer.
|
||||
- name: shm
|
||||
mountPath: /dev/shm
|
||||
# Probes use a generous timeout on purpose: the data lives on a
|
||||
# rotational disk on a loaded single node, and pg_isready can take
|
||||
# seconds during WAL recovery. A 1s timeout kills the container
|
||||
# mid-recovery and restarts the spiral.
|
||||
startupProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", "pg_isready -U immich -d immich"]
|
||||
failureThreshold: 60
|
||||
periodSeconds: 5
|
||||
timeoutSeconds: 5
|
||||
readinessProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", "pg_isready -U immich -d immich"]
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
livenessProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", "pg_isready -U immich -d immich"]
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 20
|
||||
timeoutSeconds: 5
|
||||
# The image template sets shared_buffers to 512MB, and the vchord and
|
||||
# vectors workers are Rust binaries with a real RSS footprint on top
|
||||
# of postmaster, checkpointer and friends. 1Gi was enough to start
|
||||
# the server but the vectors worker kept dying in it, so the limit
|
||||
# sits at 2Gi. The request stays at the idle cost.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "50m"
|
||||
memory: "256Mi"
|
||||
limits:
|
||||
cpu: "1000m"
|
||||
memory: "2Gi"
|
||||
volumes:
|
||||
- name: shm
|
||||
emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 128Mi
|
||||
volumeClaimTemplates:
|
||||
- metadata:
|
||||
name: postgres-data
|
||||
spec:
|
||||
accessModes: ["ReadWriteOnce"]
|
||||
# Retain: this is the metadata for a library that only exists in one
|
||||
# place, and local-path cannot expand a bound volume, so this size has
|
||||
# to hold until the library is rebuilt or dumped elsewhere.
|
||||
storageClassName: local-path-retain
|
||||
resources:
|
||||
requests:
|
||||
storage: 32Gi
|
||||
@@ -0,0 +1,13 @@
|
||||
apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
name: immich-secrets
|
||||
namespace: immich
|
||||
type: Opaque
|
||||
stringData:
|
||||
# Creates the immich superuser in this namespace's own postgres on first
|
||||
# boot, and is the same value the server connects with. Nothing outside the
|
||||
# immich namespace needs it. Letters and digits only: immich reads this into
|
||||
# a connection string.
|
||||
DB_PASSWORD: "changeme"
|
||||
REDIS_PASSWORD: "changeme"
|
||||
@@ -0,0 +1,84 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: immich-valkey
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-valkey
|
||||
spec:
|
||||
clusterIP: None
|
||||
selector:
|
||||
app: immich-valkey
|
||||
ports:
|
||||
- name: valkey
|
||||
port: 6379
|
||||
targetPort: valkey
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
name: immich-valkey
|
||||
namespace: immich
|
||||
labels:
|
||||
app: immich-valkey
|
||||
spec:
|
||||
serviceName: immich-valkey
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app: immich-valkey
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: immich-valkey
|
||||
spec:
|
||||
containers:
|
||||
- name: valkey
|
||||
image: docker.io/valkey/valkey:9.1.2-alpine
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- valkey-server --appendonly yes --save 30 1 --loglevel warning --requirepass "$REDIS_PASSWORD"
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: immich-secrets
|
||||
ports:
|
||||
- name: valkey
|
||||
containerPort: 6379
|
||||
volumeMounts:
|
||||
- name: valkey-data
|
||||
mountPath: /data
|
||||
# Same reasoning as postgres: 1s probe timeouts flap on a loaded
|
||||
# single node with rotational storage.
|
||||
startupProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
|
||||
failureThreshold: 20
|
||||
periodSeconds: 5
|
||||
timeoutSeconds: 5
|
||||
readinessProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
livenessProbe:
|
||||
exec:
|
||||
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
|
||||
initialDelaySeconds: 20
|
||||
periodSeconds: 20
|
||||
timeoutSeconds: 5
|
||||
resources:
|
||||
requests:
|
||||
cpu: "25m"
|
||||
memory: "64Mi"
|
||||
limits:
|
||||
cpu: "250m"
|
||||
memory: "256Mi"
|
||||
volumeClaimTemplates:
|
||||
- metadata:
|
||||
name: valkey-data
|
||||
spec:
|
||||
accessModes: ["ReadWriteOnce"]
|
||||
resources:
|
||||
requests:
|
||||
storage: 1Gi
|
||||
@@ -7,18 +7,32 @@
|
||||
|
||||
controller:
|
||||
type: daemonset
|
||||
|
||||
# config-reloader sidecar: p95 33M, max 43M. The chart keeps it at the top level,
|
||||
# not under `alloy:`.
|
||||
configReloader:
|
||||
resources:
|
||||
requests:
|
||||
memory: "128Mi"
|
||||
cpu: "50m"
|
||||
memory: "32Mi"
|
||||
cpu: "10m"
|
||||
limits:
|
||||
memory: "512Mi"
|
||||
cpu: "500m"
|
||||
memory: "128Mi"
|
||||
|
||||
image:
|
||||
tag: "v1.19.2"
|
||||
|
||||
alloy:
|
||||
# p95 275M, max 287M. Alloy tails every pod log and ships it to Loki, so it sits
|
||||
# on the same IronWolf read path the node is I/O bound on. Request is set at p95.
|
||||
# The chart key is `alloy.resources`. `controller.resources` is ignored silently,
|
||||
# which is why this pod shipped with no limits at all.
|
||||
resources:
|
||||
requests:
|
||||
memory: "288Mi"
|
||||
cpu: "50m"
|
||||
limits:
|
||||
memory: "512Mi"
|
||||
|
||||
configMap:
|
||||
create: true
|
||||
content: |
|
||||
|
||||
@@ -39,6 +39,16 @@ loki:
|
||||
local:
|
||||
directory: /var/loki/rules
|
||||
|
||||
# p95 84M, max 85M for the rules sidecar that shares the singleBinary pod.
|
||||
# The chart exposes it as `sidecar.resources`, shared with any other sidecar.
|
||||
sidecar:
|
||||
resources:
|
||||
requests:
|
||||
memory: "96Mi"
|
||||
cpu: "10m"
|
||||
limits:
|
||||
memory: "192Mi"
|
||||
|
||||
singleBinary:
|
||||
replicas: 1
|
||||
persistence:
|
||||
@@ -47,10 +57,10 @@ singleBinary:
|
||||
storageClass: local-path-retain
|
||||
resources:
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
memory: "256Mi"
|
||||
cpu: "200m"
|
||||
limits:
|
||||
memory: "2Gi"
|
||||
memory: "1Gi"
|
||||
cpu: "1000m"
|
||||
|
||||
# Zeroed: unused in SingleBinary mode (chart validation requires it).
|
||||
@@ -63,12 +73,25 @@ backend:
|
||||
|
||||
gateway:
|
||||
replicas: 1
|
||||
# Single node: chart default is required podAntiAffinity on hostname +
|
||||
# RollingUpdate 25%/25% (effective maxUnavailable=0 at replicas=1).
|
||||
# That deadlocks the rollout: the new pod stays Unschedulable while the
|
||||
# old one lives, and the old one never leaves while the new one is not
|
||||
# Ready. Null clears the default (an empty map would deep-merge with it
|
||||
# and keep the required rule); maxUnavailable=1 allows a brief gateway
|
||||
# outage during rollouts instead of a stuck deploy.
|
||||
affinity: null
|
||||
deploymentStrategy:
|
||||
type: RollingUpdate
|
||||
rollingUpdate:
|
||||
maxSurge: 1
|
||||
maxUnavailable: 1
|
||||
resources:
|
||||
requests:
|
||||
memory: "64Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "50m"
|
||||
limits:
|
||||
memory: "256Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "300m"
|
||||
|
||||
monitoring:
|
||||
|
||||
@@ -36,12 +36,13 @@ spec:
|
||||
volumeMounts:
|
||||
- name: downloads
|
||||
mountPath: /downloads
|
||||
# p95 72M, max 80M over 7 days. Was 600Mi/2Gi.
|
||||
resources:
|
||||
requests:
|
||||
memory: "600Mi"
|
||||
memory: "96Mi"
|
||||
cpu: "400m"
|
||||
limits:
|
||||
memory: "2Gi"
|
||||
memory: "384Mi"
|
||||
cpu: "1700m"
|
||||
volumes:
|
||||
- name: downloads
|
||||
|
||||
@@ -7,12 +7,18 @@ spec:
|
||||
entryPoints:
|
||||
- websecure
|
||||
routes:
|
||||
# NO crowdsec-bouncer on the API routes. These are the mesh client's own
|
||||
# endpoints: gRPC-gateway management calls plus signal/relay long-polling,
|
||||
# authenticated by NetBird's token rather than by a login form. A ban here
|
||||
# is self-defeating - the client needs the mesh to reach anything else, so
|
||||
# CrowdSec banning it locks the peer out of the network it needs to
|
||||
# function. It also backfires: a banned peer keeps retrying, every retry
|
||||
# is another 403, and LePresidente/http-generic-403-bf turns five 403s in
|
||||
# ten seconds into a 4h ban, so one 403 loop kept re-arming the ban.
|
||||
# netbird-local below has always been exempt; this makes prod match.
|
||||
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`))
|
||||
kind: Rule
|
||||
priority: 100
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: netbird-server-service
|
||||
port: 80
|
||||
@@ -20,9 +26,6 @@ spec:
|
||||
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`))
|
||||
kind: Rule
|
||||
priority: 100
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: netbird-server-service
|
||||
port: 80
|
||||
|
||||
@@ -88,12 +88,13 @@ spec:
|
||||
periodSeconds: 30
|
||||
timeoutSeconds: 5
|
||||
failureThreshold: 5
|
||||
# p95 97M, max 102M over 7 days. Was 256Mi/1Gi.
|
||||
resources:
|
||||
requests:
|
||||
memory: "256Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "250m"
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
memory: "384Mi"
|
||||
cpu: "1000m"
|
||||
volumes:
|
||||
- name: netbird-data
|
||||
@@ -162,10 +163,10 @@ spec:
|
||||
failureThreshold: 5
|
||||
resources:
|
||||
requests:
|
||||
memory: "64Mi"
|
||||
memory: "32Mi"
|
||||
cpu: "50m"
|
||||
limits:
|
||||
memory: "256Mi"
|
||||
memory: "128Mi"
|
||||
cpu: "300m"
|
||||
---
|
||||
apiVersion: v1
|
||||
|
||||
@@ -97,7 +97,7 @@ spec:
|
||||
resources:
|
||||
requests:
|
||||
cpu: "100m"
|
||||
memory: "512Mi"
|
||||
memory: "1Gi"
|
||||
limits:
|
||||
cpu: "2"
|
||||
memory: "2Gi"
|
||||
@@ -164,7 +164,7 @@ spec:
|
||||
memory: "256Mi"
|
||||
limits:
|
||||
cpu: "1"
|
||||
memory: "1Gi"
|
||||
memory: "512Mi"
|
||||
volumes:
|
||||
- name: netbox-config
|
||||
configMap:
|
||||
|
||||
@@ -51,8 +51,8 @@ spec:
|
||||
key: NETRONOME__DB_PASSWORD
|
||||
resources:
|
||||
requests:
|
||||
memory: "100Mi"
|
||||
memory: "64Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
memory: "512Mi"
|
||||
memory: "256Mi"
|
||||
cpu: "500m"
|
||||
@@ -60,16 +60,26 @@ spec:
|
||||
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
|
||||
initialDelaySeconds: 10
|
||||
periodSeconds: 10
|
||||
# Generous timeout: on an I/O-bound single node even exec can take
|
||||
# seconds, and a 1s default kills a healthy postgres mid-recovery.
|
||||
timeoutSeconds: 5
|
||||
startupProbe:
|
||||
exec:
|
||||
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
|
||||
failureThreshold: 30
|
||||
# Crash recovery on an I/O-starved single node can fsync for 10+
|
||||
# minutes; killing postgres mid-recovery restarts the fsync from
|
||||
# zero and loops forever. 90x10s = 15 minutes of grace.
|
||||
failureThreshold: 90
|
||||
periodSeconds: 10
|
||||
livenessProbe:
|
||||
exec:
|
||||
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 20
|
||||
# Same I/O reasoning as readiness, plus more misses before a kill:
|
||||
# restarting postgres on a loaded node only makes recovery longer.
|
||||
timeoutSeconds: 5
|
||||
failureThreshold: 5
|
||||
resources:
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
|
||||
@@ -27,6 +27,33 @@ spec:
|
||||
summary: "Pod is crash looping"
|
||||
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} is in CrashLoopBackOff."
|
||||
|
||||
- alert: ContainerOOMKilled
|
||||
expr: max_over_time(kube_pod_container_status_terminated_reason{reason="OOMKilled"}[15m]) >= 1
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Container was OOMKilled"
|
||||
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} was killed for exceeding its memory limit. Raise the limit or reduce the workload."
|
||||
|
||||
- alert: ContainerRestartingTooOften
|
||||
expr: max by (namespace, pod, container) (increase(kube_pod_container_status_restarts_total[30m])) > 3
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Container restarting too often"
|
||||
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} restarted {{ $value }} times in the last 30 minutes."
|
||||
|
||||
- alert: PodEvicted
|
||||
expr: max_over_time(kube_pod_status_reason{reason="Evicted"}[15m]) >= 1
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Pod was evicted"
|
||||
description: "Pod {{ $labels.namespace }}/{{ $labels.pod }} was evicted, usually for node disk or memory pressure."
|
||||
|
||||
- alert: PersistentVolumeClaimFillingUp
|
||||
expr: kubelet_volume_stats_available_bytes / kubelet_volume_stats_capacity_bytes < 0.15
|
||||
for: 15m
|
||||
|
||||
@@ -26,6 +26,25 @@ grafana:
|
||||
service:
|
||||
port: 80
|
||||
|
||||
# p95 391M, observed max 1046M with no limit at all. Request is set at p95 so the
|
||||
# scheduler sees reality; the limit is a manual exception above the 1.3x max
|
||||
# formula, because a single query burst reached 1046M.
|
||||
resources:
|
||||
requests:
|
||||
memory: "416Mi"
|
||||
cpu: 100m
|
||||
limits:
|
||||
memory: "1536Mi"
|
||||
|
||||
# One block covers both the dashboards and datasources sidecars (p95 91M / 80M).
|
||||
sidecar:
|
||||
resources:
|
||||
requests:
|
||||
memory: "96Mi"
|
||||
cpu: 10m
|
||||
limits:
|
||||
memory: "192Mi"
|
||||
|
||||
additionalDataSources:
|
||||
- name: Loki
|
||||
type: loki
|
||||
@@ -48,13 +67,21 @@ prometheus:
|
||||
storage: 40Gi
|
||||
resources:
|
||||
requests:
|
||||
memory: "700Mi"
|
||||
memory: "768Mi"
|
||||
cpu: 200m
|
||||
limits:
|
||||
memory: "2Gi"
|
||||
memory: "2560Mi"
|
||||
alertmanager:
|
||||
alertmanagerSpec:
|
||||
configSecret: alertmanager-config
|
||||
# p95 66M, max 68M. Silences and notification state live here, so the request
|
||||
# stays above p95 to keep the pod out of the eviction candidates.
|
||||
resources:
|
||||
requests:
|
||||
memory: "96Mi"
|
||||
cpu: 10m
|
||||
limits:
|
||||
memory: "192Mi"
|
||||
storage:
|
||||
volumeClaimTemplate:
|
||||
spec:
|
||||
@@ -66,6 +93,39 @@ alertmanager:
|
||||
requests:
|
||||
storage: 20Gi
|
||||
|
||||
# p95 103M, max 107M, and it grows with the number of cluster objects.
|
||||
kube-state-metrics:
|
||||
# Values key is the dependency name from Chart.yaml, not the `kubeStateMetrics`
|
||||
# condition key. Setting resources under `kubeStateMetrics:` is silently ignored.
|
||||
resources:
|
||||
requests:
|
||||
memory: "128Mi"
|
||||
cpu: 50m
|
||||
limits:
|
||||
memory: "256Mi"
|
||||
|
||||
# p95 39M, max 40M. One per node, so it scales with node count.
|
||||
prometheus-node-exporter:
|
||||
resources:
|
||||
requests:
|
||||
memory: "32Mi"
|
||||
cpu: 20m
|
||||
limits:
|
||||
memory: "128Mi"
|
||||
|
||||
# p95 75M, max 75M. Creates and reconciles every PrometheusRule in the cluster.
|
||||
prometheusOperator:
|
||||
resources:
|
||||
requests:
|
||||
memory: "96Mi"
|
||||
cpu: 50m
|
||||
limits:
|
||||
memory: "192Mi"
|
||||
|
||||
# config-reloader sidecars (p95 33M, max 43M) are not covered: the chart does not
|
||||
# template `prometheusSpec.configReloader.resources`, so there is no values key
|
||||
# for them. They keep shipping with requests only.
|
||||
|
||||
defaultRules:
|
||||
disabled:
|
||||
CPUThrottlingHigh: true
|
||||
|
||||
@@ -66,10 +66,10 @@ spec:
|
||||
periodSeconds: 30
|
||||
resources:
|
||||
requests:
|
||||
memory: "128Mi"
|
||||
memory: "160Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
memory: "512Mi"
|
||||
memory: "384Mi"
|
||||
cpu: "500m"
|
||||
volumes:
|
||||
- name: rackpeek-config
|
||||
|
||||
@@ -11,10 +11,13 @@ reloader:
|
||||
replicas: 1
|
||||
# The chart defaults to no requests or limits, so the pod is evictable under node
|
||||
# pressure and the restarts go with it.
|
||||
# Memory was raised from 64Mi: measured p95 over 7 days is 73M, so the pod was
|
||||
# running above its own request and sitting in the eviction candidates. This pod
|
||||
# is the one that restarts every other pod, so it must not be evicted.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "10m"
|
||||
memory: "64Mi"
|
||||
memory: "96Mi"
|
||||
limits:
|
||||
cpu: "100m"
|
||||
memory: "128Mi"
|
||||
memory: "192Mi"
|
||||
@@ -10,6 +10,9 @@ spec:
|
||||
failedJobsHistoryLimit: 3
|
||||
jobTemplate:
|
||||
spec:
|
||||
# Reap finished job pods (including manual `create job --from` runs,
|
||||
# which history limits never delete). Keeps a day for debugging.
|
||||
ttlSecondsAfterFinished: 86400
|
||||
backoffLimit: 1
|
||||
template:
|
||||
spec:
|
||||
|
||||
@@ -38,12 +38,13 @@ spec:
|
||||
volumeMounts:
|
||||
- name: cache
|
||||
mountPath: /var/cache/searxng
|
||||
# p95 134M, max 145M over 7 days. Was 300Mi/700Mi.
|
||||
resources:
|
||||
requests:
|
||||
memory: "300Mi"
|
||||
memory: "160Mi"
|
||||
cpu: "30m"
|
||||
limits:
|
||||
memory: "700Mi"
|
||||
memory: "512Mi"
|
||||
cpu: "500m"
|
||||
volumes:
|
||||
- name: cache
|
||||
|
||||
@@ -30,6 +30,13 @@ spec:
|
||||
containers:
|
||||
- name: valkey
|
||||
image: docker.io/valkey/valkey:9.1.2-alpine
|
||||
# p95 15M, max 17M, no limit before. Matches the netbox valkey pod.
|
||||
resources:
|
||||
requests:
|
||||
cpu: "50m"
|
||||
memory: "64Mi"
|
||||
limits:
|
||||
memory: "256Mi"
|
||||
command:
|
||||
- valkey-server
|
||||
- --save
|
||||
|
||||
@@ -38,10 +38,10 @@ spec:
|
||||
mountPath: /app/data
|
||||
resources:
|
||||
requests:
|
||||
memory: "128Mi"
|
||||
memory: "160Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
memory: "256Mi"
|
||||
memory: "384Mi"
|
||||
cpu: "300m"
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: traefik-plugins
|
||||
namespace: traefik
|
||||
spec:
|
||||
accessModes: ["ReadWriteOnce"]
|
||||
storageClassName: local-path
|
||||
resources:
|
||||
requests:
|
||||
storage: 256Mi
|
||||
@@ -28,6 +28,13 @@ updateStrategy:
|
||||
|
||||
deployment:
|
||||
enabled: true
|
||||
additionalVolumes:
|
||||
- name: plugins
|
||||
persistentVolumeClaim:
|
||||
claimName: traefik-plugins
|
||||
additionalVolumeMounts:
|
||||
- name: plugins
|
||||
mountPath: /plugins-storage
|
||||
|
||||
providers:
|
||||
kubernetesIngress:
|
||||
|
||||
@@ -28,9 +28,10 @@ spec:
|
||||
containers:
|
||||
- name: uptime-kuma
|
||||
image: louislam/uptime-kuma:2.5.5
|
||||
# p95 469M, max 471M over 7 days. Was a 3Gi limit on 469M of real use.
|
||||
resources:
|
||||
limits:
|
||||
memory: "3Gi"
|
||||
memory: "1Gi"
|
||||
cpu: "1"
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
|
||||
@@ -3,7 +3,7 @@ services:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
image: gcr.forust.xyz/forust/userbot:latest
|
||||
image: gcr.forust.xyz/forust/userbot:prod
|
||||
pull_policy: build
|
||||
restart: unless-stopped
|
||||
env_file:
|
||||
@@ -18,7 +18,7 @@ services:
|
||||
- 1.1.1.1
|
||||
|
||||
anna:
|
||||
image: gcr.forust.xyz/forust/userbot:latest
|
||||
image: gcr.forust.xyz/forust/userbot:prod
|
||||
pull_policy: build
|
||||
depends_on:
|
||||
- forust
|
||||
|
||||
@@ -210,6 +210,10 @@ spec:
|
||||
- name: USERBOT_IMAGE
|
||||
# The build pushes main/prod only. The deploy resolves every gcr ref in
|
||||
# this file, so a :latest here aborts the whole apply as unresolvable.
|
||||
# Deliberately left on the tag: render_pinned only rewrites plain
|
||||
# `image:` lines to a digest, and this ref is what the panel injects
|
||||
# into the per-instance Deployments it creates. Instances track prod
|
||||
# rather than the panel's own resolved digest.
|
||||
value: gcr.forust.xyz/forust/userbot:prod
|
||||
- name: USERBOT_STORAGE_CLASS
|
||||
value: local-path-retain
|
||||
|
||||
@@ -36,10 +36,10 @@ spec:
|
||||
resources:
|
||||
requests:
|
||||
cpu: "100m"
|
||||
memory: "128Mi"
|
||||
memory: "80Mi"
|
||||
limits:
|
||||
cpu: "500m"
|
||||
memory: "512Mi"
|
||||
memory: "256Mi"
|
||||
volumeMounts:
|
||||
- name: vaultwarden-data
|
||||
mountPath: /data
|
||||
|
||||
@@ -51,10 +51,10 @@ spec:
|
||||
mountPath: /etc/x-ui
|
||||
resources:
|
||||
requests:
|
||||
memory: "128Mi"
|
||||
memory: "192Mi"
|
||||
cpu: "100m"
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
memory: "512Mi"
|
||||
cpu: "1000m"
|
||||
volumes:
|
||||
- name: x-ui-db
|
||||
|
||||
Reference in new issue
Block a user