From 9c5705b66adbb17ebf7faa6fcae8661cf00b6b20 Mon Sep 17 00:00:00 2001 From: Sabine Maennel <5292683+sabinem@users.noreply.github.com> Date: Mon, 5 Oct 2026 14:24:40 +0200 Subject: [PATCH 1/2] chore(just): add cluster module to back up, wipe and reseed dev The deployed dev instance had no supported way to reset its data: emptying it meant hand-running kubectl and psql against the cluster, and the seed binary is not shipped in the backend image. `just cluster::{backup,wipe,reseed} ` does it from a laptop. Every recipe, including the private _wipe-db, takes the kube context explicitly and refuses a cluster whose ingress does not serve hackagon-dev.dscompute.ch; wipe and reseed take a pg_dumpall first and ask for the context name to be typed back. Only the `hackagon` database is recreated, so Keycloak and its logins survive. reseed refuses unless HEAD contains origin/main and components/backend matches it, since the seed migrates the schema before writing anything. --- CLAUDE.md | 5 + justfile | 1 + tools/just/cluster.just | 200 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 206 insertions(+) create mode 100644 tools/just/cluster.just diff --git a/CLAUDE.md b/CLAUDE.md index cceac870..c0015c21 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -108,6 +108,11 @@ just helm::lint # helm lint + full render against tools/helm/lin just helm::template [args] # render the chart to stdout just helm::check-bump # fail if helm-chart/ changed without a version bump just helm::publish # push the chart if Chart.yaml's version is unpublished + +# Deployed dev instance only — takes the kube context, refuses any other cluster +just cluster::backup # pg_dumpall to ~/hackagon-backups +just cluster::wipe # backup, then drop + recreate the app DB; dev restarts empty +just cluster::reseed # backup, wipe, then seed; refuses if components/backend differs from origin/main ``` Backend listens on **:3000**, frontend on **:8081**. Dev users (Keycloak diff --git a/justfile b/justfile index 6e4ab262..da7b8444 100644 --- a/justfile +++ b/justfile @@ -15,6 +15,7 @@ mod codegen "./tools/just/codegen.just" mod clean "./tools/just/clean.just" mod version "./tools/just/version.just" mod helm "./tools/just/helm.just" +mod cluster "./tools/just/cluster.just" [private] default: diff --git a/tools/just/cluster.just b/tools/just/cluster.just new file mode 100644 index 00000000..ee83951c --- /dev/null +++ b/tools/just/cluster.just @@ -0,0 +1,200 @@ +set positional-arguments +set shell := ["bash", "-cue"] + +root_dir := `git rev-parse --show-toplevel` + +# Names the chart gives a release called `hackagon` (see helm-chart/templates). +namespace := "hackagon" +backend_deploy := "hackagon-backend" +backend_config := "hackagon-backend-config" +backend_db_secret := "hackagon-backend-db" +postgres_sts := "hackagon-postgresql" +app_db := "hackagon" +app_db_owner := "hackagon" + +# Refuse any cluster whose ingress does not serve this domain, so a recipe +# pointed at the prod context by mistake stops before touching anything. +dev_domain := "hackagon-dev.dscompute.ch" + +backup_dir := env("HOME") + "/hackagon-backups" +local_pg_port := "15432" + +# Runs inside the postgres pod: finds the superuser password the bitnami image +# was started with (env var or mounted file) so it never leaves the pod. +pg_as_superuser := ''' +pw="${POSTGRES_POSTGRES_PASSWORD:-${POSTGRES_PASSWORD:-}}" +if [ -z "$pw" ]; then + f="${POSTGRES_POSTGRES_PASSWORD_FILE:-${POSTGRES_PASSWORD_FILE:-}}" + [ -n "$f" ] && pw="$(cat "$f")" +fi +PGPASSWORD="$pw" exec "$@" +''' + +default: + just --list --unsorted -f "{{source_file()}}" + +# Show usage examples for cluster commands. +[group('cluster')] +help: + #!/usr/bin/env bash + bold="\033[1m" + dim="\033[2m" + cyan="\033[36m" + reset="\033[0m" + + echo "" + echo -e " ${bold}Dev Cluster Commands${reset}" + echo -e " ${dim}Act on the deployed dev instance. Every recipe takes the kube context${reset}" + echo -e " ${dim}explicitly and refuses a cluster not serving {{dev_domain}}.${reset}" + echo "" + echo -e " ${cyan}just cluster::backup${reset} " + echo -e " ${dim}pg_dumpall the dev postgres to {{backup_dir}}${reset}" + echo "" + echo -e " ${cyan}just cluster::wipe${reset} " + echo -e " ${dim}Back up, then drop and recreate the app database. Keycloak is untouched.${reset}" + echo -e " ${dim}The backend recreates the schema on startup, so dev comes back empty.${reset}" + echo "" + echo -e " ${cyan}just cluster::reseed${reset} " + echo -e " ${dim}Back up, wipe, then run cmd/seed (backend code must match origin/main).${reset}" + echo "" + echo -e " ${dim}e.g. just cluster::wipe sck-sit-dev${reset}" + echo "" + +# pg_dumpall the dev postgres (app and keycloak databases) to ~/hackagon-backups. +[group('cluster')] +backup context: + #!/usr/bin/env bash + set -euo pipefail + just -f "{{source_file()}}" _guard "{{context}}" + mkdir -p "{{backup_dir}}" + out="{{backup_dir}}/{{context}}-$(date +%Y%m%d-%H%M%S).sql" + echo "==> Backing up to $out" + kubectl --context "{{context}}" -n "{{namespace}}" exec "sts/{{postgres_sts}}" -- \ + sh -c '{{pg_as_superuser}}' sh pg_dumpall -h 127.0.0.1 -U postgres > "$out" + echo "✓ $(du -h "$out" | cut -f1) written" + +# Back up, then drop and recreate the app database; dev restarts empty. +[group('cluster')] +wipe context: + #!/usr/bin/env bash + set -euo pipefail + just -f "{{source_file()}}" _confirm "{{context}}" "WIPE the app database" + just -f "{{source_file()}}" backup "{{context}}" + just -f "{{source_file()}}" _wipe-db "{{context}}" + +# Back up, wipe, then seed the dev fixture from this checkout. +[group('cluster')] +reseed context: + #!/usr/bin/env bash + set -euo pipefail + cd "{{root_dir}}" + + # The seed migrates the schema before it writes anything, so its backend + # code must be the code dev is serving — temporary/*:latest is built from + # origin/main. Changes outside components/backend (these recipes, docs) + # do not matter, so a branch that only touches those may seed. + git fetch --quiet origin main + if ! git merge-base --is-ancestor origin/main HEAD; then + echo "✗ This checkout is behind origin/main. Rebase or merge main first." >&2 + exit 1 + fi + if ! git diff --quiet origin/main -- components/backend; then + echo "✗ components/backend differs from origin/main, which dev is serving." >&2 + exit 1 + fi + if [ -n "$(git status --porcelain components/backend)" ]; then + echo "✗ components/backend has uncommitted changes." >&2 + exit 1 + fi + + just -f "{{source_file()}}" _confirm "{{context}}" "WIPE and RESEED the app database" + just -f "{{source_file()}}" backup "{{context}}" + + kc=(kubectl --context "{{context}}" -n "{{namespace}}") + replicas="$("${kc[@]}" get "deploy/{{backend_deploy}}" -o jsonpath='{.spec.replicas}')" + [ "${replicas:-0}" -gt 0 ] || replicas=1 + tmp="$(mktemp -d)" + cleanup() { + [ -n "${pf_pid:-}" ] && kill "$pf_pid" 2>/dev/null || true + rm -rf "$tmp" + # Starting fresh is also what loads the casbin rows the seed wrote. + echo "==> Scaling backend back to $replicas" + "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas="$replicas" + } + trap cleanup EXIT + + KEEP_BACKEND_DOWN=1 just -f "{{source_file()}}" _wipe-db "{{context}}" + + "${kc[@]}" get cm "{{backend_config}}" -o jsonpath='{.data.config\.yaml}' > "$tmp/config.yaml" + + echo "==> Port-forwarding postgres to localhost:{{local_pg_port}}" + "${kc[@]}" port-forward "svc/{{postgres_sts}}" "{{local_pg_port}}:5432" >/dev/null & + pf_pid=$! + for _ in $(seq 20); do + nc -z 127.0.0.1 "{{local_pg_port}}" 2>/dev/null && break + sleep 0.5 + done + + echo "==> Seeding" + cd components/backend + HACKAGON_DATABASE_HOST=127.0.0.1 \ + HACKAGON_DATABASE_PORT="{{local_pg_port}}" \ + HACKAGON_DATABASE_PASSWORD="$("${kc[@]}" get secret "{{backend_db_secret}}" -o jsonpath='{.data.password}' | base64 -d)" \ + go run ./cmd/seed/ --config-dir "$tmp/" + echo "✓ Seeded" + +# Scale the backend down, drop and recreate the app database, scale it back up. +[private] +_wipe-db context: + #!/usr/bin/env bash + set -euo pipefail + # [private] only hides a recipe from --list; it can still be called directly. + just -f "{{source_file()}}" _guard "{{context}}" + kc=(kubectl --context "{{context}}" -n "{{namespace}}") + + replicas="$("${kc[@]}" get "deploy/{{backend_deploy}}" -o jsonpath='{.spec.replicas}')" + [ "${replicas:-0}" -gt 0 ] || replicas=1 + + # Under reseed the caller scales the backend back up after seeding; + # otherwise bring it back whether or not the wipe succeeded. + if [ -z "${KEEP_BACKEND_DOWN:-}" ]; then + trap 'echo "==> Scaling backend back to $replicas (it recreates the schema on startup)" + "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas="$replicas"' EXIT + fi + + echo "==> Scaling backend to 0" + "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas=0 + while [ -n "$("${kc[@]}" get pods -o name | grep "{{backend_deploy}}-" || true)" ]; do + sleep 2 + done + + echo "==> Recreating database {{app_db}}" + "${kc[@]}" exec "sts/{{postgres_sts}}" -- sh -c '{{pg_as_superuser}}' sh \ + psql -h 127.0.0.1 -U postgres -v ON_ERROR_STOP=1 \ + -c 'DROP DATABASE IF EXISTS {{app_db}} WITH (FORCE);' \ + -c 'CREATE DATABASE {{app_db}} OWNER {{app_db_owner}};' + echo "✓ Database {{app_db}} is empty" + +# Refuse a context whose hackagon ingress does not serve the dev domain. +[private] +_guard context: + #!/usr/bin/env bash + set -euo pipefail + hosts="$(kubectl --context "{{context}}" -n "{{namespace}}" get ingress \ + -o jsonpath='{.items[*].spec.rules[*].host}')" + if [[ " $hosts " != *"{{dev_domain}}"* ]]; then + echo "✗ Context '{{context}}' does not serve {{dev_domain}} (found: ${hosts:-none})." >&2 + echo " Refusing — this does not look like the dev instance." >&2 + exit 1 + fi + +# Guard, then make the operator type the context name back. +[private] +_confirm context action: + #!/usr/bin/env bash + set -euo pipefail + just -f "{{source_file()}}" _guard "{{context}}" + echo "About to {{action}} on context '{{context}}' (namespace {{namespace}})." + echo "Everything in the app database is lost; a backup is taken first." + read -r -p "Type the context name to continue: " answer + [ "$answer" = "{{context}}" ] || { echo "Aborted."; exit 1; } From eed9695e4a77e699a336423957b23bbe50e88e01 Mon Sep 17 00:00:00 2001 From: Sabine Maennel <5292683+sabinem@users.noreply.github.com> Date: Tue, 6 Oct 2026 19:54:31 +0200 Subject: [PATCH 2/2] refactor(just): fail before wiping, and make backend restarts explicit reseed now builds the seed and fetches the backend's config and database password before it confirms, backs up or wipes, so a compile error or a wrong resource name leaves dev untouched rather than empty. _wipe-db no longer restarts the backend behind an environment flag; wipe and reseed each bring it back through a named restore_backend in an EXIT trap. The backup is written to .partial and renamed only on success, the port-forward wait fails with a message, and the wait for backend pods gives up after two minutes. --- tools/just/cluster.just | 131 +++++++++++++++++++++++++++------------- 1 file changed, 90 insertions(+), 41 deletions(-) diff --git a/tools/just/cluster.just b/tools/just/cluster.just index ee83951c..cff7188d 100644 --- a/tools/just/cluster.just +++ b/tools/just/cluster.just @@ -68,9 +68,16 @@ backup context: just -f "{{source_file()}}" _guard "{{context}}" mkdir -p "{{backup_dir}}" out="{{backup_dir}}/{{context}}-$(date +%Y%m%d-%H%M%S).sql" + + # Dump to a .partial file and rename it only once the dump succeeded, so a + # failed run never leaves behind something that looks like a backup. + discard_partial() { rm -f "$out.partial"; } + trap discard_partial EXIT + echo "==> Backing up to $out" kubectl --context "{{context}}" -n "{{namespace}}" exec "sts/{{postgres_sts}}" -- \ - sh -c '{{pg_as_superuser}}' sh pg_dumpall -h 127.0.0.1 -U postgres > "$out" + sh -c '{{pg_as_superuser}}' sh pg_dumpall -h 127.0.0.1 -U postgres > "$out.partial" + mv "$out.partial" "$out" echo "✓ $(du -h "$out" | cut -f1) written" # Back up, then drop and recreate the app database; dev restarts empty. @@ -80,6 +87,18 @@ wipe context: set -euo pipefail just -f "{{source_file()}}" _confirm "{{context}}" "WIPE the app database" just -f "{{source_file()}}" backup "{{context}}" + + kc=(kubectl --context "{{context}}" -n "{{namespace}}") + replicas="$("${kc[@]}" get "deploy/{{backend_deploy}}" -o jsonpath='{.spec.replicas}')" + [ "${replicas:-0}" -gt 0 ] || replicas=1 + + restore_backend() { + echo "==> Scaling backend back to $replicas (it recreates the schema on startup)" + "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas="$replicas" + } + # _wipe-db leaves the backend down; bring it back however this ends. + trap restore_backend EXIT + just -f "{{source_file()}}" _wipe-db "{{context}}" # Back up, wipe, then seed the dev fixture from this checkout. @@ -89,84 +108,93 @@ reseed context: set -euo pipefail cd "{{root_dir}}" - # The seed migrates the schema before it writes anything, so its backend - # code must be the code dev is serving — temporary/*:latest is built from - # origin/main. Changes outside components/backend (these recipes, docs) - # do not matter, so a branch that only touches those may seed. - git fetch --quiet origin main - if ! git merge-base --is-ancestor origin/main HEAD; then - echo "✗ This checkout is behind origin/main. Rebase or merge main first." >&2 - exit 1 - fi - if ! git diff --quiet origin/main -- components/backend; then - echo "✗ components/backend differs from origin/main, which dev is serving." >&2 - exit 1 - fi - if [ -n "$(git status --porcelain components/backend)" ]; then - echo "✗ components/backend has uncommitted changes." >&2 - exit 1 - fi - - just -f "{{source_file()}}" _confirm "{{context}}" "WIPE and RESEED the app database" - just -f "{{source_file()}}" backup "{{context}}" + just -f "{{source_file()}}" _backend-matches-main + just -f "{{source_file()}}" _guard "{{context}}" kc=(kubectl --context "{{context}}" -n "{{namespace}}") - replicas="$("${kc[@]}" get "deploy/{{backend_deploy}}" -o jsonpath='{.spec.replicas}')" - [ "${replicas:-0}" -gt 0 ] || replicas=1 tmp="$(mktemp -d)" - cleanup() { - [ -n "${pf_pid:-}" ] && kill "$pf_pid" 2>/dev/null || true - rm -rf "$tmp" + pf_pid="" + backend_down="" + + restore_backend() { # Starting fresh is also what loads the casbin rows the seed wrote. echo "==> Scaling backend back to $replicas" "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas="$replicas" } + # However this ends: close the tunnel, delete the temp dir, and bring the + # backend back if _wipe-db was reached (it leaves the backend down). + cleanup() { + [ -n "$pf_pid" ] && kill "$pf_pid" 2>/dev/null || true + rm -rf "$tmp" + [ -z "$backend_down" ] || restore_backend + } trap cleanup EXIT - KEEP_BACKEND_DOWN=1 just -f "{{source_file()}}" _wipe-db "{{context}}" + # Everything that can fail without changing the cluster happens first, so + # a failure here leaves dev exactly as it was. + echo "==> Building the seed" + (cd components/backend && go build -o "$tmp/seed" ./cmd/seed/) + echo "==> Fetching backend config and database password from the cluster" "${kc[@]}" get cm "{{backend_config}}" -o jsonpath='{.data.config\.yaml}' > "$tmp/config.yaml" + [ -s "$tmp/config.yaml" ] || { echo "✗ {{backend_config}} has no config.yaml." >&2; exit 1; } + db_password="$("${kc[@]}" get secret "{{backend_db_secret}}" -o jsonpath='{.data.password}' | base64 -d)" + [ -n "$db_password" ] || { echo "✗ {{backend_db_secret}} has no password." >&2; exit 1; } + + replicas="$("${kc[@]}" get "deploy/{{backend_deploy}}" -o jsonpath='{.spec.replicas}')" + [ "${replicas:-0}" -gt 0 ] || replicas=1 + + just -f "{{source_file()}}" _confirm "{{context}}" "WIPE and RESEED the app database" + just -f "{{source_file()}}" backup "{{context}}" + + # Wipe the app database, leaving keycloak untouched. The seed creates the schema. + backend_down=yes + just -f "{{source_file()}}" _wipe-db "{{context}}" echo "==> Port-forwarding postgres to localhost:{{local_pg_port}}" "${kc[@]}" port-forward "svc/{{postgres_sts}}" "{{local_pg_port}}:5432" >/dev/null & pf_pid=$! + + # Wait up to 10s for the tunnel to open. for _ in $(seq 20); do nc -z 127.0.0.1 "{{local_pg_port}}" 2>/dev/null && break sleep 0.5 done + nc -z 127.0.0.1 "{{local_pg_port}}" 2>/dev/null \ + || { echo "✗ The port-forward to postgres did not open within 10s." >&2; exit 1; } echo "==> Seeding" cd components/backend HACKAGON_DATABASE_HOST=127.0.0.1 \ HACKAGON_DATABASE_PORT="{{local_pg_port}}" \ - HACKAGON_DATABASE_PASSWORD="$("${kc[@]}" get secret "{{backend_db_secret}}" -o jsonpath='{.data.password}' | base64 -d)" \ - go run ./cmd/seed/ --config-dir "$tmp/" + HACKAGON_DATABASE_PASSWORD="$db_password" \ + "$tmp/seed" --config-dir "$tmp/" echo "✓ Seeded" -# Scale the backend down, drop and recreate the app database, scale it back up. +# Scale the backend down, then drop and recreate the app database. Leaves the +# backend down: scaling it back up is the caller's job, done in an EXIT trap set +# before calling, so it happens even when this fails halfway. [private] _wipe-db context: #!/usr/bin/env bash set -euo pipefail - # [private] only hides a recipe from --list; it can still be called directly. + + # Guard the context, so a recipe pointed at the prod context by mistake stops before touching anything. just -f "{{source_file()}}" _guard "{{context}}" kc=(kubectl --context "{{context}}" -n "{{namespace}}") - replicas="$("${kc[@]}" get "deploy/{{backend_deploy}}" -o jsonpath='{.spec.replicas}')" - [ "${replicas:-0}" -gt 0 ] || replicas=1 - - # Under reseed the caller scales the backend back up after seeding; - # otherwise bring it back whether or not the wipe succeeded. - if [ -z "${KEEP_BACKEND_DOWN:-}" ]; then - trap 'echo "==> Scaling backend back to $replicas (it recreates the schema on startup)" - "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas="$replicas"' EXIT - fi + backend_pods() { "${kc[@]}" get pods -o name | grep "{{backend_deploy}}-" || true; } echo "==> Scaling backend to 0" "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas=0 - while [ -n "$("${kc[@]}" get pods -o name | grep "{{backend_deploy}}-" || true)" ]; do + + # Wait up to 2 minutes for its pods to be gone. + for _ in $(seq 60); do + [ -z "$(backend_pods)" ] && break sleep 2 done + [ -z "$(backend_pods)" ] \ + || { echo "✗ Backend pods still running after 2 minutes." >&2; exit 1; } echo "==> Recreating database {{app_db}}" "${kc[@]}" exec "sts/{{postgres_sts}}" -- sh -c '{{pg_as_superuser}}' sh \ @@ -175,6 +203,27 @@ _wipe-db context: -c 'CREATE DATABASE {{app_db}} OWNER {{app_db_owner}};' echo "✓ Database {{app_db}} is empty" +# Refuse unless this checkout's backend is the code dev serves. +[private] +_backend-matches-main: + #!/usr/bin/env bash + set -euo pipefail + cd "{{root_dir}}" + git fetch --quiet origin main + if ! git merge-base --is-ancestor origin/main HEAD; then + echo "✗ This checkout is behind origin/main. Rebase or merge main first." >&2 + exit 1 + fi + if ! git diff --quiet origin/main -- components/backend; then + echo "✗ components/backend differs from origin/main, which dev is serving." >&2 + exit 1 + fi + if [ -n "$(git status --porcelain components/backend)" ]; then + echo "✗ components/backend has uncommitted changes." >&2 + exit 1 + fi + echo "✓ components/backend matches origin/main" + # Refuse a context whose hackagon ingress does not serve the dev domain. [private] _guard context: