feat(erp): manifeste d'exercice, rejeu à blanc, gardes manquantes, trunk réconcilié (#92)

Co-authored-by: Gabriel Radureau <[email protected]>
This commit was merged in pull request #92.
This commit is contained in:
2026-08-14 22:05:44 +02:00
committed by arcodange
parent cefba7b367
commit c00cf948ed
9 changed files with 1702 additions and 37 deletions
+49 -24
View File
@@ -37,6 +37,29 @@ TMP_S3_SECRET="dolibarr-backup-s3-temp"
log() { printf '\033[1;36m==>\033[0m %s\n' "$*"; }
die() { printf '\033[1;31mABORT:\033[0m %s\n' "$*" >&2; exit 1; }
# --- garde de cluster -------------------------------------------------------
# Ce script lit des secrets, crée des Jobs et peut RESTAURER une base. Lancé sur
# le mauvais contexte kubectl, il part sur l'infrastructure de quelqu'un d'autre.
# Le contexte courant d'une station de travail n'est pas une garantie : il suffit
# d'un `K config use-context` oublié. On l'épingle donc, et on vérifie une
# empreinte POSITIVE du homelab avant d'agir — mêmes garde-fous que
# ops/sandbox/sandbox-lifecycle.sh.
ERP_KUBE_CONTEXT="${ERP_KUBE_CONTEXT:-default}"
K() { kubectl --context "$ERP_KUBE_CONTEXT" "$@"; }
assert_arcodange_cluster() {
kubectl config get-contexts -o name 2>/dev/null | grep -qx "$ERP_KUBE_CONTEXT" \
|| die "kube-context '$ERP_KUBE_CONTEXT' n'existe pas (définir ERP_KUBE_CONTEXT)"
for ns in erp erp-sandbox "$S3_SRC_NS"; do
K get ns "$ns" >/dev/null 2>&1 \
|| die "le contexte '$ERP_KUBE_CONTEXT' n'a pas de namespace '$ns' — refus de s'y exécuter.
Ce script lit des secrets et peut restaurer une base ; il ne doit viser que le homelab Arcodange.
Contexte courant : '$(kubectl config current-context 2>/dev/null)'.
Définir ERP_KUBE_CONTEXT sur le contexte du homelab et réessayer."
done
log "garde de cluster OK — contexte '$ERP_KUBE_CONTEXT' (namespaces erp/erp-sandbox/$S3_SRC_NS)"
}
CMD="${1:-}"; shift || true
ENV="prod"; KEY=""; KIND=""; YES=0
while [[ $# -gt 0 ]]; do
@@ -72,11 +95,11 @@ SH
copy_s3_secret() {
command -v python3 >/dev/null || die "python3 required to copy the S3 secret without exposing it"
kubectl get secret "$S3_SRC_SECRET" -n "$S3_SRC_NS" -o json \
K get secret "$S3_SRC_SECRET" -n "$S3_SRC_NS" -o json \
| python3 -c "import json,sys; d=json.load(sys.stdin); d['metadata']={'name':'$TMP_S3_SECRET','namespace':'$NS'}; d.pop('status',None); d['data']={k:d['data'][k] for k in ('AWS_ACCESS_KEY_ID','AWS_SECRET_ACCESS_KEY','AWS_ENDPOINTS')}; print(json.dumps(d))" \
| kubectl apply -f - >/dev/null
| K apply -f - >/dev/null
}
cleanup_secret() { kubectl delete secret "$TMP_S3_SECRET" -n "$NS" --ignore-not-found >/dev/null 2>&1 || true; }
cleanup_secret() { K delete secret "$TMP_S3_SECRET" -n "$NS" --ignore-not-found >/dev/null 2>&1 || true; }
# b64-encode an in-container script (host vars already substituted by the caller)
b64() { printf '%s' "$1" | base64 | tr -d '\n'; }
@@ -87,8 +110,8 @@ run_backup() {
copy_s3_secret
log "Backup ${ENV}: DB=$DB PVC=$PVC -> s3://$BUCKET/$PREFIX/{db,docs}/"
local B64; B64="$(b64 "$(cat "${SCRIPT_DIR}/../../chart/files/backup-job.sh")")"
kubectl delete job dolibarr-backup -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
kubectl apply -f - >/dev/null <<EOF
K delete job dolibarr-backup -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
K apply -f - >/dev/null <<EOF
apiVersion: batch/v1
kind: Job
metadata: { name: dolibarr-backup, namespace: $NS }
@@ -118,10 +141,10 @@ spec:
command: ["/bin/sh","-c"]
args: ["echo $B64 | base64 -d | sh"]
EOF
kubectl wait --for=condition=complete job/dolibarr-backup -n "$NS" --timeout=300s >/dev/null 2>&1 \
|| die "backup Job did not complete — kubectl logs -n $NS job/dolibarr-backup"
kubectl logs -n "$NS" job/dolibarr-backup | sed 's/^/ /'
kubectl delete job dolibarr-backup -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
K wait --for=condition=complete job/dolibarr-backup -n "$NS" --timeout=300s >/dev/null 2>&1 \
|| die "backup Job did not complete — K logs -n $NS job/dolibarr-backup"
K logs -n "$NS" job/dolibarr-backup | sed 's/^/ /'
K delete job dolibarr-backup -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
cleanup_secret; trap - EXIT
log "Backup complete."
}
@@ -135,8 +158,8 @@ echo "db/:"; S3 ls "s3://$BUCKET/$PREFIX/db/" || echo " (empty)"
echo "docs/:"; S3 ls "s3://$BUCKET/$PREFIX/docs/" || echo " (empty)"
EOF
)"
kubectl delete job dolibarr-backup-list -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
kubectl apply -f - >/dev/null <<EOF
K delete job dolibarr-backup-list -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
K apply -f - >/dev/null <<EOF
apiVersion: batch/v1
kind: Job
metadata: { name: dolibarr-backup-list, namespace: $NS }
@@ -153,9 +176,9 @@ spec:
command: ["/bin/sh","-c"]
args: ["echo $(b64 "$SCRIPT") | base64 -d | sh"]
EOF
kubectl wait --for=condition=complete job/dolibarr-backup-list -n "$NS" --timeout=180s >/dev/null 2>&1 || true
kubectl logs -n "$NS" job/dolibarr-backup-list 2>/dev/null | sed 's/^/ /'
kubectl delete job dolibarr-backup-list -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
K wait --for=condition=complete job/dolibarr-backup-list -n "$NS" --timeout=180s >/dev/null 2>&1 || true
K logs -n "$NS" job/dolibarr-backup-list 2>/dev/null | sed 's/^/ /'
K delete job dolibarr-backup-list -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
cleanup_secret; trap - EXIT
}
@@ -167,8 +190,8 @@ run_restore() {
copy_s3_secret
local DEPLOY="$NS" # release/instance name == namespace (erp / erp-sandbox)
log "Restore ${KIND} on '${ENV}' from ${KEY} (scaling ${DEPLOY} to 0)"
kubectl scale deploy "$DEPLOY" -n "$NS" --replicas=0 >/dev/null 2>&1 || true
kubectl wait --for=delete pod -l app.kubernetes.io/instance="$NS" -n "$NS" --timeout=120s >/dev/null 2>&1 || true
K scale deploy "$DEPLOY" -n "$NS" --replicas=0 >/dev/null 2>&1 || true
K wait --for=delete pod -l app.kubernetes.io/instance="$NS" -n "$NS" --timeout=120s >/dev/null 2>&1 || true
local SCRIPT VOLS="[]" MOUNTS="[]"
if [[ "$KIND" == "db" ]]; then
@@ -197,8 +220,8 @@ EOF
)"
fi
local B64; B64="$(b64 "$SCRIPT")"
kubectl delete job dolibarr-restore -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
kubectl apply -f - >/dev/null <<EOF
K delete job dolibarr-restore -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
K apply -f - >/dev/null <<EOF
apiVersion: batch/v1
kind: Job
metadata: { name: dolibarr-restore, namespace: $NS }
@@ -221,19 +244,21 @@ spec:
command: ["/bin/sh","-c"]
args: ["echo $B64 | base64 -d | sh"]
EOF
if ! kubectl wait --for=condition=complete job/dolibarr-restore -n "$NS" --timeout=300s >/dev/null 2>&1; then
kubectl logs -n "$NS" job/dolibarr-restore 2>&1 | sed 's/^/ /'
kubectl scale deploy "$DEPLOY" -n "$NS" --replicas=1 >/dev/null 2>&1 || true
if ! K wait --for=condition=complete job/dolibarr-restore -n "$NS" --timeout=300s >/dev/null 2>&1; then
K logs -n "$NS" job/dolibarr-restore 2>&1 | sed 's/^/ /'
K scale deploy "$DEPLOY" -n "$NS" --replicas=1 >/dev/null 2>&1 || true
die "restore Job did not complete"
fi
kubectl logs -n "$NS" job/dolibarr-restore | sed 's/^/ /'
kubectl delete job dolibarr-restore -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
K logs -n "$NS" job/dolibarr-restore | sed 's/^/ /'
K delete job dolibarr-restore -n "$NS" --ignore-not-found >/dev/null 2>&1 || true
log "Scaling ${DEPLOY} back to 1"
kubectl scale deploy "$DEPLOY" -n "$NS" --replicas=1 >/dev/null 2>&1 || true
K scale deploy "$DEPLOY" -n "$NS" --replicas=1 >/dev/null 2>&1 || true
cleanup_secret; trap - EXIT
log "Restore complete."
}
assert_arcodange_cluster
case "$CMD" in
backup) run_backup ;;
list) run_list ;;