fix(ops): pin the kube-context — never run destructive steps on the ambient one #76
@@ -38,20 +38,48 @@ TMP_PROD_SECRET="prod-db-ro-temp" # transient copy of prod creds, deleted on e
|
|||||||
log() { printf '\033[1;36m==>\033[0m %s\n' "$*"; }
|
log() { printf '\033[1;36m==>\033[0m %s\n' "$*"; }
|
||||||
die() { printf '\033[1;31mABORT:\033[0m %s\n' "$*" >&2; exit 1; }
|
die() { printf '\033[1;31mABORT:\033[0m %s\n' "$*" >&2; exit 1; }
|
||||||
|
|
||||||
sb_pod() { kubectl get pod -n "$SB_NS" -l app.kubernetes.io/instance=erp-sandbox -o name 2>/dev/null | head -1; }
|
sb_pod() { K get pod -n "$SB_NS" -l app.kubernetes.io/instance=erp-sandbox -o name 2>/dev/null | head -1; }
|
||||||
prod_pod() { kubectl get pod -n "$PROD_NS" -l app.kubernetes.io/instance=erp -o name 2>/dev/null | head -1; }
|
prod_pod() { K get pod -n "$PROD_NS" -l app.kubernetes.io/instance=erp -o name 2>/dev/null | head -1; }
|
||||||
|
|
||||||
cleanup_secret() { kubectl delete secret "$TMP_PROD_SECRET" -n "$SB_NS" --ignore-not-found >/dev/null 2>&1 || true; }
|
cleanup_secret() { K delete secret "$TMP_PROD_SECRET" -n "$SB_NS" --ignore-not-found >/dev/null 2>&1 || true; }
|
||||||
|
|
||||||
# erp-sandbox is ArgoCD-managed with self-heal ON, which reverts `kubectl scale
|
# erp-sandbox is ArgoCD-managed with self-heal ON, which reverts `kubectl scale
|
||||||
# --replicas=0` within seconds — so without pausing it the seed runs while Dolibarr
|
# --replicas=0` within seconds — so without pausing it the seed runs while Dolibarr
|
||||||
# is still connected, and the restore collides with the app re-creating tables.
|
# is still connected, and the restore collides with the app re-creating tables.
|
||||||
# Pause self-heal for the duration so the scale-down holds; always re-arm it.
|
# Pause self-heal for the duration so the scale-down holds; always re-arm it.
|
||||||
ARGOCD_NS="argocd"; ARGOCD_APP="erp-sandbox"
|
ARGOCD_NS="argocd"; ARGOCD_APP="erp-sandbox"
|
||||||
set_selfheal() { kubectl patch application "$ARGOCD_APP" -n "$ARGOCD_NS" --type merge \
|
|
||||||
|
# --- Cluster guard -----------------------------------------------------------
|
||||||
|
# This script scales deployments to zero, patches the ArgoCD Application and runs
|
||||||
|
# `DROP OWNED ... CASCADE`. Until now every one of those ran against whatever
|
||||||
|
# kube-context happened to be current — and this workstation also carries a
|
||||||
|
# CLIENT production cluster (observed 2026-07-25: the current context was the
|
||||||
|
# client's, and refresh only failed because that cluster has no `erp` namespace).
|
||||||
|
# Never trust the ambient context: resolve an explicit one and prove it is the
|
||||||
|
# Arcodange homelab before touching anything.
|
||||||
|
ERP_KUBE_CONTEXT="${ERP_KUBE_CONTEXT:-default}"
|
||||||
|
K() { kubectl --context "$ERP_KUBE_CONTEXT" "$@"; }
|
||||||
|
|
||||||
|
assert_arcodange_cluster() {
|
||||||
|
kubectl config get-contexts -o name 2>/dev/null | grep -qx "$ERP_KUBE_CONTEXT" \
|
||||||
|
|| die "kube-context '$ERP_KUBE_CONTEXT' does not exist (set ERP_KUBE_CONTEXT)"
|
||||||
|
# Positive fingerprint: the three namespaces AND the ArgoCD Application this
|
||||||
|
# script drives. A client cluster cannot match all four by accident.
|
||||||
|
for ns in "$PROD_NS" "$SB_NS" "$ARGOCD_NS"; do
|
||||||
|
K get ns "$ns" >/dev/null 2>&1 \
|
||||||
|
|| die "kube-context '$ERP_KUBE_CONTEXT' has no '$ns' namespace — refusing to run against it.
|
||||||
|
This script is destructive (scale-to-0, DROP OWNED CASCADE, ArgoCD patch) and must only
|
||||||
|
ever target the Arcodange homelab. Current context is '$(kubectl config current-context 2>/dev/null)'.
|
||||||
|
Set ERP_KUBE_CONTEXT to the homelab context and retry."
|
||||||
|
done
|
||||||
|
K get application "$ARGOCD_APP" -n "$ARGOCD_NS" >/dev/null 2>&1 \
|
||||||
|
|| die "kube-context '$ERP_KUBE_CONTEXT' has no ArgoCD Application '$ARGOCD_APP' — refusing to run."
|
||||||
|
log "cluster guard OK — context '$ERP_KUBE_CONTEXT' (namespaces $PROD_NS/$SB_NS/$ARGOCD_NS + app $ARGOCD_APP)"
|
||||||
|
}
|
||||||
|
set_selfheal() { K patch application "$ARGOCD_APP" -n "$ARGOCD_NS" --type merge \
|
||||||
-p "{\"spec\":{\"syncPolicy\":{\"automated\":{\"selfHeal\":$1,\"prune\":true}}}}" >/dev/null 2>&1 || true; }
|
-p "{\"spec\":{\"syncPolicy\":{\"automated\":{\"selfHeal\":$1,\"prune\":true}}}}" >/dev/null 2>&1 || true; }
|
||||||
# Safety net (EXIT trap): whatever happens, bring the app back, re-arm self-heal, drop the secret.
|
# Safety net (EXIT trap): whatever happens, bring the app back, re-arm self-heal, drop the secret.
|
||||||
restore_state() { set_selfheal true; kubectl scale deploy erp-sandbox -n "$SB_NS" --replicas=1 >/dev/null 2>&1 || true; cleanup_secret; }
|
restore_state() { set_selfheal true; K scale deploy erp-sandbox -n "$SB_NS" --replicas=1 >/dev/null 2>&1 || true; cleanup_secret; }
|
||||||
|
|
||||||
refresh_from_prod() {
|
refresh_from_prod() {
|
||||||
command -v python3 >/dev/null || die "python3 required to copy the prod secret without exposing it"
|
command -v python3 >/dev/null || die "python3 required to copy the prod secret without exposing it"
|
||||||
@@ -61,17 +89,17 @@ refresh_from_prod() {
|
|||||||
set_selfheal false
|
set_selfheal false
|
||||||
|
|
||||||
log "Copying prod DB creds into a transient, read-only-intent secret in $SB_NS (values stay base64)"
|
log "Copying prod DB creds into a transient, read-only-intent secret in $SB_NS (values stay base64)"
|
||||||
kubectl get secret vso-db-credentials -n "$PROD_NS" -o json \
|
K get secret vso-db-credentials -n "$PROD_NS" -o json \
|
||||||
| python3 -c "import json,sys; d=json.load(sys.stdin); d['metadata']={'name':'$TMP_PROD_SECRET','namespace':'$SB_NS'}; d.pop('status',None); d['data']={k:d['data'][k] for k in ('username','password')}; print(json.dumps(d))" \
|
| python3 -c "import json,sys; d=json.load(sys.stdin); d['metadata']={'name':'$TMP_PROD_SECRET','namespace':'$SB_NS'}; d.pop('status',None); d['data']={k:d['data'][k] for k in ('username','password')}; print(json.dumps(d))" \
|
||||||
| kubectl apply -f - >/dev/null
|
| K apply -f - >/dev/null
|
||||||
|
|
||||||
log "Scaling erp-sandbox to 0 (exclusive DB access for the restore)"
|
log "Scaling erp-sandbox to 0 (exclusive DB access for the restore)"
|
||||||
kubectl scale deploy erp-sandbox -n "$SB_NS" --replicas=0 >/dev/null
|
K scale deploy erp-sandbox -n "$SB_NS" --replicas=0 >/dev/null
|
||||||
kubectl wait --for=delete pod -l app.kubernetes.io/instance=erp-sandbox -n "$SB_NS" --timeout=120s >/dev/null 2>&1 || true
|
K wait --for=delete pod -l app.kubernetes.io/instance=erp-sandbox -n "$SB_NS" --timeout=120s >/dev/null 2>&1 || true
|
||||||
|
|
||||||
log "Running the seed Job (pg_dump prod read-only -> DROP OWNED -> pg_restore into sandbox)"
|
log "Running the seed Job (pg_dump prod read-only -> DROP OWNED -> pg_restore into sandbox)"
|
||||||
kubectl delete job sandbox-seed -n "$SB_NS" --ignore-not-found >/dev/null 2>&1 || true
|
K delete job sandbox-seed -n "$SB_NS" --ignore-not-found >/dev/null 2>&1 || true
|
||||||
kubectl apply -f - >/dev/null <<EOF
|
K apply -f - >/dev/null <<EOF
|
||||||
apiVersion: batch/v1
|
apiVersion: batch/v1
|
||||||
kind: Job
|
kind: Job
|
||||||
metadata: { name: sandbox-seed, namespace: $SB_NS }
|
metadata: { name: sandbox-seed, namespace: $SB_NS }
|
||||||
@@ -122,14 +150,14 @@ spec:
|
|||||||
echo "llx tables=\$N company=\$(Q "select value from llx_const where name='MAIN_INFO_SOCIETE_NOM'") lang=\$(Q "select value from llx_const where name='MAIN_LANG_DEFAULT'") owner=\$(Q "select tableowner from pg_tables where tablename='llx_societe'")"
|
echo "llx tables=\$N company=\$(Q "select value from llx_const where name='MAIN_INFO_SOCIETE_NOM'") lang=\$(Q "select value from llx_const where name='MAIN_LANG_DEFAULT'") owner=\$(Q "select tableowner from pg_tables where tablename='llx_societe'")"
|
||||||
echo "DONE."
|
echo "DONE."
|
||||||
EOF
|
EOF
|
||||||
kubectl wait --for=condition=complete job/sandbox-seed -n "$SB_NS" --timeout=300s >/dev/null 2>&1 \
|
K wait --for=condition=complete job/sandbox-seed -n "$SB_NS" --timeout=300s >/dev/null 2>&1 \
|
||||||
|| die "seed Job did not complete — see: kubectl logs -n $SB_NS job/sandbox-seed"
|
|| die "seed Job did not complete — see: K logs -n $SB_NS job/sandbox-seed"
|
||||||
kubectl logs -n "$SB_NS" job/sandbox-seed | sed 's/^/ /'
|
K logs -n "$SB_NS" job/sandbox-seed | sed 's/^/ /'
|
||||||
kubectl delete job sandbox-seed -n "$SB_NS" --ignore-not-found >/dev/null 2>&1 || true
|
K delete job sandbox-seed -n "$SB_NS" --ignore-not-found >/dev/null 2>&1 || true
|
||||||
|
|
||||||
log "Restoring app (replicas=1) + re-arming ArgoCD self-heal"
|
log "Restoring app (replicas=1) + re-arming ArgoCD self-heal"
|
||||||
set_selfheal true
|
set_selfheal true
|
||||||
kubectl scale deploy erp-sandbox -n "$SB_NS" --replicas=1 >/dev/null
|
K scale deploy erp-sandbox -n "$SB_NS" --replicas=1 >/dev/null
|
||||||
cleanup_secret; trap - EXIT
|
cleanup_secret; trap - EXIT
|
||||||
log "Refresh complete. Run 'sync-documents' to also copy the company logo + uploads."
|
log "Refresh complete. Run 'sync-documents' to also copy the company logo + uploads."
|
||||||
}
|
}
|
||||||
@@ -140,14 +168,14 @@ sync_documents() {
|
|||||||
[ -n "$pp" ] || die "no prod erp pod found"
|
[ -n "$pp" ] || die "no prod erp pod found"
|
||||||
[ -n "$sp" ] || die "no erp-sandbox pod found"
|
[ -n "$sp" ] || die "no erp-sandbox pod found"
|
||||||
log "Syncing $DOC_ROOT/mycompany (logo + uploads) ${pp##*/} -> ${sp##*/} via tar pipe"
|
log "Syncing $DOC_ROOT/mycompany (logo + uploads) ${pp##*/} -> ${sp##*/} via tar pipe"
|
||||||
kubectl exec -n "$PROD_NS" "${pp#pod/}" -- tar -C "$DOC_ROOT" -cf - mycompany 2>/dev/null \
|
K exec -n "$PROD_NS" "${pp#pod/}" -- tar -C "$DOC_ROOT" -cf - mycompany 2>/dev/null \
|
||||||
| kubectl exec -i -n "$SB_NS" "${sp#pod/}" -- tar -C "$DOC_ROOT" -xf -
|
| K exec -i -n "$SB_NS" "${sp#pod/}" -- tar -C "$DOC_ROOT" -xf -
|
||||||
log "Documents synced. (For a one-shot logo only, scope the tar to mycompany/logos.)"
|
log "Documents synced. (For a one-shot logo only, scope the tar to mycompany/logos.)"
|
||||||
}
|
}
|
||||||
|
|
||||||
case "${1:-}" in
|
case "${1:-}" in
|
||||||
refresh-from-prod) refresh_from_prod ;;
|
refresh-from-prod) assert_arcodange_cluster; refresh_from_prod ;;
|
||||||
sync-documents) sync_documents ;;
|
sync-documents) assert_arcodange_cluster; sync_documents ;;
|
||||||
refresh) refresh_from_prod; sync_documents ;;
|
refresh) assert_arcodange_cluster; refresh_from_prod; sync_documents ;;
|
||||||
*) echo "usage: $0 {refresh-from-prod|sync-documents|refresh}" >&2; exit 2 ;;
|
*) echo "usage: $0 {refresh-from-prod|sync-documents|refresh}" >&2; exit 2 ;;
|
||||||
esac
|
esac
|
||||||
|
|||||||
Reference in New Issue
Block a user