package handlers import ( "context" "encoding/json" "log/slog" "net/http" "os" "os/exec" "time" "github.com/gin-gonic/gin" "git.netcell-it.de/projekte/edgeguard-native/internal/handlers/response" "git.netcell-it.de/projekte/edgeguard-native/internal/models" ) const rollingUpdateStateFile = "/var/lib/edgeguard/rolling-update-state.json" const ( phaseIdle = "idle" phaseUpdatingSecondary = "updating-secondary" phaseWaitingSecondary = "waiting-secondary" phaseUpdatingPrimary = "updating-primary" phaseFailed = "failed" ) // RollingUpdateState hält den Fortschritt des Rolling-Updates. // Persistiert in rollingUpdateStateFile damit der Status über // einen kurzen API-Neustart hinaus lesbar bleibt. type RollingUpdateState struct { Phase string `json:"phase"` SecondaryID string `json:"secondary_id,omitempty"` SecondaryFQDN string `json:"secondary_fqdn,omitempty"` StartedAt time.Time `json:"started_at,omitempty"` UpdatedAt time.Time `json:"updated_at"` Error string `json:"error,omitempty"` } func readRollingUpdateState() RollingUpdateState { data, err := os.ReadFile(rollingUpdateStateFile) if err != nil { return RollingUpdateState{Phase: phaseIdle, UpdatedAt: time.Now()} } var s RollingUpdateState if err := json.Unmarshal(data, &s); err != nil { return RollingUpdateState{Phase: phaseIdle, UpdatedAt: time.Now()} } return s } func writeRollingUpdateState(s RollingUpdateState) { s.UpdatedAt = time.Now() data, err := json.Marshal(s) if err != nil { slog.Warn("rolling-update: failed to marshal state", "error", err) return } if err := os.WriteFile(rollingUpdateStateFile, data, 0o600); err != nil { slog.Warn("rolling-update: failed to write state file", "error", err) } } // RollingUpdate startet den Rolling-Update-Prozess: // 1. Secondary aktualisieren (via mTLS /agent/cluster/trigger-update) // 2. Warten bis Secondary neue Version meldet // 3. Primary (dieser Node) aktualisieren (wie /system/upgrade) // // Kein Cluster vorhanden → 409 zurück damit der Client auf /system/upgrade // ausweichen kann. Wenn bereits ein Rolling-Update läuft → aktuellen State. func (h *ClusterHandler) RollingUpdate(c *gin.Context) { if h.Aggregator == nil || h.Store == nil { c.JSON(http.StatusConflict, gin.H{"error": "no cluster — use /system/upgrade"}) return } st := readRollingUpdateState() if st.Phase != phaseIdle && st.Phase != phaseFailed { response.OK(c, st) return } nodes, err := h.Store.List(c.Request.Context()) if err != nil { response.Internal(c, err) return } var secondary *models.HANode for i := range nodes { if nodes[i].ID != h.LocalID { secondary = &nodes[i] break } } if secondary == nil { c.JSON(http.StatusConflict, gin.H{"error": "no peer node — use /system/upgrade"}) return } newState := RollingUpdateState{ Phase: phaseUpdatingSecondary, SecondaryID: secondary.ID, SecondaryFQDN: secondary.FQDN, StartedAt: time.Now(), } writeRollingUpdateState(newState) slog.Info("rolling-update: started", "secondary", secondary.FQDN) go h.runRollingUpdate(secondary) c.JSON(http.StatusAccepted, newState) } // RollingUpdateStatus gibt den aktuellen Rolling-Update-State zurück. // Wenn phase == "updating-primary" soll der Client auf /system/health // umschalten (der Primary restartet gleich → State kann nicht mehr // geschrieben werden). func (h *ClusterHandler) RollingUpdateStatus(c *gin.Context) { response.OK(c, readRollingUpdateState()) } func (h *ClusterHandler) runRollingUpdate(secondary *models.HANode) { ctx := context.Background() // 1. Secondary triggern slog.Info("rolling-update: posting trigger-update to secondary", "fqdn", secondary.FQDN) result := h.Aggregator.PostPeer(ctx, *secondary, "/agent/cluster/trigger-update") if !result.OK { writeRollingUpdateState(RollingUpdateState{ Phase: phaseFailed, SecondaryID: secondary.ID, SecondaryFQDN: secondary.FQDN, Error: "trigger-update failed: " + result.Err, }) slog.Warn("rolling-update: secondary trigger failed", "error", result.Err) return } // 2. Secondary-Version pollen — der Secondary restartet nach dem // Upgrade, danach zeigt /agent/cluster/version eine neue Version. writeRollingUpdateState(RollingUpdateState{ Phase: phaseWaitingSecondary, SecondaryID: secondary.ID, SecondaryFQDN: secondary.FQDN, }) slog.Info("rolling-update: waiting for secondary version flip") // Kurze Wartezeit damit apt auf dem Secondary erst losläuft time.Sleep(20 * time.Second) deadline := time.Now().Add(10 * time.Minute) versionFlipped := false for time.Now().Before(deadline) { results := h.Aggregator.FanOut(ctx, []models.HANode{*secondary}, "/agent/cluster/version", h.LocalID) if len(results) > 0 && results[0].OK { var ver struct { Version string `json:"version"` } if err := json.Unmarshal(results[0].Data, &ver); err == nil { slog.Info("rolling-update: secondary version", "version", ver.Version, "primary", h.Version) if ver.Version != h.Version { versionFlipped = true break } } } time.Sleep(10 * time.Second) } if !versionFlipped { writeRollingUpdateState(RollingUpdateState{ Phase: phaseFailed, SecondaryID: secondary.ID, SecondaryFQDN: secondary.FQDN, Error: "timeout (10 min) waiting for secondary version flip", }) slog.Warn("rolling-update: secondary version flip timeout") return } // 3. Primary (uns selbst) aktualisieren — identisch zu /system/upgrade writeRollingUpdateState(RollingUpdateState{ Phase: phaseUpdatingPrimary, SecondaryID: secondary.ID, SecondaryFQDN: secondary.FQDN, }) slog.Info("rolling-update: triggering primary self-upgrade") const scriptPath = "/var/lib/edgeguard/upgrade.sh" const script = `#!/bin/bash set -e sleep 2 export DEBIAN_FRONTEND=noninteractive dpkg --configure -a || true retry_apt() { local attempt=0 max=3 wait_for=15 while [ $attempt -lt $max ]; do attempt=$((attempt + 1)) apt-get update -qq || true if apt-get install -y -qq -o Dpkg::Options::=--force-confold \ edgeguard-api edgeguard-ui edgeguard; then return 0; fi [ $attempt -lt $max ] && sleep $wait_for && wait_for=$((wait_for * 2)) done return 1 } retry_apt echo "[upgrade] complete" rm -f /var/lib/edgeguard/upgrade.sh ` if err := os.WriteFile(scriptPath, []byte(script), 0o755); err != nil { writeRollingUpdateState(RollingUpdateState{ Phase: phaseFailed, SecondaryID: secondary.ID, SecondaryFQDN: secondary.FQDN, Error: "write upgrade script: " + err.Error(), }) return } const unitName = "edgeguard-upgrade.service" _ = exec.Command("sudo", "-n", "/usr/bin/systemctl", "reset-failed", unitName).Run() cmd := exec.Command("sudo", "-n", "/usr/bin/systemd-run", "--unit="+unitName, "--description=EdgeGuard rolling-update (primary)", "--collect", "bash", scriptPath) if err := cmd.Run(); err != nil { writeRollingUpdateState(RollingUpdateState{ Phase: phaseFailed, SecondaryID: secondary.ID, SecondaryFQDN: secondary.FQDN, Error: "systemd-run failed: " + err.Error(), }) slog.Warn("rolling-update: primary systemd-run failed", "error", err) return } // State bleibt "updating-primary" — der Primary restartet gleich. // UI erkennt Version-Flip via /system/health und schließt den Flow. slog.Info("rolling-update: primary upgrade dispatched, process will restart") }