2 Commits

Author SHA1 Message Date
Debian
58e42eb269 fix(cluster): node-lokale Tabellen aus Drift-Hash entfernen — v1.2.89
network_interfaces + ip_addresses standen in confighash hashSpec, aber in cluster_replication.go localOnlyTables (= nicht repliziert, node-spezifische IPs). Dadurch waren die config_hash-Werte zweier Nodes ZWANGSLÄUFIG dauerhaft verschieden → Drift-Banner, das kein Resync beheben konnte. Beide Tabellen aus dem Hash entfernt; Drift erkennt jetzt nur noch wirklich replizierte Service-Config. Muss auf BEIDEN Nodes installiert sein (gleicher hashSpec für vergleichbare Hashes).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-04 13:35:32 +02:00
Debian
90f0df4c45 fix(cluster): Repair-Rollenerkennung über pg_publication statt ha_nodes.role — v1.2.88
Bug: Dispatch an utm-2 schlug fehl ('dieser Node ist der Cluster-Primary'), weil ha_nodes je Node lokal ist und sich JEDE Node selbst als role=primary markiert. Fix: Primary-Erkennung über pg_publication (edgeguard_shared, für jeden DB-User lesbar) statt role/pg_role. Primary gibt dem Subscriber seine eigene Adresse als primary_host mit (PostPeerWithBody); Agent-Handler vertraut dem mTLS-Dispatch mit Safety-Guard 'läuft nie auf dem Publication-Primary'. Funktioniert auch bei Direktzugriff auf den Subscriber. UI-Gating vereinfacht (Drift + Peer + Admin).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-04 13:18:54 +02:00
4 changed files with 130 additions and 141 deletions

View File

@@ -1 +1 @@
1.2.87 1.2.89

View File

@@ -71,33 +71,16 @@ var hashSpec = []hashTable{
{Name: "ntp_pools", MigrationDefault: true}, {Name: "ntp_pools", MigrationDefault: true},
// network_interfaces + ip_addresses werden seit 0030 repliziert — // network_interfaces + ip_addresses sind BEWUSST NICHT im Drift-Hash.
// VLAN/Bridge/Bond-Definitionen und Gateway-IPs müssen auf dem Secondary // Sie stehen in cluster_replication.go localOnlyTables, werden also NICHT
// für Failover bereitstehen. Ethernet-IPs werden im Secondary-Renderer // repliziert und sind per Design node-spezifisch (jede Node hat eigene
// herausgefiltert (eth0 = cloud-init / Keepalived). // Mgmt-/Host-IPs, z.B. utm-1=.6, utm-2=.8). Würde man sie hashen, wäre
// der config_hash zwischen zwei Nodes ZWANGSLÄUFIG dauerhaft verschieden
// → Drift-Banner, das kein Resync je beheben kann (Resync kopiert nur
// replizierte Tabellen). Migration 0030 wollte sie zwar replizieren,
// localOnlyTables schließt sie aber weiter aus → wir hashen sie nicht.
// //
// ip_addresses.interface_id ist ein node-lokaler Autoincrement-PK, der // static_routes, dns_settings, ntp_settings bleiben ebenfalls node-spezifisch.
// zwischen zwei unabhängigen DBs divergiert (utm-1: eth0=6, utm-2: eth0=1).
// Wir hashen daher semantisch: address + prefix + flags + interface_name
// statt interface_id — sonst False-Positive-Drift auf logisch identischen Nodes.
{Name: "network_interfaces"},
{Name: "ip_addresses", CustomSQL: `
SELECT COALESCE(md5(string_agg(rh, '|' ORDER BY rh)), '')
FROM (
SELECT md5(jsonb_build_object(
'address', ia.address,
'prefix', ia.prefix,
'is_vip', ia.is_vip,
'active', ia.active,
'vip_priority', ia.vip_priority,
'description', ia.description,
'iface', ni.name
)::text) AS rh
FROM ip_addresses ia
JOIN network_interfaces ni ON ia.interface_id = ni.id
) sub`},
// static_routes, dns_settings, ntp_settings bleiben node-spezifisch.
} }
// hashSQL rendert die SHA-Input-SQL für eine Tabelle. // hashSQL rendert die SHA-Input-SQL für eine Tabelle.

View File

@@ -2,6 +2,7 @@ package handlers
import ( import (
"context" "context"
"encoding/json"
"errors" "errors"
"fmt" "fmt"
"log/slog" "log/slog"
@@ -26,132 +27,146 @@ import (
// Reparatur baut die Subscription neu auf und kopiert alle geteilten // Reparatur baut die Subscription neu auf und kopiert alle geteilten
// Tabellen frisch vom Primary (einseitig: Primary = Source of Truth). // Tabellen frisch vom Primary (einseitig: Primary = Source of Truth).
// //
// Der Resync MUSS auf dem Standby/Subscriber laufen (nur der hat eine // Rollen-Erkennung: NICHT über ha_nodes.role/pg_role — die sind je Node
// Subscription). Operatoren erreichen die UI aber über die VIP, die immer // lokal und unzuverlässig (jede Node markiert sich selbst, pg_role bleibt
// auf den Primary zeigt. Deshalb: // 'standalone' bis `promote`). Verlässlich ist die PUBLICATION: nur der
// Primary hat `edgeguard_shared` (pg_publication ist für jeden DB-User
// lesbar). Der Subscriber hat sie nicht → er ist das Resync-Ziel.
// //
// - Auf dem Primary geklickt → Dispatch via mTLS an den Standby // Ablauf:
// (POST /agent/cluster/repair-replication), der dort lokal läuft. // - Klick auf dem Primary → Dispatch via mTLS an den Peer
// - Auf dem Standby direkt geklickt → läuft lokal. // (POST /agent/cluster/repair-replication) mit der eigenen Adresse als
// primary_host; der Peer resynct von dort.
// - Klick direkt auf dem Subscriber → läuft lokal (Quelle = der Peer).
// //
// Die eigentliche Arbeit läuft — analog zum Rolling-Update — in einer // Die eigentliche Arbeit läuft — analog zum Rolling-Update — in einer
// transienten systemd-Unit, die das bereits getestete // transienten systemd-Unit, die `edgeguard-ctl cluster-setup-standby
// `edgeguard-ctl cluster-setup-standby <primary>` ausführt. // <primary>` ausführt.
const ( const (
repairUnitName = "edgeguard-repair-replication.service" repairUnitName = "edgeguard-repair-replication.service"
repairScriptPath = "/var/lib/edgeguard/repair-replication.sh" repairScriptPath = "/var/lib/edgeguard/repair-replication.sh"
repairAgentPath = "/agent/cluster/repair-replication" repairAgentPath = "/agent/cluster/repair-replication"
repairPubName = "edgeguard_shared" // muss zu cmd/edgeguard-ctl egPubName passen
) )
// validRepairHost erlaubt nur IPv4/IPv6/Hostnamen — der Wert landet in // validRepairHost erlaubt nur IPv4/IPv6/Hostnamen — der Wert landet in
// einem Bash-Script das als root läuft, also strikt validieren (defense // einem Bash-Script das als root läuft, also strikt validieren.
// in depth, auch wenn er aus ha_nodes stammt).
var validRepairHost = regexp.MustCompile(`^[A-Za-z0-9._:-]{1,253}$`) var validRepairHost = regexp.MustCompile(`^[A-Za-z0-9._:-]{1,253}$`)
// RepairReplication ist der UI-Endpoint. Läuft der lokale Node als // repairDispatchBody ist der Body des Agent-Dispatch: der Primary teilt
// Primary, wird der Resync an den Standby-Peer delegiert; auf dem Standby // dem Subscriber seine Adresse mit, von der resynct werden soll.
// selbst läuft er lokal. type repairDispatchBody struct {
PrimaryHost string `json:"primary_host"`
}
// RepairReplication ist der UI-Endpoint. Hat dieser Node die Publication
// (= Primary), wird der Resync an den Peer delegiert; sonst (Subscriber)
// läuft er lokal mit dem Peer als Quelle.
func (h *ClusterHandler) RepairReplication(c *gin.Context) { func (h *ClusterHandler) RepairReplication(c *gin.Context) {
if h.Store == nil { if h.Store == nil {
response.Internal(c, errors.New("cluster store unavailable")) response.Internal(c, errors.New("cluster store unavailable"))
return return
} }
all, err := h.Store.List(c.Request.Context()) ctx := c.Request.Context()
all, err := h.Store.List(ctx)
if err != nil { if err != nil {
response.Internal(c, err) response.Internal(c, err)
return return
} }
local := findNode(all, h.LocalID) local := findNode(all, h.LocalID)
peer := findOtherPeer(all, h.LocalID)
if peer == nil {
response.BadRequest(c, errors.New("kein Peer-Node im Cluster — nichts zu resyncen"))
return
}
// Primary → an den Subscriber-Peer (Nicht-Primary) delegieren. if h.nodeHasPublication(ctx) {
if isPrimaryNode(local) { // Primary → an den Subscriber-Peer delegieren, mit eigener Adresse.
standby := findSubscriberPeer(all, h.LocalID)
if h.Aggregator == nil { if h.Aggregator == nil {
response.BadRequest(c, errors.New("kein mTLS-Aggregator verfügbar — Resync nicht delegierbar")) response.BadRequest(c, errors.New("kein mTLS-Aggregator verfügbar — Resync nicht delegierbar"))
return return
} }
if standby == nil { primaryHost := pickPrimaryHost(local)
response.BadRequest(c, errors.New("kein Standby-/Subscriber-Node gefunden, an den der Resync delegiert werden könnte")) if primaryHost == "" || !validRepairHost.MatchString(primaryHost) {
response.BadRequest(c, errors.New("eigene Primary-Adresse (Mgmt/Internal/Public-IP/FQDN) fehlt oder ist ungültig"))
return return
} }
res := h.Aggregator.PostPeer(c.Request.Context(), *standby, repairAgentPath) body, _ := json.Marshal(repairDispatchBody{PrimaryHost: primaryHost})
res := h.Aggregator.PostPeerWithBody(ctx, *peer, repairAgentPath, body)
if !res.OK { if !res.OK {
response.Internal(c, fmt.Errorf("Resync auf %s anstoßen: %s", standby.FQDN, res.Err)) response.Internal(c, fmt.Errorf("Resync auf %s anstoßen: %s", peer.FQDN, res.Err))
return return
} }
slog.Info("cluster: replication repair delegated to standby", "standby", standby.FQDN) slog.Info("cluster: replication repair delegated", "target", peer.FQDN, "primary_host", primaryHost)
if h.Audit != nil { if h.Audit != nil {
_ = h.Audit.Log(c.Request.Context(), actorOf(c), "cluster.repair-replication", _ = h.Audit.Log(ctx, actorOf(c), "cluster.repair-replication",
standby.FQDN, gin.H{"target": "standby", "standby": standby.FQDN}, h.NodeID) peer.FQDN, gin.H{"target": "peer", "peer": peer.FQDN, "primary_host": primaryHost}, h.NodeID)
} }
response.Accepted(c, gin.H{"dispatched": true, "target": "standby", "standby_fqdn": standby.FQDN}) response.Accepted(c, gin.H{"dispatched": true, "target": "peer", "peer_fqdn": peer.FQDN})
return return
} }
// Standby (oder Direktzugriff) → lokal ausführen. // Subscriber → lokal ausführen, Quelle = der Peer (Primary).
host, err := h.runLocalRepair(c.Request.Context(), all) host := pickPrimaryHost(peer)
if err != nil { if err := h.startResync(ctx, host); err != nil {
response.BadRequest(c, err) response.BadRequest(c, err)
return return
} }
if h.Audit != nil { if h.Audit != nil {
_ = h.Audit.Log(c.Request.Context(), actorOf(c), "cluster.repair-replication", _ = h.Audit.Log(ctx, actorOf(c), "cluster.repair-replication",
host, gin.H{"target": "local", "primary": host}, h.NodeID) host, gin.H{"target": "local", "primary": host}, h.NodeID)
} }
response.Accepted(c, gin.H{"dispatched": true, "target": "local", "primary": host}) response.Accepted(c, gin.H{"dispatched": true, "target": "local", "primary": host})
} }
// AgentRepairReplication wird vom Primary via mTLS auf dem Standby // AgentRepairReplication wird vom Primary via mTLS auf dem Subscriber
// aufgerufen und startet dort den lokalen Resync. // aufgerufen und startet dort den lokalen Resync von primary_host.
func (h *ClusterHandler) AgentRepairReplication(c *gin.Context) { func (h *ClusterHandler) AgentRepairReplication(c *gin.Context) {
if h.Store == nil { if h.Store == nil {
response.Internal(c, errors.New("cluster store unavailable")) response.Internal(c, errors.New("cluster store unavailable"))
return return
} }
all, err := h.Store.List(c.Request.Context()) ctx := c.Request.Context()
if err != nil { var body repairDispatchBody
response.Internal(c, err) _ = c.ShouldBindJSON(&body) // best-effort; Fallback unten
return
host := strings.TrimSpace(body.PrimaryHost)
if host == "" {
// Fallback: Quelle aus ha_nodes (der andere Node).
if all, err := h.Store.List(ctx); err == nil {
host = pickPrimaryHost(findOtherPeer(all, h.LocalID))
} }
host, err := h.runLocalRepair(c.Request.Context(), all) }
if err != nil { if err := h.startResync(ctx, host); err != nil {
response.BadRequest(c, err) response.BadRequest(c, err)
return return
} }
slog.Info("cluster: replication repair triggered by peer", "primary", host, "node", h.LocalID) slog.Info("cluster: replication repair triggered by peer", "primary", host, "node", h.LocalID)
if h.Audit != nil { if h.Audit != nil {
_ = h.Audit.Log(c.Request.Context(), "cluster-peer", "cluster.repair-replication", _ = h.Audit.Log(ctx, "cluster-peer", "cluster.repair-replication",
host, gin.H{"target": "local", "primary": host, "via": "agent"}, h.NodeID) host, gin.H{"target": "local", "primary": host, "via": "agent"}, h.NodeID)
} }
response.Accepted(c, gin.H{"dispatched": true, "primary": host}) response.Accepted(c, gin.H{"dispatched": true, "primary": host})
} }
// runLocalRepair startet den Resync auf DIESEM Node. Verweigert auf dem // startResync schreibt das Repair-Script und startet die transiente
// Primary (kein Subscriber). Gibt den ermittelten Primary-Host zurück. // systemd-Unit. Safety-Guard: läuft NIE auf dem Publication-Primary.
func (h *ClusterHandler) runLocalRepair(_ context.Context, all []models.HANode) (string, error) { func (h *ClusterHandler) startResync(ctx context.Context, primaryHost string) error {
local := findNode(all, h.LocalID) primaryHost = strings.TrimSpace(primaryHost)
primary := findPrimary(all) if primaryHost == "" {
return errors.New("keine Primary-Adresse für den Resync ermittelbar")
if isPrimaryNode(local) {
return "", errors.New("dieser Node ist der Cluster-Primary — Resync läuft nur auf einem Standby/Subscriber")
} }
if primary == nil { if !validRepairHost.MatchString(primaryHost) {
return "", errors.New("kein Cluster-Primary gefunden — Resync-Quelle unbekannt") return fmt.Errorf("ungültige Primary-Adresse: %q", primaryHost)
} }
if primary.ID == h.LocalID { // Niemals auf dem Primary (Publication-Quelle) resyncen — würde die
return "", errors.New("der lokale Node ist als Primary markiert — Resync nicht möglich") // eigene Config mit sich selbst überschreiben bzw. ist sinnlos.
} if h.nodeHasPublication(ctx) {
return errors.New("dieser Node ist der Publication-Primary — Resync läuft nur auf einem Subscriber")
host := pickPrimaryHost(primary)
if host == "" {
return "", errors.New("Primary hat keine erreichbare IP/FQDN in ha_nodes")
}
if !validRepairHost.MatchString(host) {
return "", fmt.Errorf("ungültige Primary-Adresse: %q", host)
} }
if st := repairUnitState(); st == "activating" || st == "active" { if st := repairUnitState(); st == "activating" || st == "active" {
return "", errors.New("Resync läuft bereits") return errors.New("Resync läuft bereits")
} }
script := fmt.Sprintf(`#!/bin/bash script := fmt.Sprintf(`#!/bin/bash
@@ -165,10 +180,10 @@ if [ "$rc" -ne 0 ]; then
fi fi
echo "[repair] abgeschlossen — config_hash wird beim nächsten Cluster-Status neu berechnet" echo "[repair] abgeschlossen — config_hash wird beim nächsten Cluster-Status neu berechnet"
rm -f %[2]s rm -f %[2]s
`, host, repairScriptPath) `, primaryHost, repairScriptPath)
if err := os.WriteFile(repairScriptPath, []byte(script), 0o755); err != nil { if err := os.WriteFile(repairScriptPath, []byte(script), 0o755); err != nil {
return "", fmt.Errorf("write repair script: %w", err) return fmt.Errorf("write repair script: %w", err)
} }
_ = exec.Command("sudo", "-n", "/usr/bin/systemctl", "reset-failed", repairUnitName).Run() _ = exec.Command("sudo", "-n", "/usr/bin/systemctl", "reset-failed", repairUnitName).Run()
cmd := exec.Command("sudo", "-n", "/usr/bin/systemd-run", cmd := exec.Command("sudo", "-n", "/usr/bin/systemd-run",
@@ -177,10 +192,28 @@ rm -f %[2]s
"--collect", "--collect",
"bash", repairScriptPath) "bash", repairScriptPath)
if err := cmd.Run(); err != nil { if err := cmd.Run(); err != nil {
return "", fmt.Errorf("systemd-run failed: %w", err) return fmt.Errorf("systemd-run failed: %w", err)
} }
slog.Info("cluster: replication repair dispatched (local)", "primary", host, "node", h.LocalID) slog.Info("cluster: replication repair dispatched (local)", "primary", primaryHost, "node", h.LocalID)
return host, nil return nil
}
// nodeHasPublication prüft, ob dieser Node die Replikations-Publication
// besitzt — das verlässliche Primary-Signal. pg_publication ist für jeden
// DB-User lesbar (anders als pg_subscription).
func (h *ClusterHandler) nodeHasPublication(ctx context.Context) bool {
if h.Store == nil || h.Store.Pool == nil {
return false
}
cctx, cancel := context.WithTimeout(ctx, 2*time.Second)
defer cancel()
var exists bool
if err := h.Store.Pool.QueryRow(cctx,
`SELECT EXISTS(SELECT 1 FROM pg_publication WHERE pubname = $1)`, repairPubName,
).Scan(&exists); err != nil {
return false
}
return exists
} }
// repairStatusResponse spiegelt den Zustand der transienten Repair-Unit. // repairStatusResponse spiegelt den Zustand der transienten Repair-Unit.
@@ -195,21 +228,20 @@ type repairStatusResponse struct {
} }
// RepairReplicationStatus liest den Job-Zustand. Auf dem Primary wird der // RepairReplicationStatus liest den Job-Zustand. Auf dem Primary wird der
// Status vom Standby-Peer geholt (dort läuft der Job); sonst lokal. // Status vom Subscriber-Peer geholt (dort läuft der Job); sonst lokal.
func (h *ClusterHandler) RepairReplicationStatus(c *gin.Context) { func (h *ClusterHandler) RepairReplicationStatus(c *gin.Context) {
if h.Store != nil { ctx := c.Request.Context()
if all, err := h.Store.List(c.Request.Context()); err == nil { if h.Store != nil && h.nodeHasPublication(ctx) && h.Aggregator != nil {
local := findNode(all, h.LocalID) if all, err := h.Store.List(ctx); err == nil {
standby := findSubscriberPeer(all, h.LocalID) if peer := findOtherPeer(all, h.LocalID); peer != nil {
if isPrimaryNode(local) && h.Aggregator != nil && standby != nil { results := h.Aggregator.FanOut(ctx,
results := h.Aggregator.FanOut(c.Request.Context(), []models.HANode{*peer}, repairAgentPath+"/status", h.LocalID)
[]models.HANode{*standby}, repairAgentPath+"/status", h.LocalID)
if len(results) == 1 && results[0].OK && len(results[0].Data) > 0 { if len(results) == 1 && results[0].OK && len(results[0].Data) > 0 {
c.Data(200, "application/json", wrapEnvelope(results[0].Data)) c.Data(200, "application/json", wrapEnvelope(results[0].Data))
return return
} }
// Peer nicht erreichbar → idle zurückgeben statt Fehler, // Peer nicht erreichbar → idle statt Fehler, damit das
// damit das UI-Polling nicht hart abbricht. // UI-Polling nicht hart abbricht.
response.OK(c, repairStatusResponse{Phase: "idle", Log: []string{}}) response.OK(c, repairStatusResponse{Phase: "idle", Log: []string{}})
return return
} }
@@ -295,7 +327,7 @@ func localRepairStatus() repairStatusResponse {
return out return out
} }
// findNode / findByPGRole: kleine Helfer über die ha_nodes-Liste. // findNode liefert die ha_nodes-Row mit der gegebenen ID.
func findNode(nodes []models.HANode, id string) *models.HANode { func findNode(nodes []models.HANode, id string) *models.HANode {
for i := range nodes { for i := range nodes {
if nodes[i].ID == id { if nodes[i].ID == id {
@@ -305,32 +337,13 @@ func findNode(nodes []models.HANode, id string) *models.HANode {
return nil return nil
} }
// isPrimaryNode: ein Node gilt als Primary (Publication-Quelle), wenn // findOtherPeer liefert den (einen) anderen Node im 2-Node-Cluster.
// role ODER pg_role "primary" ist. pg_role bleibt nach cluster-setup-
// standby auf "standalone" (nur `promote` setzt es), daher ist role das
// verlässliche Signal — analog zur keepalived-Logik.
func isPrimaryNode(n *models.HANode) bool {
return n != nil && (n.Role == "primary" || n.PGRole == "primary")
}
// findPrimary liefert den Primary-Node (Resync-Quelle).
func findPrimary(nodes []models.HANode) *models.HANode {
for i := range nodes {
if isPrimaryNode(&nodes[i]) {
return &nodes[i]
}
}
return nil
}
// findSubscriberPeer liefert den Resync-Ziel-Peer: ein anderer Node, der
// NICHT der Primary ist (in einem 2-Node-Cluster der Standby/Subscriber).
// Bevorzugt einen online erreichbaren Peer. // Bevorzugt einen online erreichbaren Peer.
func findSubscriberPeer(nodes []models.HANode, localID string) *models.HANode { func findOtherPeer(nodes []models.HANode, localID string) *models.HANode {
var fallback *models.HANode var fallback *models.HANode
for i := range nodes { for i := range nodes {
n := &nodes[i] n := &nodes[i]
if n.ID == localID || isPrimaryNode(n) { if n.ID == localID {
continue continue
} }
if n.Status == "online" { if n.Status == "online" {
@@ -343,10 +356,13 @@ func findSubscriberPeer(nodes []models.HANode, localID string) *models.HANode {
return fallback return fallback
} }
// pickPrimaryHost wählt die beste erreichbare Adresse des Primary: // pickPrimaryHost wählt die beste erreichbare Adresse eines Node:
// Mgmt-IP → Internal-IP → Public-IP → FQDN. Strippt eine etwaige // Mgmt-IP → Internal-IP → Public-IP → FQDN. Strippt eine etwaige
// CIDR-Maske (inet-Spalten können "10.0.0.5/32" liefern). // CIDR-Maske (inet-Spalten können "10.0.0.5/32" liefern).
func pickPrimaryHost(n *models.HANode) string { func pickPrimaryHost(n *models.HANode) string {
if n == nil {
return ""
}
for _, cand := range []*string{n.MgmtIP, n.InternalIP, n.PublicIP} { for _, cand := range []*string{n.MgmtIP, n.InternalIP, n.PublicIP} {
if cand != nil { if cand != nil {
if h := strings.TrimSpace(strings.SplitN(*cand, "/", 2)[0]); h != "" { if h := strings.TrimSpace(strings.SplitN(*cand, "/", 2)[0]); h != "" {

View File

@@ -392,17 +392,13 @@ export default function ClusterPage() {
const primaryFqdn = data?.local_node?.fqdn ?? window.location.hostname const primaryFqdn = data?.local_node?.fqdn ?? window.location.hostname
// Repair-Button: sichtbar bei Drift, für Admins, wenn ein Resync-Ziel // Repair-Button: sichtbar bei Drift, für Admins, sobald ein Peer
// existiert — auf dem Standby (lokal) oder auf dem Primary (delegiert // existiert. Welche Node Primary (Publication-Quelle) bzw. Subscriber
// an den Subscriber-Peer). Primary = role ODER pg_role 'primary' // ist, entscheidet das Backend zur Laufzeit über pg_publication — die
// (pg_role bleibt nach setup-standby 'standalone', role ist verlässlich). // UI muss das nicht raten (ha_nodes.role ist je Node lokal/unzuverlässig).
const isPrimaryNode = (n?: HANode | null) => !!n && (n.pg_role === 'primary' || n.role === 'primary')
const localIsPrimary = isPrimaryNode(data?.local_node)
const hasSubscriberPeer = data?.peers?.some(p => !isPrimaryNode(p)) ?? false
const hasPrimary = localIsPrimary || (data?.peers?.some(isPrimaryNode) ?? false)
const canRepair = !isViewer const canRepair = !isViewer
&& !!data?.drift_found && !!data?.drift_found
&& (localIsPrimary ? hasSubscriberPeer : hasPrimary) && ((data?.peers?.length ?? 0) > 0)
const peerColumns: ColumnsType<HANode> = [ const peerColumns: ColumnsType<HANode> = [
{ {
@@ -540,13 +536,7 @@ export default function ClusterPage() {
banner banner
className="mb-16" className="mb-16"
message={t('cluster.driftBanner')} message={t('cluster.driftBanner')}
description={ description={t('cluster.driftBannerDesc')}
<>
<Paragraph style={{ marginBottom: 8 }}>{t('cluster.driftBannerDesc')}</Paragraph>
{localIsPrimary && !hasSubscriberPeer
&& <Text type="secondary">{t('cluster.repair.noStandbyHint')}</Text>}
</>
}
action={ action={
canRepair ? ( canRepair ? (
<Popconfirm <Popconfirm