#15 waf/alerts.go: AlertWriter.Close() flusht gepufferte Alerts + stoppt die Goroutine (stop/done-Channels, sync.Once, atomic closed; Kanal wird NIE geschlossen → Send racet ohne Panic). Wiring in cmd/edgeguard-waf nach ListenAndServe (graceful shutdown). -race-Test alerts_test.go. #19 handlers/cluster_rollingupdate.go: (a) RollingUpdateStatus mutiert State nicht mehr beim GET — terminale Zustände altern in readRollingUpdateState nach 10 min aus (kein verlorenes 'done' bei parallelen Pollern). (b) State-File via sync.Mutex + configgen.AtomicWrite (kein partieller Read / Race zwischen Handler & Goroutine). (c) Version-Flip wird gegen die VORHER erfasste Secondary-Baseline geprüft statt gegen die Primary-Version (verhindert sofort-/nie-Flip). Bewusst belassen: geteilter upgrade.sh-Pfad ist deterministischer Inhalt + an exakte sudoers-Zeile gebunden → Überschreib-Race benign; MST-Timestamp-Parse locale (Server laufen C-Locale). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -3,6 +3,8 @@ package waf
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/jackc/pgx/v5/pgxpool"
|
||||
@@ -26,8 +28,12 @@ type Alert struct {
|
||||
// AlertWriter accepts Alert values via a buffered channel and writes
|
||||
// them to PostgreSQL asynchronously so SPOE handling stays low-latency.
|
||||
type AlertWriter struct {
|
||||
pool *pgxpool.Pool
|
||||
ch chan Alert
|
||||
pool *pgxpool.Pool
|
||||
ch chan Alert
|
||||
stop chan struct{}
|
||||
done chan struct{}
|
||||
closeOnce sync.Once
|
||||
closed atomic.Bool
|
||||
}
|
||||
|
||||
// NewAlertWriter creates an AlertWriter and starts its background goroutine.
|
||||
@@ -36,14 +42,19 @@ func NewAlertWriter(pool *pgxpool.Pool, bufSize int) *AlertWriter {
|
||||
aw := &AlertWriter{
|
||||
pool: pool,
|
||||
ch: make(chan Alert, bufSize),
|
||||
stop: make(chan struct{}),
|
||||
done: make(chan struct{}),
|
||||
}
|
||||
go aw.run()
|
||||
return aw
|
||||
}
|
||||
|
||||
// Send enqueues an alert. Drops silently if the channel is full to
|
||||
// avoid slowing down SPOE request handling.
|
||||
// Send enqueues an alert. Drops silently if the channel is full (or the
|
||||
// writer is closing) to avoid slowing down / panicking SPOE handling.
|
||||
func (aw *AlertWriter) Send(a Alert) {
|
||||
if aw.closed.Load() {
|
||||
return
|
||||
}
|
||||
select {
|
||||
case aw.ch <- a:
|
||||
default:
|
||||
@@ -51,9 +62,34 @@ func (aw *AlertWriter) Send(a Alert) {
|
||||
}
|
||||
}
|
||||
|
||||
// Close stops the writer and flushes buffered alerts (best-effort).
|
||||
// Safe to call multiple times. The channel is never closed → Send never
|
||||
// panics even if it races with Close.
|
||||
func (aw *AlertWriter) Close() {
|
||||
aw.closeOnce.Do(func() {
|
||||
aw.closed.Store(true)
|
||||
close(aw.stop)
|
||||
})
|
||||
<-aw.done
|
||||
}
|
||||
|
||||
func (aw *AlertWriter) run() {
|
||||
for a := range aw.ch {
|
||||
aw.write(a)
|
||||
defer close(aw.done)
|
||||
for {
|
||||
select {
|
||||
case a := <-aw.ch:
|
||||
aw.write(a)
|
||||
case <-aw.stop:
|
||||
// Restliche gepufferte Alerts noch wegschreiben, dann Ende.
|
||||
for {
|
||||
select {
|
||||
case a := <-aw.ch:
|
||||
aw.write(a)
|
||||
default:
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
61
internal/waf/alerts_test.go
Normal file
61
internal/waf/alerts_test.go
Normal file
@@ -0,0 +1,61 @@
|
||||
package waf
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.netcell-it.de/projekte/edgeguard-native/internal/database"
|
||||
)
|
||||
|
||||
// Beweist Fix #15: AlertWriter.Close() flusht, ist idempotent, und Send/Close
|
||||
// racen ohne Panic (Kanal wird nie geschlossen). Guarded per EG_FWTEST_DSN.
|
||||
func TestAlertWriter_CloseFlush(t *testing.T) {
|
||||
dsn := os.Getenv("EG_FWTEST_DSN")
|
||||
if dsn == "" {
|
||||
t.Skip("set EG_FWTEST_DSN to run the alert-writer test")
|
||||
}
|
||||
ctx := context.Background()
|
||||
var mErr error
|
||||
for i := 0; i < 3; i++ {
|
||||
if mErr = database.Migrate(ctx, dsn); mErr == nil {
|
||||
break
|
||||
}
|
||||
time.Sleep(700 * time.Millisecond)
|
||||
}
|
||||
if mErr != nil {
|
||||
t.Fatalf("migrate: %v", mErr)
|
||||
}
|
||||
pool, err := database.Open(ctx, dsn)
|
||||
if err != nil {
|
||||
t.Fatalf("open: %v", err)
|
||||
}
|
||||
defer pool.Close()
|
||||
|
||||
aw := NewAlertWriter(pool, 64)
|
||||
for i := 0; i < 20; i++ {
|
||||
aw.Send(Alert{Hostname: "t.local", ClientIP: "203.0.113.1", Method: "GET", URI: "/", Action: "detected"})
|
||||
}
|
||||
|
||||
// Send parallel zu Close → darf nicht paniken.
|
||||
var wg sync.WaitGroup
|
||||
for i := 0; i < 10; i++ {
|
||||
wg.Add(1)
|
||||
go func() { defer wg.Done(); aw.Send(Alert{Hostname: "t.local", Action: "detected"}) }()
|
||||
}
|
||||
|
||||
done := make(chan struct{})
|
||||
go func() { aw.Close(); close(done) }()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(10 * time.Second):
|
||||
t.Fatal("Close() did not return (flush hung)")
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
// Idempotent + Send nach Close ist No-op (kein Panic).
|
||||
aw.Close()
|
||||
aw.Send(Alert{Hostname: "after.local", Action: "detected"})
|
||||
}
|
||||
Reference in New Issue
Block a user