Enterprise-Backup-, Recovery-, Verification-, Security- und Monitoring-Plattform fuer Proxmox VE, Windows, Linux und Dateisysteme. Der Leitsatz, der fast jede Entscheidung erklaert: Ein Backup gilt erst als vertrauenswuerdig, wenn Integritaet geprueft und Wiederherstellbarkeit nachgewiesen wurde. Deshalb steigt ein Wiederherstellungspunkt erst nach einem tatsaechlich durchgefuehrten Restore-Test auf "recoverable", und Unbekanntes geht in keine Bewertung als "gut" ein. Umfang (Phasen 0-23): - Repository Engine: inhaltsadressierte Bloecke, atomares Commit-Protokoll, Katalogaufbau allein aus den Manifesten — ohne Datenbank - Backup Engine: inhaltsabhaengiges Chunking, Deduplizierung trotz Verschluesselung, zstd, AES-256-GCM, Streaming mit Gegendruck - Agenten fuer Windows und Linux mit Auftragsabholung (Pull-Modell) - Proxmox-Provider mit beiden Zugriffswegen auf die Sicherungsarchive - Scheduler, Recovery Engine mit Pruefpunkt, Verification, Unveraenderlichkeit - Weboberflaeche, Kennzahlen, Meldungen, Berichte, Security Center, Ransomware-Heuristik (meldet, handelt nie) - Disaster Recovery, Haertung, Leistungsmessung, Chaos Testing - Eingefrorene Vertraege fuer API, Migrationen, Backup-Format und Repository - Auslieferungspaket fuer linux/amd64, linux/arm64 und windows/amd64 Nicht enthalten und als solches gekennzeichnet: Kapazitaetsprognose, Backup Copy, Changed Block Tracking bei Proxmox, erweiterte Attribute und ACLs. Gebaut, aber nie auf echter Hardware gefahren: der Windows-Dienst, die systemd-Einheit und der verpflichtende Proxmox-Meilenstein — ob eine wiederhergestellte VM startet, ist ungeprueft. Einzelheiten in CHANGELOG.md und docs/release-candidate.md. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
376 lines
14 KiB
Go
376 lines
14 KiB
Go
package alerting
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"os"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/google/uuid"
|
|
"github.com/jackc/pgx/v5/pgxpool"
|
|
)
|
|
|
|
// connectTestDatabase oeffnet die Testdatenbank.
|
|
func connectTestDatabase(testInstance *testing.T) *pgxpool.Pool {
|
|
testInstance.Helper()
|
|
|
|
connectionString := os.Getenv("SYNCOVA_TEST_DATABASE_URL")
|
|
if connectionString == "" {
|
|
testInstance.Skip("SYNCOVA_TEST_DATABASE_URL ist nicht gesetzt; die Datenbanktests werden uebersprungen")
|
|
}
|
|
|
|
connectContext, cancelConnect := context.WithTimeout(context.Background(), 5*time.Second)
|
|
defer cancelConnect()
|
|
|
|
connectionPool, poolError := pgxpool.New(connectContext, connectionString)
|
|
if poolError != nil {
|
|
testInstance.Skipf("die Testdatenbank war nicht erreichbar: %v", poolError)
|
|
}
|
|
|
|
if pingError := connectionPool.Ping(connectContext); pingError != nil {
|
|
connectionPool.Close()
|
|
testInstance.Skipf("die Testdatenbank antwortete nicht: %v", pingError)
|
|
}
|
|
|
|
testInstance.Cleanup(connectionPool.Close)
|
|
|
|
return connectionPool
|
|
}
|
|
|
|
// newTestStore erzeugt eine Datenzugriffsschicht mit eigenem Regelnamen.
|
|
//
|
|
// Der eigene Name je Test trennt die Meldungen voneinander: Sonst loeste die
|
|
// Selbstaufloesung eines Tests die Meldungen des naechsten auf.
|
|
func newTestStore(testInstance *testing.T) (*Store, string) {
|
|
testInstance.Helper()
|
|
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
testRuleName := "test_" + uuid.NewString()[:8]
|
|
|
|
testInstance.Cleanup(func() {
|
|
_, _ = connectionPool.Exec(context.Background(),
|
|
`DELETE FROM alerts WHERE rule_name = $1`, testRuleName)
|
|
})
|
|
|
|
return NewStore(connectionPool), testRuleName
|
|
}
|
|
|
|
// buildFinding baut einen Befund fuer einen Test.
|
|
func buildFinding(ruleName string, fingerprintSuffix string) Finding {
|
|
return Finding{
|
|
RuleName: ruleName,
|
|
Severity: SeverityHigh,
|
|
Fingerprint: ruleName + ":" + fingerprintSuffix,
|
|
Title: "Testbefund " + fingerprintSuffix,
|
|
Message: "Ein Zustand, der jemanden interessieren sollte.",
|
|
EntityType: "backup_job",
|
|
EntityName: "Testauftrag",
|
|
Details: map[string]any{"probe": fingerprintSuffix},
|
|
}
|
|
}
|
|
|
|
// TestSameCauseYieldsOneAlert prueft die Deduplizierung.
|
|
//
|
|
// Ein System, das jede Minute dieselbe Meldung erzeugt, wird nach drei Tagen
|
|
// weggeklickt — und dann fehlt die eine, auf die es ankam.
|
|
func TestSameCauseYieldsOneAlert(testInstance *testing.T) {
|
|
store, ruleName := newTestStore(testInstance)
|
|
finding := buildFinding(ruleName, "auftrag-1")
|
|
|
|
firstAlert, isFirstNew, firstError := store.RaiseFinding(context.Background(), finding)
|
|
if firstError != nil {
|
|
testInstance.Fatalf("die erste Meldung liess sich nicht anlegen: %v", firstError)
|
|
}
|
|
|
|
if !isFirstNew {
|
|
testInstance.Error("die erste Meldung wurde nicht als neu gemeldet")
|
|
}
|
|
|
|
// Derselbe Befund, dreimal hintereinander.
|
|
for repetition := 0; repetition < 3; repetition++ {
|
|
repeatedAlert, isNew, raiseError := store.RaiseFinding(context.Background(), finding)
|
|
if raiseError != nil {
|
|
testInstance.Fatalf("die Wiederholung schlug fehl: %v", raiseError)
|
|
}
|
|
|
|
if isNew {
|
|
testInstance.Fatal("eine Wiederholung wurde als neue Meldung gemeldet; " +
|
|
"sie wuerde ein zweites Mal zugestellt")
|
|
}
|
|
|
|
if repeatedAlert.ID != firstAlert.ID {
|
|
testInstance.Fatal("die Wiederholung erzeugte eine zweite Meldung")
|
|
}
|
|
}
|
|
|
|
finalAlert, _ := store.GetAlert(context.Background(), firstAlert.ID)
|
|
|
|
// Der Zaehler ist die eigentliche Auskunft: Einmal ist ein Zwischenfall,
|
|
// viermal ein Zustand.
|
|
if finalAlert.OccurrenceCount != 4 {
|
|
testInstance.Errorf("der Zaehler steht auf %d, erwartet wurden 4", finalAlert.OccurrenceCount)
|
|
}
|
|
}
|
|
|
|
// TestVanishedCauseResolvesItself prueft die automatische Aufloesung.
|
|
//
|
|
// **Der wichtigste Vorgang des Meldungswesens.** Ohne ihn bleibt jede Meldung
|
|
// stehen, bis jemand sie wegklickt; nach zwei Wochen steht dort eine Liste
|
|
// erledigter Probleme, und die eine aktuelle geht darin unter.
|
|
func TestVanishedCauseResolvesItself(testInstance *testing.T) {
|
|
store, ruleName := newTestStore(testInstance)
|
|
|
|
firstAlert, _, _ := store.RaiseFinding(context.Background(), buildFinding(ruleName, "auftrag-1"))
|
|
secondAlert, _, _ := store.RaiseFinding(context.Background(), buildFinding(ruleName, "auftrag-2"))
|
|
|
|
// Im naechsten Durchgang trifft nur noch der erste Befund zu.
|
|
resolvedCount, resolveError := store.ResolveVanishedFindings(context.Background(), ruleName,
|
|
[]string{firstAlert.Fingerprint})
|
|
if resolveError != nil {
|
|
testInstance.Fatalf("die Aufloesung schlug fehl: %v", resolveError)
|
|
}
|
|
|
|
if resolvedCount != 1 {
|
|
testInstance.Fatalf("%d Meldungen wurden aufgeloest, erwartet wurde genau eine", resolvedCount)
|
|
}
|
|
|
|
stillOpen, _ := store.GetAlert(context.Background(), firstAlert.ID)
|
|
if stillOpen.Status != StatusOpen {
|
|
testInstance.Error("die weiterhin zutreffende Meldung wurde aufgeloest")
|
|
}
|
|
|
|
resolvedAlert, _ := store.GetAlert(context.Background(), secondAlert.ID)
|
|
if resolvedAlert.Status != StatusResolved {
|
|
testInstance.Error("die verschwundene Ursache wurde nicht aufgeloest")
|
|
}
|
|
|
|
// Der Unterschied zur Aufloesung von Hand ist eine Auskunft ueber die
|
|
// Anlage: Was sich selbst erledigt, war ein Zwischenfall.
|
|
if !resolvedAlert.WasResolvedAutomatically() {
|
|
testInstance.Error("die Aufloesung wurde nicht als automatisch gekennzeichnet")
|
|
}
|
|
}
|
|
|
|
// TestEmptyResultResolvesEverything prueft den Fall ohne einen einzigen Befund.
|
|
func TestEmptyResultResolvesEverything(testInstance *testing.T) {
|
|
store, ruleName := newTestStore(testInstance)
|
|
|
|
raisedAlert, _, _ := store.RaiseFinding(context.Background(), buildFinding(ruleName, "auftrag-1"))
|
|
|
|
// Eine leere Liste ist der Normalfall, sobald alles in Ordnung ist. In SQL
|
|
// waere `x = ANY(NULL)` niemals wahr — es wuerde nichts aufgeloest, und die
|
|
// Meldung bliebe fuer immer stehen.
|
|
resolvedCount, resolveError := store.ResolveVanishedFindings(context.Background(), ruleName, nil)
|
|
if resolveError != nil {
|
|
testInstance.Fatalf("die Aufloesung schlug fehl: %v", resolveError)
|
|
}
|
|
|
|
if resolvedCount != 1 {
|
|
testInstance.Fatalf("%d Meldungen wurden aufgeloest, erwartet wurde eine", resolvedCount)
|
|
}
|
|
|
|
resolvedAlert, _ := store.GetAlert(context.Background(), raisedAlert.ID)
|
|
if resolvedAlert.Status != StatusResolved {
|
|
testInstance.Error("die Meldung blieb trotz leerer Befundliste offen")
|
|
}
|
|
}
|
|
|
|
// TestAcknowledgementKeepsAlertVisible prueft die Bedeutung der Kenntnisnahme.
|
|
//
|
|
// „Bestaetigen" heisst „ich weiss davon", nicht „weg damit". Waere es dasselbe,
|
|
// verschwaende der Zustand aus der Uebersicht, obwohl er weiterbesteht.
|
|
func TestAcknowledgementKeepsAlertVisible(testInstance *testing.T) {
|
|
store, ruleName := newTestStore(testInstance)
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
|
|
var actingUser uuid.UUID
|
|
if scanError := connectionPool.QueryRow(context.Background(),
|
|
`SELECT id FROM users LIMIT 1`).Scan(&actingUser); scanError != nil {
|
|
testInstance.Skipf("kein Benutzerkonto vorhanden: %v", scanError)
|
|
}
|
|
|
|
raisedAlert, _, _ := store.RaiseFinding(context.Background(), buildFinding(ruleName, "auftrag-1"))
|
|
|
|
acknowledgedAlert, acknowledgeError := store.AcknowledgeAlert(context.Background(),
|
|
raisedAlert.ID, actingUser, "Wird bearbeitet.")
|
|
if acknowledgeError != nil {
|
|
testInstance.Fatalf("die Bestaetigung schlug fehl: %v", acknowledgeError)
|
|
}
|
|
|
|
if acknowledgedAlert.Status != StatusAcknowledged {
|
|
testInstance.Fatalf("der Zustand ist %q, erwartet wurde acknowledged", acknowledgedAlert.Status)
|
|
}
|
|
|
|
// Die Meldung bleibt in der Liste unerledigter Vorgaenge.
|
|
activeAlerts, _, listError := store.ListAlerts(context.Background(), AlertFilter{OnlyActive: true})
|
|
if listError != nil {
|
|
testInstance.Fatalf("die Liste liess sich nicht lesen: %v", listError)
|
|
}
|
|
|
|
foundInActiveList := false
|
|
|
|
for _, activeAlert := range activeAlerts {
|
|
if activeAlert.ID == raisedAlert.ID {
|
|
foundInActiveList = true
|
|
}
|
|
}
|
|
|
|
if !foundInActiveList {
|
|
testInstance.Error("eine bestaetigte Meldung verschwand aus den unerledigten Vorgaengen")
|
|
}
|
|
|
|
// Und sie wird weiterhin aktualisiert, solange die Ursache besteht.
|
|
updatedAlert, isNew, _ := store.RaiseFinding(context.Background(), buildFinding(ruleName, "auftrag-1"))
|
|
if isNew {
|
|
testInstance.Error("die bestaetigte Meldung wurde als neue Meldung erzeugt")
|
|
}
|
|
|
|
if updatedAlert.Status != StatusAcknowledged {
|
|
testInstance.Error("die Bestaetigung ging bei der Aktualisierung verloren")
|
|
}
|
|
}
|
|
|
|
// TestResolvedCauseCanReturn prueft eine wiederkehrende Ursache.
|
|
//
|
|
// Ein Zustand, der gestern behoben war und heute wieder auftritt, ist ein neuer
|
|
// Vorgang — und muss erneut zugestellt werden.
|
|
func TestResolvedCauseCanReturn(testInstance *testing.T) {
|
|
store, ruleName := newTestStore(testInstance)
|
|
|
|
firstAlert, _, _ := store.RaiseFinding(context.Background(), buildFinding(ruleName, "auftrag-1"))
|
|
|
|
if _, resolveError := store.ResolveVanishedFindings(context.Background(), ruleName, nil); resolveError != nil {
|
|
testInstance.Fatalf("die Aufloesung schlug fehl: %v", resolveError)
|
|
}
|
|
|
|
returnedAlert, isNew, raiseError := store.RaiseFinding(context.Background(),
|
|
buildFinding(ruleName, "auftrag-1"))
|
|
if raiseError != nil {
|
|
testInstance.Fatalf("die wiederkehrende Ursache liess sich nicht melden: %v", raiseError)
|
|
}
|
|
|
|
if !isNew {
|
|
testInstance.Error("die wiederkehrende Ursache wurde nicht als neue Meldung erkannt")
|
|
}
|
|
|
|
if returnedAlert.ID == firstAlert.ID {
|
|
testInstance.Error("die aufgeloeste Meldung wurde wiederbelebt statt neu angelegt")
|
|
}
|
|
}
|
|
|
|
// TestSeverityCanRiseWhileOpen prueft die Verschaerfung einer offenen Meldung.
|
|
func TestSeverityCanRiseWhileOpen(testInstance *testing.T) {
|
|
store, ruleName := newTestStore(testInstance)
|
|
|
|
warningFinding := buildFinding(ruleName, "auftrag-1")
|
|
warningFinding.Severity = SeverityWarning
|
|
|
|
raisedAlert, _, _ := store.RaiseFinding(context.Background(), warningFinding)
|
|
|
|
if raisedAlert.Severity != SeverityWarning {
|
|
testInstance.Fatalf("der Schweregrad ist %q, erwartet wurde warning", raisedAlert.Severity)
|
|
}
|
|
|
|
// Aus einer Warnung kann ein kritischer Zustand werden, waehrend die Meldung
|
|
// offen ist. Sie muss dann steigen — sonst bliebe ein kritischer Zustand als
|
|
// Warnung stehen und erreichte den Bereitschaftskanal nicht.
|
|
criticalFinding := buildFinding(ruleName, "auftrag-1")
|
|
criticalFinding.Severity = SeverityCritical
|
|
|
|
escalatedAlert, _, _ := store.RaiseFinding(context.Background(), criticalFinding)
|
|
|
|
if escalatedAlert.Severity != SeverityCritical {
|
|
testInstance.Errorf("der Schweregrad blieb bei %q", escalatedAlert.Severity)
|
|
}
|
|
}
|
|
|
|
// TestManualResolutionRecordsActor prueft die Aufloesung von Hand.
|
|
func TestManualResolutionRecordsActor(testInstance *testing.T) {
|
|
store, ruleName := newTestStore(testInstance)
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
|
|
var actingUser uuid.UUID
|
|
if scanError := connectionPool.QueryRow(context.Background(),
|
|
`SELECT id FROM users LIMIT 1`).Scan(&actingUser); scanError != nil {
|
|
testInstance.Skipf("kein Benutzerkonto vorhanden: %v", scanError)
|
|
}
|
|
|
|
raisedAlert, _, _ := store.RaiseFinding(context.Background(), buildFinding(ruleName, "auftrag-1"))
|
|
|
|
resolvedAlert, resolveError := store.ResolveAlert(context.Background(), raisedAlert.ID,
|
|
actingUser, "Von Hand behoben.")
|
|
if resolveError != nil {
|
|
testInstance.Fatalf("die Aufloesung schlug fehl: %v", resolveError)
|
|
}
|
|
|
|
if resolvedAlert.WasResolvedAutomatically() {
|
|
testInstance.Error("eine von Hand geschlossene Meldung gilt als automatisch aufgeloest")
|
|
}
|
|
|
|
// Ein zweiter Versuch trifft eine bereits erledigte Meldung.
|
|
if _, secondError := store.ResolveAlert(context.Background(), raisedAlert.ID,
|
|
actingUser, "Nochmal"); !errors.Is(secondError, ErrAlertNotOpen) {
|
|
testInstance.Errorf("die zweite Aufloesung wurde angenommen, Fehler war: %v", secondError)
|
|
}
|
|
}
|
|
|
|
// TestSummaryCountsUnresolvedBySeverity prueft die Meldungslage.
|
|
func TestSummaryCountsUnresolvedBySeverity(testInstance *testing.T) {
|
|
store, ruleName := newTestStore(testInstance)
|
|
|
|
criticalFinding := buildFinding(ruleName, "kritisch")
|
|
criticalFinding.Severity = SeverityCritical
|
|
|
|
_, _, _ = store.RaiseFinding(context.Background(), criticalFinding)
|
|
_, _, _ = store.RaiseFinding(context.Background(), buildFinding(ruleName, "hoch"))
|
|
|
|
summary, summaryError := store.Summary(context.Background())
|
|
if summaryError != nil {
|
|
testInstance.Fatalf("die Meldungslage liess sich nicht erheben: %v", summaryError)
|
|
}
|
|
|
|
if summary.CriticalCount < 1 {
|
|
testInstance.Error("die kritische Meldung fehlt in der Zusammenfassung")
|
|
}
|
|
|
|
if summary.HighCount < 1 {
|
|
testInstance.Error("die ernste Meldung fehlt in der Zusammenfassung")
|
|
}
|
|
}
|
|
|
|
// TestAllRulesAreNamed prueft die Vollstaendigkeit des Regelwerks.
|
|
//
|
|
// Der Plan (§16) nennt elf Regeln. Fehlt eine, ist sie nicht „noch nicht
|
|
// gebaut", sondern vergessen — und niemand sieht es.
|
|
func TestAllRulesAreNamed(testInstance *testing.T) {
|
|
allRules := Rules()
|
|
|
|
if len(allRules) != 11 {
|
|
testInstance.Errorf("das Regelwerk nennt %d Regeln, der Plan elf", len(allRules))
|
|
}
|
|
|
|
for _, ruleDefinition := range allRules {
|
|
if ruleDefinition.Title == "" || ruleDefinition.Description == "" {
|
|
testInstance.Errorf("die Regel %s ist nicht beschrieben", ruleDefinition.Name)
|
|
}
|
|
|
|
if !ruleDefinition.Available && ruleDefinition.UnavailableReason == "" {
|
|
testInstance.Errorf("die Regel %s kann nicht ausloesen und erklaert sich nicht",
|
|
ruleDefinition.Name)
|
|
}
|
|
}
|
|
|
|
// Der Zertifikatsablauf ist die einzige Regel ohne Datengrundlage: Die
|
|
// Agenten weisen sich ueber Betriebstokens aus, nicht ueber Zertifikate.
|
|
certificateRule, wasFound := FindRule(RuleCertificateExpiry)
|
|
if !wasFound || certificateRule.Available {
|
|
testInstance.Error("die Zertifikatsregel wird als ausloesbar gefuehrt, obwohl " +
|
|
"agent_certificates von keiner Stelle beschrieben wird")
|
|
}
|
|
|
|
if AvailableRuleCount() != 10 {
|
|
testInstance.Errorf("%d Regeln gelten als ausloesbar, erwartet wurden 10", AvailableRuleCount())
|
|
}
|
|
}
|