Enterprise-Backup-, Recovery-, Verification-, Security- und Monitoring-Plattform fuer Proxmox VE, Windows, Linux und Dateisysteme. Der Leitsatz, der fast jede Entscheidung erklaert: Ein Backup gilt erst als vertrauenswuerdig, wenn Integritaet geprueft und Wiederherstellbarkeit nachgewiesen wurde. Deshalb steigt ein Wiederherstellungspunkt erst nach einem tatsaechlich durchgefuehrten Restore-Test auf "recoverable", und Unbekanntes geht in keine Bewertung als "gut" ein. Umfang (Phasen 0-23): - Repository Engine: inhaltsadressierte Bloecke, atomares Commit-Protokoll, Katalogaufbau allein aus den Manifesten — ohne Datenbank - Backup Engine: inhaltsabhaengiges Chunking, Deduplizierung trotz Verschluesselung, zstd, AES-256-GCM, Streaming mit Gegendruck - Agenten fuer Windows und Linux mit Auftragsabholung (Pull-Modell) - Proxmox-Provider mit beiden Zugriffswegen auf die Sicherungsarchive - Scheduler, Recovery Engine mit Pruefpunkt, Verification, Unveraenderlichkeit - Weboberflaeche, Kennzahlen, Meldungen, Berichte, Security Center, Ransomware-Heuristik (meldet, handelt nie) - Disaster Recovery, Haertung, Leistungsmessung, Chaos Testing - Eingefrorene Vertraege fuer API, Migrationen, Backup-Format und Repository - Auslieferungspaket fuer linux/amd64, linux/arm64 und windows/amd64 Nicht enthalten und als solches gekennzeichnet: Kapazitaetsprognose, Backup Copy, Changed Block Tracking bei Proxmox, erweiterte Attribute und ACLs. Gebaut, aber nie auf echter Hardware gefahren: der Windows-Dienst, die systemd-Einheit und der verpflichtende Proxmox-Meilenstein — ob eine wiederhergestellte VM startet, ist ungeprueft. Einzelheiten in CHANGELOG.md und docs/release-candidate.md. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
569 lines
22 KiB
Go
569 lines
22 KiB
Go
package verification
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"io"
|
|
"log/slog"
|
|
"os"
|
|
"sync"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/google/uuid"
|
|
"github.com/jackc/pgx/v5/pgxpool"
|
|
)
|
|
|
|
// discardTestLogger liefert einen Logger, der nichts ausgibt.
|
|
func discardTestLogger() *slog.Logger {
|
|
return slog.New(slog.NewTextHandler(io.Discard, nil))
|
|
}
|
|
|
|
// connectTestDatabase oeffnet die Testdatenbank.
|
|
func connectTestDatabase(testInstance *testing.T) *pgxpool.Pool {
|
|
testInstance.Helper()
|
|
|
|
connectionString := os.Getenv("SYNCOVA_TEST_DATABASE_URL")
|
|
if connectionString == "" {
|
|
testInstance.Skip("SYNCOVA_TEST_DATABASE_URL ist nicht gesetzt; die Datenbanktests werden uebersprungen")
|
|
}
|
|
|
|
connectContext, cancelConnect := context.WithTimeout(context.Background(), 5*time.Second)
|
|
defer cancelConnect()
|
|
|
|
connectionPool, poolError := pgxpool.New(connectContext, connectionString)
|
|
if poolError != nil {
|
|
testInstance.Skipf("die Testdatenbank war nicht erreichbar: %v", poolError)
|
|
}
|
|
|
|
if pingError := connectionPool.Ping(connectContext); pingError != nil {
|
|
connectionPool.Close()
|
|
testInstance.Skipf("die Testdatenbank antwortete nicht: %v", pingError)
|
|
}
|
|
|
|
testInstance.Cleanup(connectionPool.Close)
|
|
|
|
return connectionPool
|
|
}
|
|
|
|
// seedBackupRow legt Repository, Auftrag, Lauf und Backup in der Datenbank an.
|
|
func seedBackupRow(testInstance *testing.T, connectionPool *pgxpool.Pool) uuid.UUID {
|
|
testInstance.Helper()
|
|
|
|
backgroundContext := context.Background()
|
|
uniqueSuffix := uuid.NewString()[:8]
|
|
|
|
var repositoryID uuid.UUID
|
|
if scanError := connectionPool.QueryRow(backgroundContext,
|
|
`INSERT INTO repositories (name, location) VALUES ($1, $2) RETURNING id`,
|
|
"ver-"+uniqueSuffix, "/tmp/ver-"+uniqueSuffix).Scan(&repositoryID); scanError != nil {
|
|
testInstance.Fatalf("das Repository liess sich nicht eintragen: %v", scanError)
|
|
}
|
|
|
|
var jobID uuid.UUID
|
|
if scanError := connectionPool.QueryRow(backgroundContext,
|
|
`INSERT INTO backup_jobs (name, schedule_type, repository_id) VALUES ($1,'manual',$2) RETURNING id`,
|
|
"ver-job-"+uniqueSuffix, repositoryID).Scan(&jobID); scanError != nil {
|
|
testInstance.Fatalf("der Auftrag liess sich nicht eintragen: %v", scanError)
|
|
}
|
|
|
|
var runID uuid.UUID
|
|
if scanError := connectionPool.QueryRow(backgroundContext,
|
|
`INSERT INTO backup_job_runs (job_id, status, completed_at, correlation_id)
|
|
VALUES ($1,'succeeded',now(),gen_random_uuid()) RETURNING id`,
|
|
jobID).Scan(&runID); scanError != nil {
|
|
testInstance.Fatalf("der Lauf liess sich nicht eintragen: %v", scanError)
|
|
}
|
|
|
|
var backupID uuid.UUID
|
|
if scanError := connectionPool.QueryRow(backgroundContext,
|
|
`INSERT INTO backups (job_run_id, repository_id, backup_id_in_repository, backup_type,
|
|
status, manifest_ref, encrypted_bytes, completed_at)
|
|
VALUES ($1,$2,$3,'full','complete','manifests/x.manifest.json',4096,now()) RETURNING id`,
|
|
runID, repositoryID, "ver-backup-"+uniqueSuffix).Scan(&backupID); scanError != nil {
|
|
testInstance.Fatalf("das Backup liess sich nicht eintragen: %v", scanError)
|
|
}
|
|
|
|
testInstance.Cleanup(func() {
|
|
cleanupContext := context.Background()
|
|
|
|
_, _ = connectionPool.Exec(cleanupContext, `DELETE FROM verification_jobs WHERE backup_id = $1`, backupID)
|
|
_, _ = connectionPool.Exec(cleanupContext, `DELETE FROM backups WHERE id = $1`, backupID)
|
|
_, _ = connectionPool.Exec(cleanupContext, `DELETE FROM backup_job_runs WHERE id = $1`, runID)
|
|
_, _ = connectionPool.Exec(cleanupContext, `DELETE FROM backup_jobs WHERE id = $1`, jobID)
|
|
_, _ = connectionPool.Exec(cleanupContext, `DELETE FROM repositories WHERE id = $1`, repositoryID)
|
|
})
|
|
|
|
return backupID
|
|
}
|
|
|
|
// readBackupClassification liest die Einstufung eines Backups.
|
|
func readBackupClassification(testInstance *testing.T, connectionPool *pgxpool.Pool, backupIdentifier uuid.UUID) (string, *time.Time, *time.Time) {
|
|
testInstance.Helper()
|
|
|
|
var (
|
|
classification *string
|
|
lastVerifiedAt *time.Time
|
|
lastRestoreTestAt *time.Time
|
|
)
|
|
|
|
if scanError := connectionPool.QueryRow(context.Background(),
|
|
`SELECT classification, last_verified_at, last_restore_test_at FROM backups WHERE id = $1`,
|
|
backupIdentifier).Scan(&classification, &lastVerifiedAt, &lastRestoreTestAt); scanError != nil {
|
|
testInstance.Fatalf("die Einstufung liess sich nicht lesen: %v", scanError)
|
|
}
|
|
|
|
if classification == nil {
|
|
return "", lastVerifiedAt, lastRestoreTestAt
|
|
}
|
|
|
|
return *classification, lastVerifiedAt, lastRestoreTestAt
|
|
}
|
|
|
|
// TestStoreRejectsSecondActiveVerification prueft die Sperre gegen Doppelpruefung.
|
|
func TestStoreRejectsSecondActiveVerification(testInstance *testing.T) {
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
store := NewStore(connectionPool)
|
|
backupIdentifier := seedBackupRow(testInstance, connectionPool)
|
|
|
|
if _, createError := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil); createError != nil {
|
|
testInstance.Fatalf("die erste Pruefung liess sich nicht anlegen: %v", createError)
|
|
}
|
|
|
|
_, secondError := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
if !errors.Is(secondError, ErrBackupBusy) {
|
|
testInstance.Fatalf("die zweite Pruefung haette abgelehnt werden muessen, Fehler war: %v", secondError)
|
|
}
|
|
}
|
|
|
|
// TestFinishJobRaisesClassificationToVerified prueft die Einstufung nach Blockpruefung.
|
|
func TestFinishJobRaisesClassificationToVerified(testInstance *testing.T) {
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
store := NewStore(connectionPool)
|
|
backupIdentifier := seedBackupRow(testInstance, connectionPool)
|
|
|
|
createdJob, createError := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
if createError != nil {
|
|
testInstance.Fatalf("die Pruefung liess sich nicht anlegen: %v", createError)
|
|
}
|
|
|
|
cleanReport := &Report{
|
|
BackupID: "ver-backup",
|
|
VerificationType: TypeChunkIntegrity,
|
|
ChunksChecked: 12,
|
|
}
|
|
|
|
if finishError := store.FinishJob(context.Background(), createdJob.ID, cleanReport, nil); finishError != nil {
|
|
testInstance.Fatalf("das Ergebnis liess sich nicht festschreiben: %v", finishError)
|
|
}
|
|
|
|
classification, lastVerifiedAt, lastRestoreTestAt := readBackupClassification(testInstance, connectionPool, backupIdentifier)
|
|
|
|
if classification != string(ClassificationVerified) {
|
|
testInstance.Errorf("Einstufung war %q, erwartet wurde verified", classification)
|
|
}
|
|
|
|
if lastVerifiedAt == nil {
|
|
testInstance.Error("der Pruefzeitpunkt wurde nicht gesetzt")
|
|
}
|
|
|
|
// Die staerkere Aussage darf eine Blockpruefung nicht mitliefern: Sie hat
|
|
// nichts wiederhergestellt.
|
|
if lastRestoreTestAt != nil {
|
|
testInstance.Error("eine Blockpruefung hat einen Wiederherstellungstest vorgetaeuscht")
|
|
}
|
|
}
|
|
|
|
// TestFinishJobWithCorruptionDowngradesBackup prueft die Herabstufung bei einem Befund.
|
|
func TestFinishJobWithCorruptionDowngradesBackup(testInstance *testing.T) {
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
store := NewStore(connectionPool)
|
|
backupIdentifier := seedBackupRow(testInstance, connectionPool)
|
|
|
|
createdJob, createError := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
if createError != nil {
|
|
testInstance.Fatalf("die Pruefung liess sich nicht anlegen: %v", createError)
|
|
}
|
|
|
|
corruptedReport := &Report{
|
|
BackupID: "ver-backup",
|
|
VerificationType: TypeChunkIntegrity,
|
|
ChunksChecked: 12,
|
|
ChunksCorrupted: 1,
|
|
Findings: []Finding{{
|
|
Code: "CHUNK_CORRUPTED",
|
|
Severity: SeverityCorruption,
|
|
Message: "Ein Block ist beschaedigt.",
|
|
}},
|
|
}
|
|
|
|
if finishError := store.FinishJob(context.Background(), createdJob.ID, corruptedReport, nil); finishError != nil {
|
|
testInstance.Fatalf("das Ergebnis liess sich nicht festschreiben: %v", finishError)
|
|
}
|
|
|
|
classification, lastVerifiedAt, _ := readBackupClassification(testInstance, connectionPool, backupIdentifier)
|
|
|
|
if classification != string(ClassificationCorrupted) {
|
|
testInstance.Errorf("Einstufung war %q, erwartet wurde corrupted", classification)
|
|
}
|
|
|
|
// Eine Pruefung, die Beschaedigungen fand, ist kein Beleg fuer
|
|
// Unversehrtheit. Setzte sie den Zeitpunkt, erschiene das Backup bei der
|
|
// naechsten Bewertung als „kuerzlich geprueft".
|
|
if lastVerifiedAt != nil {
|
|
testInstance.Error("eine Pruefung mit Befund hat den Pruefzeitpunkt gesetzt")
|
|
}
|
|
|
|
loadedJob, readError := store.GetJob(context.Background(), createdJob.ID)
|
|
if readError != nil {
|
|
testInstance.Fatalf("die Pruefung liess sich nicht lesen: %v", readError)
|
|
}
|
|
|
|
if loadedJob.Result != ResultCorrupted {
|
|
testInstance.Errorf("Ergebnis war %q, erwartet wurde corrupted", loadedJob.Result)
|
|
}
|
|
|
|
if loadedJob.Report == nil || len(loadedJob.Report.Findings) != 1 {
|
|
testInstance.Error("der Bericht wurde nicht vollstaendig abgelegt")
|
|
}
|
|
}
|
|
|
|
// TestRestoreTestDoesNotDowngradeFromRecoverable prueft die Rangfolge der Einstufungen.
|
|
func TestRestoreTestDoesNotDowngradeFromRecoverable(testInstance *testing.T) {
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
store := NewStore(connectionPool)
|
|
backupIdentifier := seedBackupRow(testInstance, connectionPool)
|
|
|
|
// Erst der Wiederherstellungstest: Er hebt auf „wiederherstellbar".
|
|
restoreJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeRestoreTest, nil)
|
|
|
|
restoreReport := &RestoreTestReport{
|
|
Report: Report{
|
|
BackupID: "ver-backup",
|
|
VerificationType: TypeRestoreTest,
|
|
ChunksChecked: 5,
|
|
DurationSeconds: 2.5,
|
|
},
|
|
FilesRestored: 3,
|
|
FilesCompared: 3,
|
|
}
|
|
|
|
if finishError := store.FinishJob(context.Background(), restoreJob.ID, &restoreReport.Report, restoreReport); finishError != nil {
|
|
testInstance.Fatalf("der Wiederherstellungstest liess sich nicht festschreiben: %v", finishError)
|
|
}
|
|
|
|
classification, _, lastRestoreTestAt := readBackupClassification(testInstance, connectionPool, backupIdentifier)
|
|
if classification != string(ClassificationRecoverable) {
|
|
testInstance.Fatalf("Einstufung war %q, erwartet wurde recoverable", classification)
|
|
}
|
|
|
|
if lastRestoreTestAt == nil {
|
|
testInstance.Fatal("der Testzeitpunkt wurde nicht gesetzt")
|
|
}
|
|
|
|
// Danach eine Blockpruefung. Sie ist die schwaechere Aussage und darf die
|
|
// staerkere nicht ueberschreiben.
|
|
chunkJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
|
|
if finishError := store.FinishJob(context.Background(), chunkJob.ID,
|
|
&Report{BackupID: "ver-backup", VerificationType: TypeChunkIntegrity, ChunksChecked: 5}, nil); finishError != nil {
|
|
testInstance.Fatalf("die Blockpruefung liess sich nicht festschreiben: %v", finishError)
|
|
}
|
|
|
|
classificationAfter, _, _ := readBackupClassification(testInstance, connectionPool, backupIdentifier)
|
|
if classificationAfter != string(ClassificationRecoverable) {
|
|
testInstance.Errorf("Einstufung fiel auf %q zurueck, erwartet wurde recoverable", classificationAfter)
|
|
}
|
|
}
|
|
|
|
// TestFailJobLeavesClassificationUntouched prueft, dass ein Fehlschlag kein Befund ist.
|
|
func TestFailJobLeavesClassificationUntouched(testInstance *testing.T) {
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
store := NewStore(connectionPool)
|
|
backupIdentifier := seedBackupRow(testInstance, connectionPool)
|
|
|
|
// Zuerst eine bestandene Pruefung, damit es etwas zu verlieren gibt.
|
|
firstJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
_ = store.FinishJob(context.Background(), firstJob.ID,
|
|
&Report{BackupID: "ver-backup", VerificationType: TypeChunkIntegrity, ChunksChecked: 5}, nil)
|
|
|
|
failingJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
|
|
if failError := store.FailJob(context.Background(), failingJob.ID,
|
|
"das repository war nicht erreichbar"); failError != nil {
|
|
testInstance.Fatalf("der Fehlschlag liess sich nicht vermerken: %v", failError)
|
|
}
|
|
|
|
classification, lastVerifiedAt, _ := readBackupClassification(testInstance, connectionPool, backupIdentifier)
|
|
|
|
// Dass die Pruefung nicht laufen konnte, sagt nichts ueber die Daten.
|
|
if classification != string(ClassificationVerified) {
|
|
testInstance.Errorf("Einstufung war %q, erwartet wurde verified", classification)
|
|
}
|
|
|
|
if lastVerifiedAt == nil {
|
|
testInstance.Error("der Pruefzeitpunkt wurde durch einen Fehlschlag geloescht")
|
|
}
|
|
}
|
|
|
|
// TestReclaimStaleJobsReleasesCrashedVerification prueft die Freigabe verwaister Pruefungen.
|
|
func TestReclaimStaleJobsReleasesCrashedVerification(testInstance *testing.T) {
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
store := NewStore(connectionPool)
|
|
backupIdentifier := seedBackupRow(testInstance, connectionPool)
|
|
|
|
createdJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
|
|
if _, execError := connectionPool.Exec(context.Background(),
|
|
`UPDATE verification_jobs SET status='running', started_at=now(),
|
|
heartbeat_at = now() - interval '1 hour' WHERE id = $1`, createdJob.ID); execError != nil {
|
|
testInstance.Fatalf("der Absturz liess sich nicht nachstellen: %v", execError)
|
|
}
|
|
|
|
reclaimedCount, reclaimError := store.ReclaimStaleJobs(context.Background(), time.Minute)
|
|
if reclaimError != nil {
|
|
testInstance.Fatalf("die Freigabe schlug fehl: %v", reclaimError)
|
|
}
|
|
|
|
if reclaimedCount < 1 {
|
|
testInstance.Fatal("die verwaiste Pruefung wurde nicht freigegeben")
|
|
}
|
|
|
|
// Nach der Freigabe muss eine neue Pruefung moeglich sein — sonst sperrte
|
|
// der Eindeutigkeitsindex das Backup dauerhaft.
|
|
if _, createError := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil); createError != nil {
|
|
testInstance.Fatalf("nach der Freigabe liess sich keine neue Pruefung anlegen: %v", createError)
|
|
}
|
|
}
|
|
|
|
// recordingRunner merkt sich die ausgefuehrten Pruefungen.
|
|
type recordingRunner struct {
|
|
// mutex schuetzt den Zustand.
|
|
mutex sync.Mutex
|
|
// executionCount zaehlt die Aufrufe.
|
|
executionCount int
|
|
// reportToReturn ist der gemeldete Bericht.
|
|
reportToReturn *Report
|
|
// errorToReturn ist der gemeldete Fehler.
|
|
errorToReturn error
|
|
}
|
|
|
|
// Run erfuellt die Runner-Schnittstelle.
|
|
func (runner *recordingRunner) Run(runContext context.Context, verificationJob Job) (*Report, *RestoreTestReport, error) {
|
|
runner.mutex.Lock()
|
|
defer runner.mutex.Unlock()
|
|
|
|
runner.executionCount++
|
|
|
|
if runner.errorToReturn != nil {
|
|
return nil, nil, runner.errorToReturn
|
|
}
|
|
|
|
returnedReport := *runner.reportToReturn
|
|
returnedReport.BackupID = verificationJob.BackupID.String()
|
|
returnedReport.VerificationType = verificationJob.VerificationType
|
|
|
|
return &returnedReport, nil, nil
|
|
}
|
|
|
|
// count liefert die Zahl der Aufrufe.
|
|
func (runner *recordingRunner) count() int {
|
|
runner.mutex.Lock()
|
|
defer runner.mutex.Unlock()
|
|
|
|
return runner.executionCount
|
|
}
|
|
|
|
// TestLoopExecutesQueuedVerification prueft den Weg von der Warteschlange zum Ergebnis.
|
|
func TestLoopExecutesQueuedVerification(testInstance *testing.T) {
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
store := NewStore(connectionPool)
|
|
backupIdentifier := seedBackupRow(testInstance, connectionPool)
|
|
|
|
createdJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
|
|
testRunner := &recordingRunner{reportToReturn: &Report{ChunksChecked: 7, BytesRead: 2048}}
|
|
|
|
verificationLoop, loopError := NewLoop(store, testRunner, LoopOptions{
|
|
InstanceName: "test-" + uuid.NewString()[:8],
|
|
TickInterval: 20 * time.Millisecond,
|
|
HeartbeatInterval: 50 * time.Millisecond,
|
|
StaleJobTimeout: time.Minute,
|
|
ShutdownGracePeriod: 2 * time.Second,
|
|
}, discardTestLogger())
|
|
if loopError != nil {
|
|
testInstance.Fatalf("die Schleife liess sich nicht bauen: %v", loopError)
|
|
}
|
|
|
|
loopContext, cancelLoop := context.WithCancel(context.Background())
|
|
loopFinished := make(chan struct{})
|
|
|
|
go func() {
|
|
defer close(loopFinished)
|
|
_ = verificationLoop.Run(loopContext)
|
|
}()
|
|
|
|
deadline := time.Now().Add(5 * time.Second)
|
|
for time.Now().Before(deadline) {
|
|
loadedJob, _ := store.GetJob(context.Background(), createdJob.ID)
|
|
if loadedJob != nil && loadedJob.Status.IsFinished() {
|
|
break
|
|
}
|
|
|
|
time.Sleep(20 * time.Millisecond)
|
|
}
|
|
|
|
cancelLoop()
|
|
<-loopFinished
|
|
|
|
if testRunner.count() != 1 {
|
|
testInstance.Fatalf("die Pruefung lief %d mal, erwartet wurde genau einmal", testRunner.count())
|
|
}
|
|
|
|
finishedJob, _ := store.GetJob(context.Background(), createdJob.ID)
|
|
|
|
if finishedJob.Status != JobStatusCompleted {
|
|
testInstance.Errorf("Zustand war %q, erwartet wurde completed", finishedJob.Status)
|
|
}
|
|
|
|
if finishedJob.Result != ResultClean {
|
|
testInstance.Errorf("Ergebnis war %q, erwartet wurde clean", finishedJob.Result)
|
|
}
|
|
|
|
if finishedJob.ChunksChecked != 7 {
|
|
testInstance.Errorf("geprueft wurden %d Bloecke, erwartet wurden 7", finishedJob.ChunksChecked)
|
|
}
|
|
|
|
classification, _, _ := readBackupClassification(testInstance, connectionPool, backupIdentifier)
|
|
if classification != string(ClassificationVerified) {
|
|
testInstance.Errorf("Einstufung war %q, erwartet wurde verified", classification)
|
|
}
|
|
}
|
|
|
|
// TestLoopMarksUnrunnableVerificationAsFailed prueft die Behandlung eines Fehlschlags.
|
|
//
|
|
// Der Unterschied ist wesentlich: Eine Pruefung, die nicht laufen konnte, ist
|
|
// kein Befund am Backup. Wuerde die Schleife sie als Beschaedigung werten, waere
|
|
// jedes kurz nicht erreichbare Repository ein Fehlalarm.
|
|
func TestLoopMarksUnrunnableVerificationAsFailed(testInstance *testing.T) {
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
store := NewStore(connectionPool)
|
|
backupIdentifier := seedBackupRow(testInstance, connectionPool)
|
|
|
|
createdJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
|
|
testRunner := &recordingRunner{errorToReturn: errors.New("das repository ist nicht erreichbar")}
|
|
|
|
verificationLoop, _ := NewLoop(store, testRunner, LoopOptions{
|
|
InstanceName: "test-" + uuid.NewString()[:8],
|
|
TickInterval: 20 * time.Millisecond,
|
|
HeartbeatInterval: 50 * time.Millisecond,
|
|
StaleJobTimeout: time.Minute,
|
|
ShutdownGracePeriod: 2 * time.Second,
|
|
}, discardTestLogger())
|
|
|
|
loopContext, cancelLoop := context.WithCancel(context.Background())
|
|
loopFinished := make(chan struct{})
|
|
|
|
go func() {
|
|
defer close(loopFinished)
|
|
_ = verificationLoop.Run(loopContext)
|
|
}()
|
|
|
|
deadline := time.Now().Add(5 * time.Second)
|
|
for time.Now().Before(deadline) {
|
|
loadedJob, _ := store.GetJob(context.Background(), createdJob.ID)
|
|
if loadedJob != nil && loadedJob.Status.IsFinished() {
|
|
break
|
|
}
|
|
|
|
time.Sleep(20 * time.Millisecond)
|
|
}
|
|
|
|
cancelLoop()
|
|
<-loopFinished
|
|
|
|
failedJob, _ := store.GetJob(context.Background(), createdJob.ID)
|
|
|
|
if failedJob.Status != JobStatusFailed {
|
|
testInstance.Errorf("Zustand war %q, erwartet wurde failed", failedJob.Status)
|
|
}
|
|
|
|
if failedJob.ErrorMessage == "" {
|
|
testInstance.Error("der Grund des Fehlschlags wurde nicht festgehalten")
|
|
}
|
|
|
|
classification, _, _ := readBackupClassification(testInstance, connectionPool, backupIdentifier)
|
|
if classification != "" {
|
|
testInstance.Errorf("das Backup wurde auf %q gestuft, obwohl die Pruefung gar nicht lief", classification)
|
|
}
|
|
}
|
|
|
|
// TestCorruptionFindingClearsPreviousEvidence prueft, dass ein Befund die Nachweise loescht.
|
|
//
|
|
// Der Fall stammt aus dem Nachweis der Phase: Nach einem Befund und dessen
|
|
// Behebung stand das Backup sofort wieder als „wiederherstellbar" da — auf
|
|
// Grundlage eines Tests, der vor dem Schaden lief.
|
|
func TestCorruptionFindingClearsPreviousEvidence(testInstance *testing.T) {
|
|
connectionPool := connectTestDatabase(testInstance)
|
|
store := NewStore(connectionPool)
|
|
backupIdentifier := seedBackupRow(testInstance, connectionPool)
|
|
|
|
// Ein bestandener Wiederherstellungstest hebt auf „wiederherstellbar".
|
|
restoreJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeRestoreTest, nil)
|
|
restoreReport := &RestoreTestReport{
|
|
Report: Report{BackupID: "ver-backup", VerificationType: TypeRestoreTest, DurationSeconds: 1.5},
|
|
}
|
|
|
|
if finishError := store.FinishJob(context.Background(), restoreJob.ID,
|
|
&restoreReport.Report, restoreReport); finishError != nil {
|
|
testInstance.Fatalf("der Wiederherstellungstest liess sich nicht festschreiben: %v", finishError)
|
|
}
|
|
|
|
// Danach ein Befund.
|
|
corruptedJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
corruptedReport := &Report{
|
|
BackupID: "ver-backup",
|
|
VerificationType: TypeChunkIntegrity,
|
|
ChunksChecked: 3,
|
|
ChunksCorrupted: 1,
|
|
Findings: []Finding{{Code: "CHUNK_CORRUPTED", Severity: SeverityCorruption}},
|
|
}
|
|
|
|
if finishError := store.FinishJob(context.Background(), corruptedJob.ID, corruptedReport, nil); finishError != nil {
|
|
testInstance.Fatalf("der Befund liess sich nicht festschreiben: %v", finishError)
|
|
}
|
|
|
|
classification, lastVerifiedAt, lastRestoreTestAt := readBackupClassification(testInstance, connectionPool, backupIdentifier)
|
|
|
|
if classification != string(ClassificationCorrupted) {
|
|
testInstance.Errorf("Einstufung war %q, erwartet wurde corrupted", classification)
|
|
}
|
|
|
|
if lastRestoreTestAt != nil {
|
|
testInstance.Error("der Wiederherstellungstest von vor dem Schaden gilt weiterhin")
|
|
}
|
|
|
|
if lastVerifiedAt != nil {
|
|
testInstance.Error("die Pruefung von vor dem Schaden gilt weiterhin")
|
|
}
|
|
|
|
// Nach der Behebung darf das Backup hoechstens „geprueft" sein — der
|
|
// Wiederherstellungstest muss wiederholt werden.
|
|
repairedJob, _ := store.CreateJob(context.Background(), backupIdentifier, TypeChunkIntegrity, nil)
|
|
if finishError := store.FinishJob(context.Background(), repairedJob.ID,
|
|
&Report{BackupID: "ver-backup", VerificationType: TypeChunkIntegrity, ChunksChecked: 3}, nil); finishError != nil {
|
|
testInstance.Fatalf("die Pruefung nach der Behebung liess sich nicht festschreiben: %v", finishError)
|
|
}
|
|
|
|
classificationAfter, _, restoreTestAfter := readBackupClassification(testInstance, connectionPool, backupIdentifier)
|
|
|
|
if classificationAfter != string(ClassificationVerified) {
|
|
testInstance.Errorf("Einstufung war %q, erwartet wurde verified", classificationAfter)
|
|
}
|
|
|
|
if restoreTestAfter != nil {
|
|
testInstance.Error("nach der Behebung galt der alte Wiederherstellungstest wieder")
|
|
}
|
|
}
|