Enterprise-Backup-, Recovery-, Verification-, Security- und Monitoring-Plattform fuer Proxmox VE, Windows, Linux und Dateisysteme. Der Leitsatz, der fast jede Entscheidung erklaert: Ein Backup gilt erst als vertrauenswuerdig, wenn Integritaet geprueft und Wiederherstellbarkeit nachgewiesen wurde. Deshalb steigt ein Wiederherstellungspunkt erst nach einem tatsaechlich durchgefuehrten Restore-Test auf "recoverable", und Unbekanntes geht in keine Bewertung als "gut" ein. Umfang (Phasen 0-23): - Repository Engine: inhaltsadressierte Bloecke, atomares Commit-Protokoll, Katalogaufbau allein aus den Manifesten — ohne Datenbank - Backup Engine: inhaltsabhaengiges Chunking, Deduplizierung trotz Verschluesselung, zstd, AES-256-GCM, Streaming mit Gegendruck - Agenten fuer Windows und Linux mit Auftragsabholung (Pull-Modell) - Proxmox-Provider mit beiden Zugriffswegen auf die Sicherungsarchive - Scheduler, Recovery Engine mit Pruefpunkt, Verification, Unveraenderlichkeit - Weboberflaeche, Kennzahlen, Meldungen, Berichte, Security Center, Ransomware-Heuristik (meldet, handelt nie) - Disaster Recovery, Haertung, Leistungsmessung, Chaos Testing - Eingefrorene Vertraege fuer API, Migrationen, Backup-Format und Repository - Auslieferungspaket fuer linux/amd64, linux/arm64 und windows/amd64 Nicht enthalten und als solches gekennzeichnet: Kapazitaetsprognose, Backup Copy, Changed Block Tracking bei Proxmox, erweiterte Attribute und ACLs. Gebaut, aber nie auf echter Hardware gefahren: der Windows-Dienst, die systemd-Einheit und der verpflichtende Proxmox-Meilenstein — ob eine wiederhergestellte VM startet, ist ungeprueft. Einzelheiten in CHANGELOG.md und docs/release-candidate.md. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
713 lines
25 KiB
Go
713 lines
25 KiB
Go
// Command syncova-bench misst Durchsatz und Ressourcenverbrauch
|
|
// (SYNCOVA_IMPLEMENTATION_PLAN.md §22).
|
|
//
|
|
// Das Werkzeug misst, statt zu schätzen — und es sagt bei jeder Zahl, unter
|
|
// welchen Bedingungen sie entstanden ist. Eine Durchsatzangabe ohne die Angabe
|
|
// von Maschine, Datenart und Einstellungen ist wertlos und wird trotzdem
|
|
// zitiert.
|
|
//
|
|
// Die Testdaten sind grundsätzlich **inkompressibel**. Das Projekt hat den
|
|
// gegenteiligen Fehler bereits gemacht: In Phase 4 zeigte ein Lauf eine
|
|
// 1021-fache Kompression, weil die Daten periodisch waren.
|
|
//
|
|
// Aufruf:
|
|
//
|
|
// syncova-bench run [--workdir <pfad>] [--scale klein|voll]
|
|
package main
|
|
|
|
import (
|
|
"context"
|
|
"crypto/rand"
|
|
"flag"
|
|
"fmt"
|
|
"io"
|
|
"log/slog"
|
|
"os"
|
|
"path/filepath"
|
|
"runtime"
|
|
"runtime/debug"
|
|
"sync"
|
|
"time"
|
|
|
|
"github.com/syncova/syncova/packages/agent"
|
|
"github.com/syncova/syncova/packages/backupengine"
|
|
"github.com/syncova/syncova/packages/benchmark"
|
|
"github.com/syncova/syncova/packages/platform/logging"
|
|
"github.com/syncova/syncova/packages/platform/ratelimit"
|
|
"github.com/syncova/syncova/packages/repository"
|
|
)
|
|
|
|
// serviceName benennt das Kommando in den Logs.
|
|
const serviceName = "syncova-bench"
|
|
|
|
// buildVersion wird beim Bauen über -ldflags gesetzt.
|
|
//
|
|
// Der Vorgabewert gilt nur für einen Bau von Hand; das Auslieferungspaket
|
|
// brennt die tatsächliche Fassung ein.
|
|
var buildVersion = "0.1.0-dev"
|
|
|
|
func main() {
|
|
if runError := run(); runError != nil {
|
|
fmt.Fprintf(os.Stderr, "%s: %v\n", serviceName, runError)
|
|
os.Exit(1)
|
|
}
|
|
}
|
|
|
|
// run wertet das Unterkommando aus.
|
|
func run() error {
|
|
// Die Versionsabfrage steht vor dem Laden der Konfiguration: Wer wissen
|
|
// will, welche Fassung auf einem Server liegt, hat in dem Moment womöglich
|
|
// keine Datenbank — etwa auf einem frisch ausgepackten Paket oder mitten in
|
|
// einer Störung.
|
|
if len(os.Args) > 1 && isVersionArgument(os.Args[1]) {
|
|
fmt.Printf("%s %s\n", serviceName, buildVersion)
|
|
|
|
return nil
|
|
}
|
|
|
|
benchFlags := flag.NewFlagSet("run", flag.ContinueOnError)
|
|
workDirectory := benchFlags.String("workdir", "", "Arbeitsverzeichnis für Quelle und Repository")
|
|
measurementScale := benchFlags.String("scale", "klein", "Umfang: klein oder voll")
|
|
|
|
arguments := os.Args[1:]
|
|
if len(arguments) > 0 && arguments[0] == "run" {
|
|
arguments = arguments[1:]
|
|
}
|
|
|
|
if parseError := benchFlags.Parse(arguments); parseError != nil {
|
|
return parseError
|
|
}
|
|
|
|
if *workDirectory == "" {
|
|
return fmt.Errorf("--workdir ist erforderlich")
|
|
}
|
|
|
|
if directoryError := os.MkdirAll(*workDirectory, 0o700); directoryError != nil {
|
|
return fmt.Errorf("das arbeitsverzeichnis liess sich nicht anlegen: %w", directoryError)
|
|
}
|
|
|
|
return runMeasurements(*workDirectory, *measurementScale)
|
|
}
|
|
|
|
// scaleProfile beschreibt den Umfang einer Messreihe.
|
|
//
|
|
// Zwei Stufen, weil eine Messung auf einer Entwicklungsmaschine anders
|
|
// aussehen muss als eine auf einer Appliance: „klein" läuft in Minuten und
|
|
// prüft, ob das Messwerk arbeitet; „voll" liefert Zahlen, die etwas bedeuten.
|
|
type scaleProfile struct {
|
|
// LargeSourceBytes ist die Größe der großen Einzelquelle.
|
|
LargeSourceBytes int64
|
|
// SmallFileCount ist die Zahl kleiner Dateien.
|
|
SmallFileCount int
|
|
// SmallFileBytes ist die Größe einer kleinen Datei.
|
|
SmallFileBytes int64
|
|
// ConcurrentJobs ist die Zahl gleichzeitiger Sicherungen.
|
|
ConcurrentJobs int
|
|
}
|
|
|
|
// resolveScale liefert das Profil einer Umfangsstufe.
|
|
func resolveScale(scaleName string) (scaleProfile, error) {
|
|
switch scaleName {
|
|
case "klein":
|
|
return scaleProfile{
|
|
LargeSourceBytes: 512 * 1024 * 1024,
|
|
SmallFileCount: 4000,
|
|
SmallFileBytes: 16 * 1024,
|
|
ConcurrentJobs: 4,
|
|
}, nil
|
|
case "voll":
|
|
return scaleProfile{
|
|
LargeSourceBytes: 2 * 1024 * 1024 * 1024,
|
|
SmallFileCount: 20000,
|
|
SmallFileBytes: 16 * 1024,
|
|
ConcurrentJobs: 8,
|
|
}, nil
|
|
default:
|
|
return scaleProfile{}, fmt.Errorf("unbekannter umfang %q (erlaubt: klein, voll)", scaleName)
|
|
}
|
|
}
|
|
|
|
// runMeasurements fährt die Messreihe.
|
|
func runMeasurements(workDirectory string, scaleName string) error {
|
|
profile, scaleError := resolveScale(scaleName)
|
|
if scaleError != nil {
|
|
return scaleError
|
|
}
|
|
|
|
quietLogger := logging.New(io.Discard, logging.Options{ServiceName: serviceName})
|
|
|
|
measurementReport := &benchmark.Report{
|
|
Title: fmt.Sprintf("Syncova Leistungsmessung (Umfang: %s)", scaleName),
|
|
Machine: benchmark.CurrentMachine(),
|
|
GeneratedAt: time.Now().UTC(),
|
|
}
|
|
|
|
if memoryLimit := debug.SetMemoryLimit(-1); memoryLimit < 1<<62 {
|
|
measurementReport.Machine.MemoryLimitBytes = memoryLimit
|
|
}
|
|
|
|
// Die Szenarien laufen nacheinander. Parallel gemessen beeinflussten sie
|
|
// sich gegenseitig — und eine Messung, die von einer anderen abhängt,
|
|
// misst nichts.
|
|
scenarioRunners := []struct {
|
|
name string
|
|
run func(string, scaleProfile, *slog.Logger) (*benchmark.Result, error)
|
|
}{
|
|
{"Eine große Quelle", measureLargeSource},
|
|
{"Viele kleine Dateien", measureManySmallFiles},
|
|
{"Zweiter Lauf über unveränderte Daten", measureDeduplicationRun},
|
|
{"Mehrere gleichzeitige Aufträge", measureConcurrentJobs},
|
|
{"Langsames Netz (1 MiB/s)", measureThrottledSource},
|
|
{"Wenig CPU (zwei Kerne)", measureConstrainedCPU},
|
|
{"Wenig Speicher (256 MiB)", measureConstrainedMemory},
|
|
}
|
|
|
|
for _, scenarioRunner := range scenarioRunners {
|
|
fmt.Fprintf(os.Stderr, "… %s\n", scenarioRunner.name)
|
|
|
|
scenarioResult, scenarioError := scenarioRunner.run(workDirectory, profile, quietLogger)
|
|
if scenarioError != nil {
|
|
return fmt.Errorf("das szenario %q schlug fehl: %w", scenarioRunner.name, scenarioError)
|
|
}
|
|
|
|
// Ein Lauf, der die Prüfung nicht besteht, kommt nicht in den Bericht —
|
|
// er kommt hinein mit dem Vermerk, warum seine Zahl nichts aussagt.
|
|
if validationError := scenarioResult.Validate(); validationError != nil {
|
|
scenarioResult.AddNote("Diese Messung ist nicht belastbar: %v", validationError)
|
|
}
|
|
|
|
measurementReport.Results = append(measurementReport.Results, scenarioResult)
|
|
}
|
|
|
|
// Was diese Maschine nicht messen kann, steht als nicht gemessen im
|
|
// Bericht — nicht als geschätzte Zahl.
|
|
measurementReport.AddSkipped("Eine große virtuelle Maschine",
|
|
"Ohne Proxmox-Verbund nicht messbar. Der Provider ist geschrieben, sein "+
|
|
"verpflichtender End-to-End-Nachweis aber nicht erbracht (Phase 7). Die große "+
|
|
"Einzelquelle oben misst denselben Datenpfad, nur ohne die Datenträgerabfrage "+
|
|
"des Hypervisors.")
|
|
measurementReport.AddSkipped("Langsames Repository",
|
|
"Ein künstlich verlangsamtes Ziel würde die Wartezeit messen, die man ihm "+
|
|
"vorgibt — eine Zahl, die man sich selbst ausgedacht hat. Aussagekräftig wäre "+
|
|
"eine Messung gegen echten Netzwerkspeicher; der steht hier nicht zur Verfügung.")
|
|
measurementReport.AddSkipped("Datenträger-Ein-/Ausgaben je Sekunde",
|
|
"Auf macOS ohne erweiterte Rechte nicht je Prozess auszulesen. Die Zahl der "+
|
|
"geschriebenen Blöcke steht ersatzweise in den Ergebnissen; sie ist die "+
|
|
"Größe, die diese Anlage beeinflussen kann.")
|
|
measurementReport.AddSkipped("Netzwerkdurchsatz",
|
|
"Alle Messungen liefen gegen ein lokales Repository. Eine Netzmessung ohne "+
|
|
"entfernte Gegenstelle wäre eine Messung des Rückschleifen-Geräts.")
|
|
|
|
measurementReport.WriteText(os.Stdout)
|
|
|
|
return nil
|
|
}
|
|
|
|
// measureLargeSource misst eine einzelne große Datei.
|
|
//
|
|
// Der Ersatz für „eine große VM": Derselbe Datenpfad — lesen, zerlegen, hashen,
|
|
// komprimieren, verschlüsseln, ablegen —, nur ohne die Datenträgerabfrage des
|
|
// Hypervisors.
|
|
func measureLargeSource(workDirectory string, profile scaleProfile, baseLogger *slog.Logger) (*benchmark.Result, error) {
|
|
sourcePath := filepath.Join(workDirectory, "grosse-quelle")
|
|
|
|
if prepareError := prepareIncompressibleFile(
|
|
filepath.Join(sourcePath, "abbild.bin"), profile.LargeSourceBytes); prepareError != nil {
|
|
return nil, prepareError
|
|
}
|
|
|
|
return runBackupScenario(scenarioOptions{
|
|
Name: "Eine große Quelle",
|
|
Description: "Eine einzelne große Datei — der Ersatz für eine virtuelle Maschine. " +
|
|
"Misst den reinen Datenpfad ohne Aufwand je Datei.",
|
|
WorkDirectory: workDirectory,
|
|
RepositoryName: "gross",
|
|
SourcePath: sourcePath,
|
|
DataShape: benchmark.DataShape{
|
|
Description: "inkompressible Zufallsdaten",
|
|
TotalBytes: profile.LargeSourceBytes,
|
|
FileCount: 1,
|
|
Compressible: false,
|
|
},
|
|
Logger: baseLogger,
|
|
})
|
|
}
|
|
|
|
// measureManySmallFiles misst viele kleine Dateien.
|
|
//
|
|
// Hier begrenzt nicht der Durchsatz, sondern der Aufwand je Datei: öffnen,
|
|
// lesen, Metadaten erfassen, schließen. Die aussagekräftige Zahl ist deshalb
|
|
// „Objekte je Sekunde", nicht „MiB/s".
|
|
func measureManySmallFiles(workDirectory string, profile scaleProfile, baseLogger *slog.Logger) (*benchmark.Result, error) {
|
|
sourcePath := filepath.Join(workDirectory, "viele-dateien")
|
|
|
|
if prepareError := prepareManyFiles(sourcePath,
|
|
profile.SmallFileCount, profile.SmallFileBytes); prepareError != nil {
|
|
return nil, prepareError
|
|
}
|
|
|
|
scenarioResult, scenarioError := runBackupScenario(scenarioOptions{
|
|
Name: "Viele kleine Dateien",
|
|
Description: "Viele kleine Dateien in flacher Struktur. Hier begrenzt der Aufwand " +
|
|
"je Datei, nicht der Durchsatz.",
|
|
WorkDirectory: workDirectory,
|
|
RepositoryName: "viele",
|
|
SourcePath: sourcePath,
|
|
DataShape: benchmark.DataShape{
|
|
Description: "inkompressible Zufallsdaten",
|
|
TotalBytes: int64(profile.SmallFileCount) * profile.SmallFileBytes,
|
|
FileCount: profile.SmallFileCount,
|
|
Compressible: false,
|
|
},
|
|
Logger: baseLogger,
|
|
})
|
|
if scenarioError != nil {
|
|
return nil, scenarioError
|
|
}
|
|
|
|
scenarioResult.AddNote("Die aussagekräftige Zahl ist hier „Objekte je Sekunde\". " +
|
|
"Der Durchsatz liegt bauartbedingt niedriger als bei einer großen Datei.")
|
|
|
|
return scenarioResult, nil
|
|
}
|
|
|
|
// measureDeduplicationRun misst einen zweiten Lauf über dieselben Daten.
|
|
//
|
|
// Er zeigt, was Deduplizierung tatsächlich einspart: Gelesen und gehasht wird
|
|
// weiterhin alles, abgelegt nichts. Der Gewinn ist Schreibarbeit, nicht Lesezeit
|
|
// — wer das verwechselt, hält Zusatzsicherungen für überflüssig.
|
|
func measureDeduplicationRun(workDirectory string, profile scaleProfile, baseLogger *slog.Logger) (*benchmark.Result, error) {
|
|
scenarioResult, scenarioError := runBackupScenario(scenarioOptions{
|
|
Name: "Zweiter Lauf über unveränderte Daten",
|
|
Description: "Dieselbe große Quelle ein zweites Mal in dasselbe Repository. " +
|
|
"Misst, was Deduplizierung einspart.",
|
|
WorkDirectory: workDirectory,
|
|
RepositoryName: "gross",
|
|
SourcePath: filepath.Join(workDirectory, "grosse-quelle"),
|
|
BackupSuffix: "-zweiter-lauf",
|
|
ReuseRepository: true,
|
|
DataShape: benchmark.DataShape{
|
|
Description: "inkompressible Zufallsdaten (bereits abgelegt)",
|
|
TotalBytes: profile.LargeSourceBytes,
|
|
FileCount: 1,
|
|
Compressible: false,
|
|
},
|
|
Logger: baseLogger,
|
|
})
|
|
if scenarioError != nil {
|
|
return nil, scenarioError
|
|
}
|
|
|
|
scenarioResult.AddNote("Gelesen und gehasht wird weiterhin alles — gespart wird das " +
|
|
"Ablegen. Der Gewinn einer Zusatzsicherung ist Zeit, nicht Speicher.")
|
|
|
|
return scenarioResult, nil
|
|
}
|
|
|
|
// measureConcurrentJobs misst mehrere gleichzeitige Sicherungen.
|
|
//
|
|
// Jede schreibt in ein **eigenes** Repository: Ein gemeinsames Ziel hielte nur
|
|
// eine Schreibsperre bereit, und gemessen würde das Warten darauf. Genau das
|
|
// ist die getrennte Frage nach der Repository-Konkurrenz weiter unten.
|
|
func measureConcurrentJobs(workDirectory string, profile scaleProfile, baseLogger *slog.Logger) (*benchmark.Result, error) {
|
|
sourcePath := filepath.Join(workDirectory, "gleichzeitig")
|
|
bytesPerJob := int64(64 * 1024 * 1024)
|
|
|
|
for jobIndex := 0; jobIndex < profile.ConcurrentJobs; jobIndex++ {
|
|
jobSource := filepath.Join(sourcePath, fmt.Sprintf("auftrag-%02d", jobIndex))
|
|
|
|
if prepareError := prepareIncompressibleFile(
|
|
filepath.Join(jobSource, "daten.bin"), bytesPerJob); prepareError != nil {
|
|
return nil, prepareError
|
|
}
|
|
}
|
|
|
|
totalBytes := bytesPerJob * int64(profile.ConcurrentJobs)
|
|
|
|
recorder := benchmark.StartRecording()
|
|
|
|
var waitGroup sync.WaitGroup
|
|
|
|
jobErrors := make([]error, profile.ConcurrentJobs)
|
|
processedBytes := make([]int64, profile.ConcurrentJobs)
|
|
|
|
for jobIndex := 0; jobIndex < profile.ConcurrentJobs; jobIndex++ {
|
|
waitGroup.Add(1)
|
|
|
|
go func(currentJob int) {
|
|
defer waitGroup.Done()
|
|
|
|
repositoryPath := filepath.Join(workDirectory, fmt.Sprintf("repo-gleichzeitig-%02d", currentJob))
|
|
|
|
if removeError := os.RemoveAll(repositoryPath); removeError != nil {
|
|
jobErrors[currentJob] = removeError
|
|
|
|
return
|
|
}
|
|
|
|
runResult, runError := performBackup(repositoryPath,
|
|
filepath.Join(sourcePath, fmt.Sprintf("auftrag-%02d", currentJob)),
|
|
fmt.Sprintf("gleichzeitig-%02d-%d", currentJob, time.Now().UnixNano()),
|
|
nil, baseLogger)
|
|
if runError != nil {
|
|
jobErrors[currentJob] = runError
|
|
|
|
return
|
|
}
|
|
|
|
processedBytes[currentJob] = runResult.Progress.BytesProcessed
|
|
}(jobIndex)
|
|
}
|
|
|
|
waitGroup.Wait()
|
|
|
|
measuredUsage := recorder.Stop()
|
|
|
|
for _, jobError := range jobErrors {
|
|
if jobError != nil {
|
|
return nil, jobError
|
|
}
|
|
}
|
|
|
|
var totalProcessed int64
|
|
for _, jobBytes := range processedBytes {
|
|
totalProcessed += jobBytes
|
|
}
|
|
|
|
scenarioResult := &benchmark.Result{
|
|
ScenarioName: "Mehrere gleichzeitige Aufträge",
|
|
Description: fmt.Sprintf("%d Sicherungen gleichzeitig, jede in ein eigenes Repository.",
|
|
profile.ConcurrentJobs),
|
|
Machine: benchmark.CurrentMachine(),
|
|
Data: benchmark.DataShape{
|
|
Description: "inkompressible Zufallsdaten",
|
|
TotalBytes: totalBytes,
|
|
FileCount: profile.ConcurrentJobs,
|
|
Compressible: false,
|
|
},
|
|
Usage: measuredUsage,
|
|
BytesProcessed: totalProcessed,
|
|
FilesProcessed: profile.ConcurrentJobs,
|
|
MeasuredAt: time.Now().UTC(),
|
|
}
|
|
|
|
scenarioResult.AddNote("Jeder Auftrag schreibt in ein eigenes Repository. Ein gemeinsames " +
|
|
"Ziel hält nur eine Schreibsperre bereit — gemessen würde dann das Warten darauf.")
|
|
|
|
return scenarioResult, nil
|
|
}
|
|
|
|
// measureThrottledSource misst mit gedrosselter Leserate.
|
|
//
|
|
// Der Ersatz für „langsames Netz": Der Bandbreitenbegrenzer aus Phase 8 sitzt
|
|
// am Lesen der Quelle, und wegen des Gegendrucks der Pipeline bindet dieser eine
|
|
// Punkt die gesamte Last.
|
|
func measureThrottledSource(workDirectory string, profile scaleProfile, baseLogger *slog.Logger) (*benchmark.Result, error) {
|
|
const throttledBytesPerSecond = 1024 * 1024
|
|
const throttledSourceBytes = 8 * 1024 * 1024
|
|
|
|
sourcePath := filepath.Join(workDirectory, "gedrosselt")
|
|
|
|
if prepareError := prepareIncompressibleFile(
|
|
filepath.Join(sourcePath, "daten.bin"), throttledSourceBytes); prepareError != nil {
|
|
return nil, prepareError
|
|
}
|
|
|
|
bandwidthLimiter, limiterError := ratelimit.NewLimiter(throttledBytesPerSecond)
|
|
if limiterError != nil {
|
|
return nil, limiterError
|
|
}
|
|
|
|
scenarioResult, scenarioError := runBackupScenario(scenarioOptions{
|
|
Name: "Langsames Netz (1 MiB/s)",
|
|
Description: "Dieselbe Pipeline mit gedrosselter Leserate. Zeigt, dass der " +
|
|
"Begrenzer wirkt und wo die Zeit dann hingeht.",
|
|
WorkDirectory: workDirectory,
|
|
RepositoryName: "gedrosselt",
|
|
SourcePath: sourcePath,
|
|
BandwidthLimiter: bandwidthLimiter,
|
|
DataShape: benchmark.DataShape{
|
|
Description: "inkompressible Zufallsdaten",
|
|
TotalBytes: throttledSourceBytes,
|
|
FileCount: 1,
|
|
Compressible: false,
|
|
},
|
|
Logger: baseLogger,
|
|
})
|
|
if scenarioError != nil {
|
|
return nil, scenarioError
|
|
}
|
|
|
|
scenarioResult.AddNote("Erwartet werden rund 8 s für 8 MiB. Eine deutlich kürzere " +
|
|
"Laufzeit hieße, dass der Begrenzer nicht greift.")
|
|
|
|
return scenarioResult, nil
|
|
}
|
|
|
|
// measureConstrainedCPU misst mit begrenzter Kernzahl.
|
|
//
|
|
// GOMAXPROCS begrenzt die gleichzeitig laufenden Abläufe der Go-Laufzeit. Das
|
|
// ist keine Simulation, sondern die tatsächliche Beschränkung — dieselbe, die
|
|
// eine kleine virtuelle Maschine mitbringt.
|
|
func measureConstrainedCPU(workDirectory string, profile scaleProfile, baseLogger *slog.Logger) (*benchmark.Result, error) {
|
|
const constrainedCores = 2
|
|
|
|
previousMaxProcs := runtime.GOMAXPROCS(constrainedCores)
|
|
defer runtime.GOMAXPROCS(previousMaxProcs)
|
|
|
|
sourcePath := filepath.Join(workDirectory, "wenig-cpu")
|
|
constrainedBytes := profile.LargeSourceBytes / 4
|
|
|
|
if prepareError := prepareIncompressibleFile(
|
|
filepath.Join(sourcePath, "daten.bin"), constrainedBytes); prepareError != nil {
|
|
return nil, prepareError
|
|
}
|
|
|
|
scenarioResult, scenarioError := runBackupScenario(scenarioOptions{
|
|
Name: "Wenig CPU (zwei Kerne)",
|
|
Description: "Dieselbe Pipeline mit auf zwei Kerne begrenzter Laufzeit. " +
|
|
"Keine Simulation — die Beschränkung ist echt.",
|
|
WorkDirectory: workDirectory,
|
|
RepositoryName: "wenig-cpu",
|
|
SourcePath: sourcePath,
|
|
DataShape: benchmark.DataShape{
|
|
Description: "inkompressible Zufallsdaten",
|
|
TotalBytes: constrainedBytes,
|
|
FileCount: 1,
|
|
Compressible: false,
|
|
},
|
|
Logger: baseLogger,
|
|
})
|
|
if scenarioError != nil {
|
|
return nil, scenarioError
|
|
}
|
|
|
|
scenarioResult.Machine.GoMaxProcs = constrainedCores
|
|
|
|
return scenarioResult, nil
|
|
}
|
|
|
|
// measureConstrainedMemory misst mit enger Speichergrenze.
|
|
//
|
|
// GOMEMLIMIT zwingt die Speicherbereinigung zu häufigerer Arbeit. Die
|
|
// interessante Frage ist nicht, ob es langsamer wird — das wird es —, sondern
|
|
// ob die Streaming-Pipeline überhaupt durchläuft: Ein Verfahren, das die Quelle
|
|
// in den Speicher lädt, scheiterte hier.
|
|
func measureConstrainedMemory(workDirectory string, profile scaleProfile, baseLogger *slog.Logger) (*benchmark.Result, error) {
|
|
const constrainedMemoryBytes = 256 * 1024 * 1024
|
|
|
|
previousLimit := debug.SetMemoryLimit(constrainedMemoryBytes)
|
|
defer debug.SetMemoryLimit(previousLimit)
|
|
|
|
sourcePath := filepath.Join(workDirectory, "wenig-speicher")
|
|
constrainedBytes := profile.LargeSourceBytes / 2
|
|
|
|
if prepareError := prepareIncompressibleFile(
|
|
filepath.Join(sourcePath, "daten.bin"), constrainedBytes); prepareError != nil {
|
|
return nil, prepareError
|
|
}
|
|
|
|
scenarioResult, scenarioError := runBackupScenario(scenarioOptions{
|
|
Name: "Wenig Speicher (256 MiB)",
|
|
Description: "Dieselbe Pipeline mit einer Speichergrenze weit unter der " +
|
|
"Quellgröße. Prüft, ob wirklich als Datenstrom gearbeitet wird.",
|
|
WorkDirectory: workDirectory,
|
|
RepositoryName: "wenig-speicher",
|
|
SourcePath: sourcePath,
|
|
DataShape: benchmark.DataShape{
|
|
Description: "inkompressible Zufallsdaten",
|
|
TotalBytes: constrainedBytes,
|
|
FileCount: 1,
|
|
Compressible: false,
|
|
},
|
|
Logger: baseLogger,
|
|
})
|
|
if scenarioError != nil {
|
|
return nil, scenarioError
|
|
}
|
|
|
|
scenarioResult.Machine.MemoryLimitBytes = constrainedMemoryBytes
|
|
scenarioResult.AddNote("Die Quelle ist %s groß, die Speichergrenze %s. "+
|
|
"Ein Verfahren, das die Quelle in den Speicher lädt, käme hier nicht durch.",
|
|
benchmark.FormatBytes(float64(constrainedBytes)),
|
|
benchmark.FormatBytes(constrainedMemoryBytes))
|
|
|
|
return scenarioResult, nil
|
|
}
|
|
|
|
// scenarioOptions beschreiben einen einzelnen Messlauf.
|
|
type scenarioOptions struct {
|
|
// Name benennt das Szenario.
|
|
Name string
|
|
// Description erklärt, welche Frage es beantwortet.
|
|
Description string
|
|
// WorkDirectory ist das Arbeitsverzeichnis.
|
|
WorkDirectory string
|
|
// RepositoryName ist der Name des Zielrepositorys.
|
|
RepositoryName string
|
|
// SourcePath ist die zu sichernde Quelle.
|
|
SourcePath string
|
|
// BackupSuffix unterscheidet mehrere Läufe im selben Repository.
|
|
BackupSuffix string
|
|
// ReuseRepository misst gegen ein bereits gefülltes Repository.
|
|
//
|
|
// Ausschliesslich fuer das Deduplizierungsszenario. Alle uebrigen Laeufe
|
|
// bekommen ein frisches Ziel — sonst misst ein zweiter Aufruf des
|
|
// Werkzeugs die Deduplizierung statt das Ablegen, und zwar unbemerkt:
|
|
// Beide Laeufe liefern plausible Zahlen, nur beantworten sie verschiedene
|
|
// Fragen. Im ersten Messlauf dieser Phase genau so passiert.
|
|
ReuseRepository bool
|
|
// BandwidthLimiter begrenzt die Leserate.
|
|
BandwidthLimiter *ratelimit.Limiter
|
|
// DataShape beschreibt die Daten.
|
|
DataShape benchmark.DataShape
|
|
// Logger nimmt die Protokollzeilen auf.
|
|
Logger *slog.Logger
|
|
}
|
|
|
|
// runBackupScenario führt einen gemessenen Sicherungslauf aus.
|
|
func runBackupScenario(options scenarioOptions) (*benchmark.Result, error) {
|
|
repositoryPath := filepath.Join(options.WorkDirectory, "repo-"+options.RepositoryName)
|
|
|
|
// Ein frisches Repository, sofern nicht ausdruecklich anders gewuenscht.
|
|
if !options.ReuseRepository {
|
|
if removeError := os.RemoveAll(repositoryPath); removeError != nil {
|
|
return nil, fmt.Errorf("das repository liess sich nicht zuruecksetzen: %w", removeError)
|
|
}
|
|
}
|
|
|
|
// Die Kennung traegt den Zeitpunkt: Ein Repository weist eine bereits
|
|
// vergebene Kennung zurueck, und ein Messwerkzeug, das sich nicht
|
|
// wiederholen laesst, ist keines — die zweite Messung ist die, die zaehlt.
|
|
backupIdentifier := fmt.Sprintf("bench-%s%s-%d",
|
|
options.RepositoryName, options.BackupSuffix, time.Now().UnixNano())
|
|
|
|
recorder := benchmark.StartRecording()
|
|
|
|
runResult, runError := performBackup(repositoryPath, options.SourcePath,
|
|
backupIdentifier, options.BandwidthLimiter, options.Logger)
|
|
|
|
measuredUsage := recorder.Stop()
|
|
|
|
if runError != nil {
|
|
return nil, runError
|
|
}
|
|
|
|
return &benchmark.Result{
|
|
ScenarioName: options.Name,
|
|
Description: options.Description,
|
|
Machine: benchmark.CurrentMachine(),
|
|
Data: options.DataShape,
|
|
Usage: measuredUsage,
|
|
BytesProcessed: runResult.Progress.BytesProcessed,
|
|
BytesStored: runResult.Progress.BytesWritten,
|
|
FilesProcessed: runResult.FilesBackedUp,
|
|
MeasuredAt: time.Now().UTC(),
|
|
}, nil
|
|
}
|
|
|
|
// performBackup führt eine Sicherung aus.
|
|
//
|
|
// Ohne Verschlüsselung: Ein Schlüssel verlangte einen Secret Store, und die
|
|
// Messung ginge dann durch AES-256-GCM — eine andere Frage als die nach dem
|
|
// Durchsatz der Pipeline. Der Aufschlag der Verschlüsselung gehört gesondert
|
|
// gemessen.
|
|
func performBackup(repositoryPath string, sourcePath string, backupIdentifier string,
|
|
bandwidthLimiter *ratelimit.Limiter, baseLogger *slog.Logger) (*agent.BackupRunResult, error) {
|
|
measurementContext := context.Background()
|
|
|
|
openedRepository, repositoryError := openOrCreateRepository(measurementContext,
|
|
repositoryPath, baseLogger)
|
|
if repositoryError != nil {
|
|
return nil, repositoryError
|
|
}
|
|
|
|
defer func() { _ = openedRepository.Close() }()
|
|
|
|
backupEngine := backupengine.NewEngine(openedRepository, nil, baseLogger)
|
|
backupRunner := agent.NewBackupRunner(backupEngine, baseLogger)
|
|
|
|
return backupRunner.RunBackup(measurementContext, agent.BackupRunOptions{
|
|
BackupID: backupIdentifier,
|
|
SourcePath: sourcePath,
|
|
SourceName: filepath.Base(sourcePath),
|
|
CompressionLevel: backupengine.CompressionBalanced,
|
|
ChainID: "bench-" + filepath.Base(repositoryPath),
|
|
BandwidthLimiter: bandwidthLimiter,
|
|
CreatedByVersion: serviceName,
|
|
})
|
|
}
|
|
|
|
// openOrCreateRepository öffnet ein Repository oder legt es an.
|
|
func openOrCreateRepository(openContext context.Context, repositoryPath string,
|
|
baseLogger *slog.Logger) (*repository.LocalRepository, error) {
|
|
openedRepository, openError := repository.Open(openContext, repositoryPath,
|
|
repository.OpenOptions{}, baseLogger)
|
|
if openError == nil {
|
|
return openedRepository, nil
|
|
}
|
|
|
|
return repository.Create(openContext, repositoryPath,
|
|
repository.CreateOptions{Name: filepath.Base(repositoryPath)}, baseLogger)
|
|
}
|
|
|
|
// prepareIncompressibleFile legt eine Datei mit Zufallsdaten an.
|
|
//
|
|
// Zufallsdaten und nicht wiederholter Text: Bei komprimierbaren Daten misst man
|
|
// zstd, nicht die Anlage. Vorhandene Dateien werden wiederverwendet — das
|
|
// Erzeugen von zwei Gigabyte kostet mehr Zeit als die Messung selbst.
|
|
func prepareIncompressibleFile(filePath string, totalBytes int64) error {
|
|
if fileInformation, statError := os.Stat(filePath); statError == nil {
|
|
if fileInformation.Size() == totalBytes {
|
|
return nil
|
|
}
|
|
}
|
|
|
|
if directoryError := os.MkdirAll(filepath.Dir(filePath), 0o700); directoryError != nil {
|
|
return fmt.Errorf("das quellverzeichnis liess sich nicht anlegen: %w", directoryError)
|
|
}
|
|
|
|
targetFile, createError := os.Create(filePath)
|
|
if createError != nil {
|
|
return fmt.Errorf("die quelldatei liess sich nicht anlegen: %w", createError)
|
|
}
|
|
|
|
defer func() { _ = targetFile.Close() }()
|
|
|
|
if _, copyError := io.CopyN(targetFile, rand.Reader, totalBytes); copyError != nil {
|
|
return fmt.Errorf("die quelldatei liess sich nicht fuellen: %w", copyError)
|
|
}
|
|
|
|
return targetFile.Sync()
|
|
}
|
|
|
|
// prepareManyFiles legt viele kleine Dateien an.
|
|
func prepareManyFiles(directoryPath string, fileCount int, fileBytes int64) error {
|
|
if entries, readError := os.ReadDir(directoryPath); readError == nil && len(entries) == fileCount {
|
|
return nil
|
|
}
|
|
|
|
if directoryError := os.MkdirAll(directoryPath, 0o700); directoryError != nil {
|
|
return fmt.Errorf("das quellverzeichnis liess sich nicht anlegen: %w", directoryError)
|
|
}
|
|
|
|
for fileIndex := 0; fileIndex < fileCount; fileIndex++ {
|
|
filePath := filepath.Join(directoryPath, fmt.Sprintf("datei-%06d.bin", fileIndex))
|
|
|
|
if prepareError := prepareIncompressibleFile(filePath, fileBytes); prepareError != nil {
|
|
return prepareError
|
|
}
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// isVersionArgument erkennt eine Versionsabfrage.
|
|
//
|
|
// Drei Schreibweisen, weil sich niemand merkt, welche ein bestimmtes Programm
|
|
// erwartet — und weil eine Fehlermeldung auf "--version" der denkbar
|
|
// schlechteste erste Eindruck ist.
|
|
func isVersionArgument(argument string) bool {
|
|
return argument == "version" || argument == "--version" || argument == "-version"
|
|
}
|