/** * Analysiert alle PDFs in data/import, ohne etwas zu schreiben. * Prüft, ob der Textbaustein-Teil überall identisch ist und listet die * variablen Inhalte sowie Abweichungen auf. * * node scripts/analyze-pdfs.js */ const fs = require("fs") const path = require("path") const { parsePolicyPdf, reflow, normalize } = require("./lib/pdf-policy") const IMPORT_DIR = path.join(process.cwd(), "data", "import") const BOILERPLATE = [ "Name und Kontaktdaten des Verantwortlichen", "Kontaktdaten der Datenschutzbeauftragten", "Betroffenen-Rechte", "Beschwerderecht bei der Aufsichtsbehörde", "Widerrufsrecht bei Einwilligung", ] const VARIABLE = { activity: "Bezeichnung der Verarbeitungstätigkeit", purpose: "Ihre Daten werden zu folgendem Zweck erhoben", legal: "Ihre Daten wurden aufgrund folgender Rechtsgrundlage erhoben", recipients: "Wir beabsichtigen, Ihre Daten an folgende Empfänger weiterzuleiten", } async function main() { const files = fs .readdirSync(IMPORT_DIR) .filter((f) => f.toLowerCase().endsWith(".pdf")) .sort() const fingerprints = new Map() const rows = [] const problems = [] for (const file of files) { const parsed = await parsePolicyPdf(path.join(IMPORT_DIR, file)) const fingerprint = BOILERPLATE.map((h) => normalize(parsed.sections[h] || [""]) ).join(" || ") if (!fingerprints.has(fingerprint)) { fingerprints.set(fingerprint, { files: [], sample: parsed }) } fingerprints.get(fingerprint).files.push(file) const row = { file } for (const [key, heading] of Object.entries(VARIABLE)) { const section = parsed.sections[heading] row[key] = section ? reflow(section).join(" ⏎ ") : null if (!section) { problems.push(`${file}: Abschnitt fehlt – „${heading}“`) } } const extras = parsed.order.filter( (h) => !BOILERPLATE.includes(h) && !Object.values(VARIABLE).includes(h) && h !== "Zwecke und Rechtsgrundlagen der Verarbeitung" ) if (extras.length) row.extras = extras if (parsed.unknown.length) { problems.push(`${file}: unbekannte Überschriften – ${parsed.unknown.join(", ")}`) } rows.push(row) } console.log(`Dateien analysiert: ${files.length}\n`) console.log(`=== Textbaustein-Varianten: ${fingerprints.size} ===`) let variant = 0 for (const [, info] of fingerprints) { variant += 1 console.log(`\nVariante ${variant} – ${info.files.length} Datei(en)`) console.log(` Beispiel: ${info.files[0]}`) if (info.files.length <= 6) { console.log(` Dateien: ${info.files.join(", ")}`) } const resp = info.sample.sections["Name und Kontaktdaten des Verantwortlichen"] const dpo = info.sample.sections["Kontaktdaten der Datenschutzbeauftragten"] console.log(` Verantwortlicher: ${(resp || []).join(" | ")}`) console.log(` DSB: ${(dpo || []).join(" | ")}`) } console.log(`\n=== Verarbeitungstätigkeiten ===`) const titles = new Map() for (const row of rows) { const key = row.activity || "" if (!titles.has(key)) titles.set(key, []) titles.get(key).push(row.file) } for (const [title, list] of [...titles].sort((a, b) => a[0].localeCompare(b[0], "de"))) { const marker = list.length > 1 ? ` ⚠ ${list.length}×` : "" console.log(` ${title}${marker}`) if (list.length > 1) console.log(` ${list.join(", ")}`) } const extraRows = rows.filter((r) => r.extras) console.log(`\n=== Zusatzabschnitte (Speicherdauer / Herkunft): ${extraRows.length} ===`) for (const row of extraRows) { console.log(` ${row.file}: ${row.extras.join(", ")}`) } console.log(`\n=== Auffälligkeiten: ${problems.length} ===`) problems.forEach((p) => console.log(` ${p}`)) fs.writeFileSync( path.join(process.cwd(), "data", "analyse.json"), JSON.stringify(rows, null, 2), "utf8" ) console.log("\nDetails geschrieben nach data/analyse.json") } main().catch((e) => { console.error(e) process.exitCode = 1 })