/** * Liest die DSE-Gesamtliste des Amtes Leezen (Excel-Export, CSV). * * Der Export ist Windows-1252 kodiert, semikolongetrennt und enthält Felder mit * eingebetteten Zeilenumbrüchen und Semikolons in Anführungszeichen. */ const fs = require("fs") /** * Windows-1252 unterscheidet sich von Latin-1 nur im Bereich 0x80–0x9F. * Genau dort liegen die typografischen Zeichen aus Excel (Gedankenstrich, * Anführungszeichen), die sonst als Steuerzeichen ankämen. */ const CP1252_HIGH = { 0x80: "€", 0x82: "‚", 0x83: "ƒ", 0x84: "„", 0x85: "…", 0x86: "†", 0x87: "‡", 0x88: "ˆ", 0x89: "‰", 0x8a: "Š", 0x8b: "‹", 0x8c: "Œ", 0x8e: "Ž", 0x91: "‘", 0x92: "’", 0x93: "“", 0x94: "”", 0x95: "•", 0x96: "–", 0x97: "—", 0x98: "˜", 0x99: "™", 0x9a: "š", 0x9b: "›", 0x9c: "œ", 0x9e: "ž", 0x9f: "Ÿ", } function decodeCp1252(buffer) { let out = "" for (const byte of buffer) { out += byte >= 0x80 && byte <= 0x9f ? CP1252_HIGH[byte] ?? "�" : String.fromCharCode(byte) } return out } /** * Zerlegt CSV-Text nach RFC 4180: Felder in Anführungszeichen dürfen das * Trennzeichen und Zeilenumbrüche enthalten, "" steht für ein Anführungszeichen. */ function parseCsv(text, delimiter = ";") { const rows = [] let row = [] let field = "" let inQuotes = false for (let i = 0; i < text.length; i++) { const char = text[i] if (inQuotes) { if (char === '"') { if (text[i + 1] === '"') { field += '"' i++ } else { inQuotes = false } } else { field += char } continue } if (char === '"') { inQuotes = true } else if (char === delimiter) { row.push(field) field = "" } else if (char === "\r") { // Zeilenende erst beim \n verarbeiten } else if (char === "\n") { row.push(field) rows.push(row) row = [] field = "" } else { field += char } } if (field !== "" || row.length > 0) { row.push(field) rows.push(row) } return rows } /** Liest die Liste und liefert Objekte mit den Spaltenüberschriften als Schlüssel. */ function readPolicyCsv(filePath) { const text = decodeCp1252(fs.readFileSync(filePath)) const rows = parseCsv(text) if (rows.length === 0) return { header: [], records: [] } const header = rows[0].map((h) => h.replace(/\s+/g, " ").trim()) const records = rows .slice(1) // Vollständig leere Zeilen am Ende des Excel-Bereichs verwerfen .filter((r) => r.some((cell) => cell.trim() !== "")) .map((r, index) => { const record = { _zeile: index + 2 } header.forEach((name, i) => { record[name] = (r[i] ?? "").trim() }) return record }) return { header, records } } module.exports = { decodeCp1252, parseCsv, readPolicyCsv }