import { PDFDocument } from 'pdf-lib'; import { parseXJustizMetadata, type XJustizMetadata } from './services/xml-reading.service'; import { unzipArchive } from './services/zip-inflating.service'; export type ZipAttachmentKind = 'pdf' | 'image'; export type ZipAttachment = { name: string; path: string; kind: ZipAttachmentKind; data: Uint8Array; }; export type ProcessedZipArchive = { name: string; attachments: ZipAttachment[]; }; const IMAGE_EXTENSIONS = new Set([ '.avif', '.bmp', '.gif', '.heic', '.jpeg', '.jpg', '.jp2', '.png', '.tif', '.tiff', '.webp' ]); const ZIP_META_FILE_NAME = 'xjustiz_nachricht.xml'; const getBaseName = (path: string) => path.split('/').pop() ?? path; const getCurrentDateString = (date: Date) => `${date.getFullYear()}-${String(date.getMonth() + 1).padStart(2, '0')}-${String(date.getDate()).padStart(2, '0')}`; const createArchiveName = (date: Date, metadata: XJustizMetadata | null) => { const datePrefix = getCurrentDateString(date); if (!metadata) { return `${datePrefix}_unbekannt_unbekannt.pdf`; } return `${datePrefix}_${metadata.sender}_${metadata.receiver}.pdf`; }; const getExtension = (path: string) => { const baseName = getBaseName(path).toLowerCase(); const lastDotIndex = baseName.lastIndexOf('.'); return lastDotIndex === -1 ? '' : baseName.slice(lastDotIndex); }; const isPdf = (path: string) => getExtension(path) === '.pdf'; const isImage = (path: string) => IMAGE_EXTENSIONS.has(getExtension(path)); const isZipArchive = (path: string) => getExtension(path) === '.zip'; const normalizeKey = (value: string) => value.replaceAll('\\', '/').toLowerCase().trim(); const isMacMetadataEntry = (path: string) => { const normalizedPath = normalizeKey(path); const baseName = getBaseName(normalizedPath); return ( normalizedPath.startsWith('__macosx/') || normalizedPath.includes('/__macosx/') || baseName === '.ds_store' || baseName.startsWith('._') ); }; const orderAttachmentsByDocumentNames = (attachments: ZipAttachment[], documentNames: string[]) => { const buckets = new Map(); for (const attachment of attachments) { const keys = new Set([normalizeKey(attachment.path), normalizeKey(attachment.name)]); for (const key of keys) { const bucket = buckets.get(key) ?? []; bucket.push(attachment); buckets.set(key, bucket); } } const usedAttachments = new Set(); const orderedAttachments: ZipAttachment[] = []; const takeAttachment = (key: string) => { const bucket = buckets.get(key); if (!bucket) { return null; } const nextAttachment = bucket.find((attachment) => !usedAttachments.has(attachment)) ?? null; if (nextAttachment) { usedAttachments.add(nextAttachment); } return nextAttachment; }; for (const documentName of documentNames) { const normalizedName = normalizeKey(documentName); const normalizedBaseName = normalizeKey(getBaseName(documentName)); const matchedAttachment = takeAttachment(normalizedName) ?? takeAttachment(normalizedBaseName); if (matchedAttachment) { orderedAttachments.push(matchedAttachment); } } for (const attachment of attachments) { if (!usedAttachments.has(attachment)) { orderedAttachments.push(attachment); } } return orderedAttachments; }; const shouldSkipZipEntry = (path: string) => isMacMetadataEntry(path); const isMetadataFile = (path: string) => getBaseName(path).toLowerCase() === ZIP_META_FILE_NAME; const createAttachment = (path: string, data: Uint8Array): ZipAttachment => ({ name: getBaseName(path), path, kind: isPdf(path) ? 'pdf' : 'image', data }); const extractZipEntries = async (archiveBytes: Uint8Array): Promise => { const files = await unzipArchive(archiveBytes); const attachments: ZipAttachment[] = []; const nestedArchives: ProcessedZipArchive[] = []; let xjustizNachrichtXml: Uint8Array | null = null; for (const [path, data] of Object.entries(files)) { if (shouldSkipZipEntry(path)) { continue; } if (isMetadataFile(path)) { xjustizNachrichtXml = data; continue; } if (isZipArchive(path)) { const childArchives = await extractZipEntries(data); nestedArchives.push(...childArchives); continue; } if (isPdf(path) || isImage(path)) { attachments.push(createAttachment(path, data)); } } const metadata = xjustizNachrichtXml ? parseXJustizMetadata(xjustizNachrichtXml) : null; const documentNames = metadata?.documentNames ?? []; const orderedAttachments = orderAttachmentsByDocumentNames(attachments, documentNames); const archiveName = createArchiveName(new Date(), metadata); const archives: ProcessedZipArchive[] = []; if (orderedAttachments.length > 0) { archives.push({ name: archiveName, attachments: orderedAttachments }); } return [...archives, ...nestedArchives]; }; export const extractZipArchives = async (file: File): Promise => { const archiveBytes = new Uint8Array(await file.arrayBuffer()); return extractZipEntries(archiveBytes); }; const loadImageElement = (src: string) => new Promise((resolve, reject) => { const image = new Image(); image.decoding = 'async'; image.onload = () => resolve(image); image.onerror = () => reject(new Error(`Unable to decode image asset: ${src}`)); image.src = src; }); const toArrayBuffer = (bytes: Uint8Array) => (() => { const copy = new Uint8Array(bytes.byteLength); copy.set(bytes); return copy.buffer; })(); const convertImageBytesToPng = async (imageBytes: Uint8Array) => { if (typeof document === 'undefined') { throw new Error('Image conversion requires a browser environment.'); } const objectUrl = URL.createObjectURL(new Blob([toArrayBuffer(imageBytes)])); try { const image = await loadImageElement(objectUrl); const canvas = document.createElement('canvas'); canvas.width = image.naturalWidth || image.width; canvas.height = image.naturalHeight || image.height; const context = canvas.getContext('2d'); if (!context) { throw new Error('Could not create a canvas context for image conversion.'); } context.drawImage(image, 0, 0); const pngBlob = await new Promise((resolve, reject) => { canvas.toBlob((blob) => { if (blob) { resolve(blob); return; } reject(new Error('Could not encode image asset as PNG.')); }, 'image/png'); }); return new Uint8Array(await pngBlob.arrayBuffer()); } finally { URL.revokeObjectURL(objectUrl); } }; const embedAttachmentImage = async (pdfDocument: PDFDocument, attachment: ZipAttachment) => { const extension = getExtension(attachment.path); const imageData = attachment.data; if (extension === '.jpg' || extension === '.jpeg') { return pdfDocument.embedJpg(imageData); } if (extension === '.png') { return pdfDocument.embedPng(imageData); } const pngBytes = await convertImageBytesToPng(imageData); return pdfDocument.embedPng(pngBytes); }; const mergeAttachmentIntoPdf = async (mergedPdf: PDFDocument, attachment: ZipAttachment) => { if (attachment.kind === 'pdf') { const sourcePdf = await PDFDocument.load(attachment.data, { ignoreEncryption: true }); const copiedPages = await mergedPdf.copyPages(sourcePdf, sourcePdf.getPageIndices()); for (const page of copiedPages) { mergedPdf.addPage(page); } return; } try { const embeddedImage = await embedAttachmentImage(mergedPdf, attachment); const page = mergedPdf.addPage([embeddedImage.width, embeddedImage.height]); page.drawImage(embeddedImage, { x: 0, y: 0, width: page.getWidth(), height: page.getHeight() }); } catch (error) { throw error; } }; export const mergeProcessedZipArchive = async (archive: ProcessedZipArchive) => { if (archive.attachments.length === 0) { throw new Error('Cannot create a PDF from an empty archive.'); } const mergedPdf = await PDFDocument.create(); for (const attachment of archive.attachments) { await mergeAttachmentIntoPdf(mergedPdf, attachment); } return mergedPdf.save(); };