using DevExpress.Pdf; using DocumentOperator.Application.Common.Interfaces; using DocumentOperator.Domain.Common.Exceptions; namespace DocumentOperator.Infrastructure.Services.PdfProcessing; /// /// PDF processor implementation using DevExpress.Pdf library. /// Handles PDF validation and metadata extraction. /// public class DevExpressPdfProcessor : IPdfProcessor { /// /// Validates a PDF document and returns metadata. /// /// PDF content as byte array /// PDF metadata (page count, file size, version, etc.) /// Thrown when PDF is invalid or null public async Task ValidateAsync(byte[] pdfBytes) { // 1. Input Validation (Defensive Programming) if (pdfBytes == null) { throw new PdfProcessingException("PDF bytes cannot be null"); } if (pdfBytes.Length == 0) { throw new PdfProcessingException("PDF bytes cannot be empty"); } try { // 2. Load PDF with DevExpress Document API (PdfDocumentProcessor) using var processor = new PdfDocumentProcessor(); processor.LoadDocument(new MemoryStream(pdfBytes)); // 3. Extract metadata var document = processor.Document; int pageCount = document.Pages.Count; string pdfVersion = document.Version.ToString(); // z.B. "1.4", "1.7" // Attachments (embedded files) // DevExpress PdfDocument API doesn't expose EmbeddedFiles directly. // We use a simple PDF raw data scan for "/EmbeddedFiles" keyword. // This is a pragmatic approach until Feature 2 (ExtractAttachments) is implemented. bool hasAttachments = DetectEmbeddedFiles(pdfBytes); int attachmentCount = hasAttachments ? -1 : 0; // -1 = "has attachments, count unknown" // 4. Create and return PdfMetadata Value Object (fully qualified name!) return new DocumentOperator.Domain.Models.ValueObjects.PdfMetadata( pageCount: pageCount, fileSizeBytes: pdfBytes.Length, pdfVersion: pdfVersion, hasAttachments: hasAttachments, attachmentCount: attachmentCount ); } catch (Exception ex) when (ex is not PdfProcessingException) { // Wrap DevExpress exceptions in our domain exception throw new PdfProcessingException( $"Failed to validate PDF: {ex.Message}", ex); } } /// /// Detects embedded files in PDF by scanning raw PDF data for /EmbeddedFiles keyword. /// This is a pragmatic approach as DevExpress PdfDocument API doesn't expose EmbeddedFiles directly. /// /// PDF raw bytes /// True if PDF contains /EmbeddedFiles keyword in proper context, false otherwise private static bool DetectEmbeddedFiles(byte[] pdfBytes) { // PDF embedded files are declared in the document catalog: // /Names << /EmbeddedFiles << /Names [...] >> >> // We search for the pattern "/Names" followed by "/EmbeddedFiles" string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes); // Look for the specific PDF dictionary pattern: /Names and /EmbeddedFiles // This is more precise than just searching for /EmbeddedFiles alone int namesIndex = pdfText.IndexOf("/Names", StringComparison.Ordinal); if (namesIndex == -1) return false; // Check if /EmbeddedFiles appears after /Names within reasonable distance (< 1000 chars) int embeddedFilesIndex = pdfText.IndexOf("/EmbeddedFiles", namesIndex, Math.Min(1000, pdfText.Length - namesIndex), StringComparison.Ordinal); return embeddedFilesIndex > namesIndex; } }