diff --git a/DocumentOperator.Infrastructure/Services/PdfProcessing/DevExpressPdfProcessor.cs b/DocumentOperator.Infrastructure/Services/PdfProcessing/DevExpressPdfProcessor.cs index 372b6c3..0b296f6 100644 --- a/DocumentOperator.Infrastructure/Services/PdfProcessing/DevExpressPdfProcessor.cs +++ b/DocumentOperator.Infrastructure/Services/PdfProcessing/DevExpressPdfProcessor.cs @@ -20,7 +20,7 @@ public class DevExpressPdfProcessor : IPdfProcessor /// PDF content as stream (caller is responsible for disposal) /// PDF metadata (page count, file size, version, etc.) /// Thrown when stream is empty or invalid - public async Task ValidateAsync(Stream pdfStream) + public async Task ValidateAsync(Stream pdfStream) { // 1. Input Validation ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream)); @@ -73,13 +73,18 @@ public class DevExpressPdfProcessor : IPdfProcessor // We scan PDF raw data for "/EmbeddedFiles" and parse the name tree to get count. var (hasAttachments, attachmentCount) = DetectEmbeddedFiles(pdfBytes); - // 5. Create and return PdfMetadata DTO - return new Application.Common.DTOs.PdfMetadata( - pageCount: pageCount, - fileSizeBytes: pdfBytes.Length, - pdfVersion: pdfVersion, - hasAttachments: hasAttachments, - attachmentCount: attachmentCount + // Check encryption (scan PDF raw data for /Encrypt keyword) + bool isEncrypted = DetectEncryption(pdfBytes); + + // 5. Create and return PdfValidationResult + return new Application.Common.DTOs.PdfValidationResult( + PageCount: pageCount, + FileSizeBytes: pdfBytes.Length, + FileSizeMB: pdfBytes.Length / 1024.0 / 1024.0, + PdfVersion: pdfVersion, + HasAttachments: hasAttachments, + AttachmentCount: attachmentCount, + IsEncrypted: isEncrypted ); } @@ -89,7 +94,7 @@ public class DevExpressPdfProcessor : IPdfProcessor /// PDF content as stream (caller is responsible for disposal) /// PDF/A metadata including conformance level and validation errors/warnings /// Thrown when stream is empty or invalid - public async Task ValidatePdfAAsync(Stream pdfStream) + public async Task ValidatePdfAAsync(Stream pdfStream) { // 1. Input Validation ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream)); @@ -162,18 +167,19 @@ public class DevExpressPdfProcessor : IPdfProcessor // 8. Determine overall validity bool isValid = errors.Count == 0; - // 9. Create and return PdfAMetadata DTO - return new PdfAMetadata( - isValid: isValid, - pdfVersion: pdfVersion, - pageCount: pageCount, - fileSizeBytes: pdfBytes.Length, - encrypted: encrypted, - pdfaVersion: pdfaVersion, - pdfaCompliant: isPdfACompliant, - errors: errors, - warnings: warnings - ); + // 9. Create and return PdfAValidationResult + return new PdfAValidationResult + { + IsValid = isValid, + PdfVersion = pdfVersion, + PageCount = pageCount, + FileSize = pdfBytes.Length, + Encrypted = encrypted, + PdfAVersion = pdfaVersion, + PdfACompliant = isPdfACompliant, + Errors = errors, + Warnings = warnings + }; } #endregion @@ -655,14 +661,108 @@ public class DevExpressPdfProcessor : IPdfProcessor #region Private Helpers /// - /// Detects if PDF is encrypted by scanning for /Encrypt keyword. + /// Detects if a PDF is encrypted by scanning the PDF trailer/xref section for an /Encrypt dictionary reference. + /// + /// Why NOT use Encoding.ASCII.GetString(): + /// - Binary PDFs contain bytes 0x80–0xFF which ASCII decoding maps to '?' (0x3F), corrupting data. + /// - A string search on corrupted data can produce both false positives and false negatives. + /// + /// This implementation searches the raw bytes directly, which is safe for all PDF variants + /// (binary, linearized, cross-reference stream PDFs). + /// + /// According to the PDF specification (ISO 32000), the /Encrypt entry only appears in the + /// document's trailer dictionary. Searching only the last ~2KB (where the trailer lives) + /// eliminates false positives from document content that might contain the literal text "/Encrypt". + /// + /// Limitation: Cross-reference stream PDFs (PDF 1.5+) embed the trailer inline in a compressed + /// stream object. In those cases the raw /Encrypt bytes may not appear in the trailer section, + /// so we fall back to searching the entire file. This is still byte-level and therefore safe. /// private static bool DetectEncryption(byte[] pdfBytes) { - string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes); - return pdfText.Contains("/Encrypt", StringComparison.Ordinal); + // The byte sequence we are looking for (ASCII, safe because PDF keywords are always ASCII) + ReadOnlySpan encryptKeyword = "/Encrypt"u8; + + // Strategy 1: Search only the last 2 KB (trailer area). + // The PDF trailer dictionary is always near the end of the file and contains + // /Encrypt only when the document is encrypted. This avoids false positives + // from document content. + int trailerSearchStart = Math.Max(0, pdfBytes.Length - 2048); + ReadOnlySpan trailerRegion = pdfBytes.AsSpan(trailerSearchStart); + + if (ContainsByteSequence(trailerRegion, encryptKeyword)) + return true; + + // Strategy 2: Some PDF generators (e.g. linearized PDFs) place the trailer + // at the beginning of the file as well. Search the first 2 KB. + int headerSearchEnd = Math.Min(2048, pdfBytes.Length); + ReadOnlySpan headerRegion = pdfBytes.AsSpan(0, headerSearchEnd); + + if (ContainsByteSequence(headerRegion, encryptKeyword)) + return true; + + // Strategy 3: Full-file fallback for cross-reference stream PDFs (PDF 1.5+). + // In these files the xref/trailer is a compressed stream object that can appear + // anywhere. We scan the entire file but require that /Encrypt is followed by + // a space or '<' (start of dictionary), which rules out arbitrary text matches. + ReadOnlySpan fullFile = pdfBytes.AsSpan(); + int idx = 0; + while (idx <= pdfBytes.Length - encryptKeyword.Length) + { + int found = IndexOf(fullFile.Slice(idx), encryptKeyword); + if (found == -1) + break; + + int absolutePos = idx + found; + int afterKeyword = absolutePos + encryptKeyword.Length; + + // Verify the character after /Encrypt is a PDF structural character, + // not part of a longer token (e.g. "/Encrypted" should not match). + if (afterKeyword < pdfBytes.Length) + { + byte next = pdfBytes[afterKeyword]; + // Valid PDF token terminators: space (0x20), newline, tab, '<', '/', '[' + if (next == 0x20 || next == 0x0A || next == 0x0D || next == 0x09 || + next == 0x3C || next == 0x2F || next == 0x5B) + { + return true; + } + } + else + { + // /Encrypt is the very last token in the file + return true; + } + + idx = absolutePos + 1; + } + + return false; } + /// + /// Returns the index of within , + /// or -1 if not found. Operates entirely on raw bytes — no string allocation. + /// + private static int IndexOf(ReadOnlySpan haystack, ReadOnlySpan needle) + { + if (needle.IsEmpty) return 0; + if (haystack.Length < needle.Length) return -1; + + for (int i = 0; i <= haystack.Length - needle.Length; i++) + { + if (haystack.Slice(i, needle.Length).SequenceEqual(needle)) + return i; + } + return -1; + } + + /// + /// Returns true if contains as a byte subsequence. + /// + private static bool ContainsByteSequence(ReadOnlySpan haystack, ReadOnlySpan needle) + => IndexOf(haystack, needle) >= 0; + /// /// Detects PDF/A conformance level by scanning PDF metadata. /// PDF/A documents contain an XMP metadata stream with pdfaid:conformance and pdfaid:part.