diff --git a/DocumentOperator.Infrastructure/Services/PdfProcessing/DevExpressPdfProcessor.cs b/DocumentOperator.Infrastructure/Services/PdfProcessing/DevExpressPdfProcessor.cs
index 372b6c3..0b296f6 100644
--- a/DocumentOperator.Infrastructure/Services/PdfProcessing/DevExpressPdfProcessor.cs
+++ b/DocumentOperator.Infrastructure/Services/PdfProcessing/DevExpressPdfProcessor.cs
@@ -20,7 +20,7 @@ public class DevExpressPdfProcessor : IPdfProcessor
/// PDF content as stream (caller is responsible for disposal)
/// PDF metadata (page count, file size, version, etc.)
/// Thrown when stream is empty or invalid
- public async Task ValidateAsync(Stream pdfStream)
+ public async Task ValidateAsync(Stream pdfStream)
{
// 1. Input Validation
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
@@ -73,13 +73,18 @@ public class DevExpressPdfProcessor : IPdfProcessor
// We scan PDF raw data for "/EmbeddedFiles" and parse the name tree to get count.
var (hasAttachments, attachmentCount) = DetectEmbeddedFiles(pdfBytes);
- // 5. Create and return PdfMetadata DTO
- return new Application.Common.DTOs.PdfMetadata(
- pageCount: pageCount,
- fileSizeBytes: pdfBytes.Length,
- pdfVersion: pdfVersion,
- hasAttachments: hasAttachments,
- attachmentCount: attachmentCount
+ // Check encryption (scan PDF raw data for /Encrypt keyword)
+ bool isEncrypted = DetectEncryption(pdfBytes);
+
+ // 5. Create and return PdfValidationResult
+ return new Application.Common.DTOs.PdfValidationResult(
+ PageCount: pageCount,
+ FileSizeBytes: pdfBytes.Length,
+ FileSizeMB: pdfBytes.Length / 1024.0 / 1024.0,
+ PdfVersion: pdfVersion,
+ HasAttachments: hasAttachments,
+ AttachmentCount: attachmentCount,
+ IsEncrypted: isEncrypted
);
}
@@ -89,7 +94,7 @@ public class DevExpressPdfProcessor : IPdfProcessor
/// PDF content as stream (caller is responsible for disposal)
/// PDF/A metadata including conformance level and validation errors/warnings
/// Thrown when stream is empty or invalid
- public async Task ValidatePdfAAsync(Stream pdfStream)
+ public async Task ValidatePdfAAsync(Stream pdfStream)
{
// 1. Input Validation
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
@@ -162,18 +167,19 @@ public class DevExpressPdfProcessor : IPdfProcessor
// 8. Determine overall validity
bool isValid = errors.Count == 0;
- // 9. Create and return PdfAMetadata DTO
- return new PdfAMetadata(
- isValid: isValid,
- pdfVersion: pdfVersion,
- pageCount: pageCount,
- fileSizeBytes: pdfBytes.Length,
- encrypted: encrypted,
- pdfaVersion: pdfaVersion,
- pdfaCompliant: isPdfACompliant,
- errors: errors,
- warnings: warnings
- );
+ // 9. Create and return PdfAValidationResult
+ return new PdfAValidationResult
+ {
+ IsValid = isValid,
+ PdfVersion = pdfVersion,
+ PageCount = pageCount,
+ FileSize = pdfBytes.Length,
+ Encrypted = encrypted,
+ PdfAVersion = pdfaVersion,
+ PdfACompliant = isPdfACompliant,
+ Errors = errors,
+ Warnings = warnings
+ };
}
#endregion
@@ -655,14 +661,108 @@ public class DevExpressPdfProcessor : IPdfProcessor
#region Private Helpers
///
- /// Detects if PDF is encrypted by scanning for /Encrypt keyword.
+ /// Detects if a PDF is encrypted by scanning the PDF trailer/xref section for an /Encrypt dictionary reference.
+ ///
+ /// Why NOT use Encoding.ASCII.GetString():
+ /// - Binary PDFs contain bytes 0x80–0xFF which ASCII decoding maps to '?' (0x3F), corrupting data.
+ /// - A string search on corrupted data can produce both false positives and false negatives.
+ ///
+ /// This implementation searches the raw bytes directly, which is safe for all PDF variants
+ /// (binary, linearized, cross-reference stream PDFs).
+ ///
+ /// According to the PDF specification (ISO 32000), the /Encrypt entry only appears in the
+ /// document's trailer dictionary. Searching only the last ~2KB (where the trailer lives)
+ /// eliminates false positives from document content that might contain the literal text "/Encrypt".
+ ///
+ /// Limitation: Cross-reference stream PDFs (PDF 1.5+) embed the trailer inline in a compressed
+ /// stream object. In those cases the raw /Encrypt bytes may not appear in the trailer section,
+ /// so we fall back to searching the entire file. This is still byte-level and therefore safe.
///
private static bool DetectEncryption(byte[] pdfBytes)
{
- string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes);
- return pdfText.Contains("/Encrypt", StringComparison.Ordinal);
+ // The byte sequence we are looking for (ASCII, safe because PDF keywords are always ASCII)
+ ReadOnlySpan encryptKeyword = "/Encrypt"u8;
+
+ // Strategy 1: Search only the last 2 KB (trailer area).
+ // The PDF trailer dictionary is always near the end of the file and contains
+ // /Encrypt only when the document is encrypted. This avoids false positives
+ // from document content.
+ int trailerSearchStart = Math.Max(0, pdfBytes.Length - 2048);
+ ReadOnlySpan trailerRegion = pdfBytes.AsSpan(trailerSearchStart);
+
+ if (ContainsByteSequence(trailerRegion, encryptKeyword))
+ return true;
+
+ // Strategy 2: Some PDF generators (e.g. linearized PDFs) place the trailer
+ // at the beginning of the file as well. Search the first 2 KB.
+ int headerSearchEnd = Math.Min(2048, pdfBytes.Length);
+ ReadOnlySpan headerRegion = pdfBytes.AsSpan(0, headerSearchEnd);
+
+ if (ContainsByteSequence(headerRegion, encryptKeyword))
+ return true;
+
+ // Strategy 3: Full-file fallback for cross-reference stream PDFs (PDF 1.5+).
+ // In these files the xref/trailer is a compressed stream object that can appear
+ // anywhere. We scan the entire file but require that /Encrypt is followed by
+ // a space or '<' (start of dictionary), which rules out arbitrary text matches.
+ ReadOnlySpan fullFile = pdfBytes.AsSpan();
+ int idx = 0;
+ while (idx <= pdfBytes.Length - encryptKeyword.Length)
+ {
+ int found = IndexOf(fullFile.Slice(idx), encryptKeyword);
+ if (found == -1)
+ break;
+
+ int absolutePos = idx + found;
+ int afterKeyword = absolutePos + encryptKeyword.Length;
+
+ // Verify the character after /Encrypt is a PDF structural character,
+ // not part of a longer token (e.g. "/Encrypted" should not match).
+ if (afterKeyword < pdfBytes.Length)
+ {
+ byte next = pdfBytes[afterKeyword];
+ // Valid PDF token terminators: space (0x20), newline, tab, '<', '/', '['
+ if (next == 0x20 || next == 0x0A || next == 0x0D || next == 0x09 ||
+ next == 0x3C || next == 0x2F || next == 0x5B)
+ {
+ return true;
+ }
+ }
+ else
+ {
+ // /Encrypt is the very last token in the file
+ return true;
+ }
+
+ idx = absolutePos + 1;
+ }
+
+ return false;
}
+ ///
+ /// Returns the index of within ,
+ /// or -1 if not found. Operates entirely on raw bytes — no string allocation.
+ ///
+ private static int IndexOf(ReadOnlySpan haystack, ReadOnlySpan needle)
+ {
+ if (needle.IsEmpty) return 0;
+ if (haystack.Length < needle.Length) return -1;
+
+ for (int i = 0; i <= haystack.Length - needle.Length; i++)
+ {
+ if (haystack.Slice(i, needle.Length).SequenceEqual(needle))
+ return i;
+ }
+ return -1;
+ }
+
+ ///
+ /// Returns true if contains as a byte subsequence.
+ ///
+ private static bool ContainsByteSequence(ReadOnlySpan haystack, ReadOnlySpan needle)
+ => IndexOf(haystack, needle) >= 0;
+
///
/// Detects PDF/A conformance level by scanning PDF metadata.
/// PDF/A documents contain an XMP metadata stream with pdfaid:conformance and pdfaid:part.