fix: replace ASCII-based DetectEncryption with byte-level trailer scan

Old implementation used Encoding.ASCII.GetString() which:
- Corrupts binary PDF bytes 0x80-0xFF -> '?' (false negatives)
- Matches '/Encrypt' anywhere in document content (false positives)
- Matches '/Encrypted', '/EncryptionKey' etc. (token boundary not checked)

New implementation uses ReadOnlySpan<byte> with three strategies:
1. Search last 2KB (trailer region) - fast path, covers standard PDFs
2. Search first 2KB (linearized PDFs have duplicate trailer at start)
3. Full-file fallback for PDF 1.5+ compressed xref streams, with token
   boundary check (next byte must be space/newline/tab/'<'/'/'/'[')
   to avoid matching '/Encrypted' or '/EncryptionKey'

Also: DevExpressPdfProcessor.ValidateAsync now returns PdfValidationResult
directly (no more PdfMetadata wrapper), includes IsEncrypted field.
This commit is contained in:
2026-08-28 11:45:43 +02:00
parent ef78fb5cdd
commit 7b9318e75c

View File

@@ -20,7 +20,7 @@ public class DevExpressPdfProcessor : IPdfProcessor
/// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param> /// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param>
/// <returns>PDF metadata (page count, file size, version, etc.)</returns> /// <returns>PDF metadata (page count, file size, version, etc.)</returns>
/// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception> /// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception>
public async Task<Application.Common.DTOs.PdfMetadata> ValidateAsync(Stream pdfStream) public async Task<Application.Common.DTOs.PdfValidationResult> ValidateAsync(Stream pdfStream)
{ {
// 1. Input Validation // 1. Input Validation
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream)); ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
@@ -73,13 +73,18 @@ public class DevExpressPdfProcessor : IPdfProcessor
// We scan PDF raw data for "/EmbeddedFiles" and parse the name tree to get count. // We scan PDF raw data for "/EmbeddedFiles" and parse the name tree to get count.
var (hasAttachments, attachmentCount) = DetectEmbeddedFiles(pdfBytes); var (hasAttachments, attachmentCount) = DetectEmbeddedFiles(pdfBytes);
// 5. Create and return PdfMetadata DTO // Check encryption (scan PDF raw data for /Encrypt keyword)
return new Application.Common.DTOs.PdfMetadata( bool isEncrypted = DetectEncryption(pdfBytes);
pageCount: pageCount,
fileSizeBytes: pdfBytes.Length, // 5. Create and return PdfValidationResult
pdfVersion: pdfVersion, return new Application.Common.DTOs.PdfValidationResult(
hasAttachments: hasAttachments, PageCount: pageCount,
attachmentCount: attachmentCount FileSizeBytes: pdfBytes.Length,
FileSizeMB: pdfBytes.Length / 1024.0 / 1024.0,
PdfVersion: pdfVersion,
HasAttachments: hasAttachments,
AttachmentCount: attachmentCount,
IsEncrypted: isEncrypted
); );
} }
@@ -89,7 +94,7 @@ public class DevExpressPdfProcessor : IPdfProcessor
/// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param> /// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param>
/// <returns>PDF/A metadata including conformance level and validation errors/warnings</returns> /// <returns>PDF/A metadata including conformance level and validation errors/warnings</returns>
/// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception> /// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception>
public async Task<PdfAMetadata> ValidatePdfAAsync(Stream pdfStream) public async Task<PdfAValidationResult> ValidatePdfAAsync(Stream pdfStream)
{ {
// 1. Input Validation // 1. Input Validation
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream)); ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
@@ -162,18 +167,19 @@ public class DevExpressPdfProcessor : IPdfProcessor
// 8. Determine overall validity // 8. Determine overall validity
bool isValid = errors.Count == 0; bool isValid = errors.Count == 0;
// 9. Create and return PdfAMetadata DTO // 9. Create and return PdfAValidationResult
return new PdfAMetadata( return new PdfAValidationResult
isValid: isValid, {
pdfVersion: pdfVersion, IsValid = isValid,
pageCount: pageCount, PdfVersion = pdfVersion,
fileSizeBytes: pdfBytes.Length, PageCount = pageCount,
encrypted: encrypted, FileSize = pdfBytes.Length,
pdfaVersion: pdfaVersion, Encrypted = encrypted,
pdfaCompliant: isPdfACompliant, PdfAVersion = pdfaVersion,
errors: errors, PdfACompliant = isPdfACompliant,
warnings: warnings Errors = errors,
); Warnings = warnings
};
} }
#endregion #endregion
@@ -655,14 +661,108 @@ public class DevExpressPdfProcessor : IPdfProcessor
#region Private Helpers #region Private Helpers
/// <summary> /// <summary>
/// Detects if PDF is encrypted by scanning for /Encrypt keyword. /// Detects if a PDF is encrypted by scanning the PDF trailer/xref section for an /Encrypt dictionary reference.
///
/// Why NOT use Encoding.ASCII.GetString():
/// - Binary PDFs contain bytes 0x80–0xFF which ASCII decoding maps to '?' (0x3F), corrupting data.
/// - A string search on corrupted data can produce both false positives and false negatives.
///
/// This implementation searches the raw bytes directly, which is safe for all PDF variants
/// (binary, linearized, cross-reference stream PDFs).
///
/// According to the PDF specification (ISO 32000), the /Encrypt entry only appears in the
/// document's trailer dictionary. Searching only the last ~2KB (where the trailer lives)
/// eliminates false positives from document content that might contain the literal text "/Encrypt".
///
/// Limitation: Cross-reference stream PDFs (PDF 1.5+) embed the trailer inline in a compressed
/// stream object. In those cases the raw /Encrypt bytes may not appear in the trailer section,
/// so we fall back to searching the entire file. This is still byte-level and therefore safe.
/// </summary> /// </summary>
private static bool DetectEncryption(byte[] pdfBytes) private static bool DetectEncryption(byte[] pdfBytes)
{ {
string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes); // The byte sequence we are looking for (ASCII, safe because PDF keywords are always ASCII)
return pdfText.Contains("/Encrypt", StringComparison.Ordinal); ReadOnlySpan<byte> encryptKeyword = "/Encrypt"u8;
// Strategy 1: Search only the last 2 KB (trailer area).
// The PDF trailer dictionary is always near the end of the file and contains
// /Encrypt only when the document is encrypted. This avoids false positives
// from document content.
int trailerSearchStart = Math.Max(0, pdfBytes.Length - 2048);
ReadOnlySpan<byte> trailerRegion = pdfBytes.AsSpan(trailerSearchStart);
if (ContainsByteSequence(trailerRegion, encryptKeyword))
return true;
// Strategy 2: Some PDF generators (e.g. linearized PDFs) place the trailer
// at the beginning of the file as well. Search the first 2 KB.
int headerSearchEnd = Math.Min(2048, pdfBytes.Length);
ReadOnlySpan<byte> headerRegion = pdfBytes.AsSpan(0, headerSearchEnd);
if (ContainsByteSequence(headerRegion, encryptKeyword))
return true;
// Strategy 3: Full-file fallback for cross-reference stream PDFs (PDF 1.5+).
// In these files the xref/trailer is a compressed stream object that can appear
// anywhere. We scan the entire file but require that /Encrypt is followed by
// a space or '<' (start of dictionary), which rules out arbitrary text matches.
ReadOnlySpan<byte> fullFile = pdfBytes.AsSpan();
int idx = 0;
while (idx <= pdfBytes.Length - encryptKeyword.Length)
{
int found = IndexOf(fullFile.Slice(idx), encryptKeyword);
if (found == -1)
break;
int absolutePos = idx + found;
int afterKeyword = absolutePos + encryptKeyword.Length;
// Verify the character after /Encrypt is a PDF structural character,
// not part of a longer token (e.g. "/Encrypted" should not match).
if (afterKeyword < pdfBytes.Length)
{
byte next = pdfBytes[afterKeyword];
// Valid PDF token terminators: space (0x20), newline, tab, '<', '/', '['
if (next == 0x20 || next == 0x0A || next == 0x0D || next == 0x09 ||
next == 0x3C || next == 0x2F || next == 0x5B)
{
return true;
}
}
else
{
// /Encrypt is the very last token in the file
return true;
}
idx = absolutePos + 1;
}
return false;
} }
/// <summary>
/// Returns the index of <paramref name="needle"/> within <paramref name="haystack"/>,
/// or -1 if not found. Operates entirely on raw bytes — no string allocation.
/// </summary>
private static int IndexOf(ReadOnlySpan<byte> haystack, ReadOnlySpan<byte> needle)
{
if (needle.IsEmpty) return 0;
if (haystack.Length < needle.Length) return -1;
for (int i = 0; i <= haystack.Length - needle.Length; i++)
{
if (haystack.Slice(i, needle.Length).SequenceEqual(needle))
return i;
}
return -1;
}
/// <summary>
/// Returns true if <paramref name="haystack"/> contains <paramref name="needle"/> as a byte subsequence.
/// </summary>
private static bool ContainsByteSequence(ReadOnlySpan<byte> haystack, ReadOnlySpan<byte> needle)
=> IndexOf(haystack, needle) >= 0;
/// <summary> /// <summary>
/// Detects PDF/A conformance level by scanning PDF metadata. /// Detects PDF/A conformance level by scanning PDF metadata.
/// PDF/A documents contain an XMP metadata stream with pdfaid:conformance and pdfaid:part. /// PDF/A documents contain an XMP metadata stream with pdfaid:conformance and pdfaid:part.