fix: replace ASCII-based DetectEncryption with byte-level trailer scan
Old implementation used Encoding.ASCII.GetString() which: - Corrupts binary PDF bytes 0x80-0xFF -> '?' (false negatives) - Matches '/Encrypt' anywhere in document content (false positives) - Matches '/Encrypted', '/EncryptionKey' etc. (token boundary not checked) New implementation uses ReadOnlySpan<byte> with three strategies: 1. Search last 2KB (trailer region) - fast path, covers standard PDFs 2. Search first 2KB (linearized PDFs have duplicate trailer at start) 3. Full-file fallback for PDF 1.5+ compressed xref streams, with token boundary check (next byte must be space/newline/tab/'<'/'/'/'[') to avoid matching '/Encrypted' or '/EncryptionKey' Also: DevExpressPdfProcessor.ValidateAsync now returns PdfValidationResult directly (no more PdfMetadata wrapper), includes IsEncrypted field.
This commit is contained in:
@@ -20,7 +20,7 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
||||
/// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param>
|
||||
/// <returns>PDF metadata (page count, file size, version, etc.)</returns>
|
||||
/// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception>
|
||||
public async Task<Application.Common.DTOs.PdfMetadata> ValidateAsync(Stream pdfStream)
|
||||
public async Task<Application.Common.DTOs.PdfValidationResult> ValidateAsync(Stream pdfStream)
|
||||
{
|
||||
// 1. Input Validation
|
||||
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
|
||||
@@ -73,13 +73,18 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
||||
// We scan PDF raw data for "/EmbeddedFiles" and parse the name tree to get count.
|
||||
var (hasAttachments, attachmentCount) = DetectEmbeddedFiles(pdfBytes);
|
||||
|
||||
// 5. Create and return PdfMetadata DTO
|
||||
return new Application.Common.DTOs.PdfMetadata(
|
||||
pageCount: pageCount,
|
||||
fileSizeBytes: pdfBytes.Length,
|
||||
pdfVersion: pdfVersion,
|
||||
hasAttachments: hasAttachments,
|
||||
attachmentCount: attachmentCount
|
||||
// Check encryption (scan PDF raw data for /Encrypt keyword)
|
||||
bool isEncrypted = DetectEncryption(pdfBytes);
|
||||
|
||||
// 5. Create and return PdfValidationResult
|
||||
return new Application.Common.DTOs.PdfValidationResult(
|
||||
PageCount: pageCount,
|
||||
FileSizeBytes: pdfBytes.Length,
|
||||
FileSizeMB: pdfBytes.Length / 1024.0 / 1024.0,
|
||||
PdfVersion: pdfVersion,
|
||||
HasAttachments: hasAttachments,
|
||||
AttachmentCount: attachmentCount,
|
||||
IsEncrypted: isEncrypted
|
||||
);
|
||||
}
|
||||
|
||||
@@ -89,7 +94,7 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
||||
/// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param>
|
||||
/// <returns>PDF/A metadata including conformance level and validation errors/warnings</returns>
|
||||
/// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception>
|
||||
public async Task<PdfAMetadata> ValidatePdfAAsync(Stream pdfStream)
|
||||
public async Task<PdfAValidationResult> ValidatePdfAAsync(Stream pdfStream)
|
||||
{
|
||||
// 1. Input Validation
|
||||
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
|
||||
@@ -162,18 +167,19 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
||||
// 8. Determine overall validity
|
||||
bool isValid = errors.Count == 0;
|
||||
|
||||
// 9. Create and return PdfAMetadata DTO
|
||||
return new PdfAMetadata(
|
||||
isValid: isValid,
|
||||
pdfVersion: pdfVersion,
|
||||
pageCount: pageCount,
|
||||
fileSizeBytes: pdfBytes.Length,
|
||||
encrypted: encrypted,
|
||||
pdfaVersion: pdfaVersion,
|
||||
pdfaCompliant: isPdfACompliant,
|
||||
errors: errors,
|
||||
warnings: warnings
|
||||
);
|
||||
// 9. Create and return PdfAValidationResult
|
||||
return new PdfAValidationResult
|
||||
{
|
||||
IsValid = isValid,
|
||||
PdfVersion = pdfVersion,
|
||||
PageCount = pageCount,
|
||||
FileSize = pdfBytes.Length,
|
||||
Encrypted = encrypted,
|
||||
PdfAVersion = pdfaVersion,
|
||||
PdfACompliant = isPdfACompliant,
|
||||
Errors = errors,
|
||||
Warnings = warnings
|
||||
};
|
||||
}
|
||||
|
||||
#endregion
|
||||
@@ -655,14 +661,108 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
||||
#region Private Helpers
|
||||
|
||||
/// <summary>
|
||||
/// Detects if PDF is encrypted by scanning for /Encrypt keyword.
|
||||
/// Detects if a PDF is encrypted by scanning the PDF trailer/xref section for an /Encrypt dictionary reference.
|
||||
///
|
||||
/// Why NOT use Encoding.ASCII.GetString():
|
||||
/// - Binary PDFs contain bytes 0x80–0xFF which ASCII decoding maps to '?' (0x3F), corrupting data.
|
||||
/// - A string search on corrupted data can produce both false positives and false negatives.
|
||||
///
|
||||
/// This implementation searches the raw bytes directly, which is safe for all PDF variants
|
||||
/// (binary, linearized, cross-reference stream PDFs).
|
||||
///
|
||||
/// According to the PDF specification (ISO 32000), the /Encrypt entry only appears in the
|
||||
/// document's trailer dictionary. Searching only the last ~2KB (where the trailer lives)
|
||||
/// eliminates false positives from document content that might contain the literal text "/Encrypt".
|
||||
///
|
||||
/// Limitation: Cross-reference stream PDFs (PDF 1.5+) embed the trailer inline in a compressed
|
||||
/// stream object. In those cases the raw /Encrypt bytes may not appear in the trailer section,
|
||||
/// so we fall back to searching the entire file. This is still byte-level and therefore safe.
|
||||
/// </summary>
|
||||
private static bool DetectEncryption(byte[] pdfBytes)
|
||||
{
|
||||
string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes);
|
||||
return pdfText.Contains("/Encrypt", StringComparison.Ordinal);
|
||||
// The byte sequence we are looking for (ASCII, safe because PDF keywords are always ASCII)
|
||||
ReadOnlySpan<byte> encryptKeyword = "/Encrypt"u8;
|
||||
|
||||
// Strategy 1: Search only the last 2 KB (trailer area).
|
||||
// The PDF trailer dictionary is always near the end of the file and contains
|
||||
// /Encrypt only when the document is encrypted. This avoids false positives
|
||||
// from document content.
|
||||
int trailerSearchStart = Math.Max(0, pdfBytes.Length - 2048);
|
||||
ReadOnlySpan<byte> trailerRegion = pdfBytes.AsSpan(trailerSearchStart);
|
||||
|
||||
if (ContainsByteSequence(trailerRegion, encryptKeyword))
|
||||
return true;
|
||||
|
||||
// Strategy 2: Some PDF generators (e.g. linearized PDFs) place the trailer
|
||||
// at the beginning of the file as well. Search the first 2 KB.
|
||||
int headerSearchEnd = Math.Min(2048, pdfBytes.Length);
|
||||
ReadOnlySpan<byte> headerRegion = pdfBytes.AsSpan(0, headerSearchEnd);
|
||||
|
||||
if (ContainsByteSequence(headerRegion, encryptKeyword))
|
||||
return true;
|
||||
|
||||
// Strategy 3: Full-file fallback for cross-reference stream PDFs (PDF 1.5+).
|
||||
// In these files the xref/trailer is a compressed stream object that can appear
|
||||
// anywhere. We scan the entire file but require that /Encrypt is followed by
|
||||
// a space or '<' (start of dictionary), which rules out arbitrary text matches.
|
||||
ReadOnlySpan<byte> fullFile = pdfBytes.AsSpan();
|
||||
int idx = 0;
|
||||
while (idx <= pdfBytes.Length - encryptKeyword.Length)
|
||||
{
|
||||
int found = IndexOf(fullFile.Slice(idx), encryptKeyword);
|
||||
if (found == -1)
|
||||
break;
|
||||
|
||||
int absolutePos = idx + found;
|
||||
int afterKeyword = absolutePos + encryptKeyword.Length;
|
||||
|
||||
// Verify the character after /Encrypt is a PDF structural character,
|
||||
// not part of a longer token (e.g. "/Encrypted" should not match).
|
||||
if (afterKeyword < pdfBytes.Length)
|
||||
{
|
||||
byte next = pdfBytes[afterKeyword];
|
||||
// Valid PDF token terminators: space (0x20), newline, tab, '<', '/', '['
|
||||
if (next == 0x20 || next == 0x0A || next == 0x0D || next == 0x09 ||
|
||||
next == 0x3C || next == 0x2F || next == 0x5B)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// /Encrypt is the very last token in the file
|
||||
return true;
|
||||
}
|
||||
|
||||
idx = absolutePos + 1;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Returns the index of <paramref name="needle"/> within <paramref name="haystack"/>,
|
||||
/// or -1 if not found. Operates entirely on raw bytes — no string allocation.
|
||||
/// </summary>
|
||||
private static int IndexOf(ReadOnlySpan<byte> haystack, ReadOnlySpan<byte> needle)
|
||||
{
|
||||
if (needle.IsEmpty) return 0;
|
||||
if (haystack.Length < needle.Length) return -1;
|
||||
|
||||
for (int i = 0; i <= haystack.Length - needle.Length; i++)
|
||||
{
|
||||
if (haystack.Slice(i, needle.Length).SequenceEqual(needle))
|
||||
return i;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Returns true if <paramref name="haystack"/> contains <paramref name="needle"/> as a byte subsequence.
|
||||
/// </summary>
|
||||
private static bool ContainsByteSequence(ReadOnlySpan<byte> haystack, ReadOnlySpan<byte> needle)
|
||||
=> IndexOf(haystack, needle) >= 0;
|
||||
|
||||
/// <summary>
|
||||
/// Detects PDF/A conformance level by scanning PDF metadata.
|
||||
/// PDF/A documents contain an XMP metadata stream with pdfaid:conformance and pdfaid:part.
|
||||
|
||||
Reference in New Issue
Block a user