fix: replace ASCII-based DetectEncryption with byte-level trailer scan
Old implementation used Encoding.ASCII.GetString() which: - Corrupts binary PDF bytes 0x80-0xFF -> '?' (false negatives) - Matches '/Encrypt' anywhere in document content (false positives) - Matches '/Encrypted', '/EncryptionKey' etc. (token boundary not checked) New implementation uses ReadOnlySpan<byte> with three strategies: 1. Search last 2KB (trailer region) - fast path, covers standard PDFs 2. Search first 2KB (linearized PDFs have duplicate trailer at start) 3. Full-file fallback for PDF 1.5+ compressed xref streams, with token boundary check (next byte must be space/newline/tab/'<'/'/'/'[') to avoid matching '/Encrypted' or '/EncryptionKey' Also: DevExpressPdfProcessor.ValidateAsync now returns PdfValidationResult directly (no more PdfMetadata wrapper), includes IsEncrypted field.
This commit is contained in:
@@ -20,7 +20,7 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
|||||||
/// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param>
|
/// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param>
|
||||||
/// <returns>PDF metadata (page count, file size, version, etc.)</returns>
|
/// <returns>PDF metadata (page count, file size, version, etc.)</returns>
|
||||||
/// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception>
|
/// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception>
|
||||||
public async Task<Application.Common.DTOs.PdfMetadata> ValidateAsync(Stream pdfStream)
|
public async Task<Application.Common.DTOs.PdfValidationResult> ValidateAsync(Stream pdfStream)
|
||||||
{
|
{
|
||||||
// 1. Input Validation
|
// 1. Input Validation
|
||||||
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
|
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
|
||||||
@@ -73,13 +73,18 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
|||||||
// We scan PDF raw data for "/EmbeddedFiles" and parse the name tree to get count.
|
// We scan PDF raw data for "/EmbeddedFiles" and parse the name tree to get count.
|
||||||
var (hasAttachments, attachmentCount) = DetectEmbeddedFiles(pdfBytes);
|
var (hasAttachments, attachmentCount) = DetectEmbeddedFiles(pdfBytes);
|
||||||
|
|
||||||
// 5. Create and return PdfMetadata DTO
|
// Check encryption (scan PDF raw data for /Encrypt keyword)
|
||||||
return new Application.Common.DTOs.PdfMetadata(
|
bool isEncrypted = DetectEncryption(pdfBytes);
|
||||||
pageCount: pageCount,
|
|
||||||
fileSizeBytes: pdfBytes.Length,
|
// 5. Create and return PdfValidationResult
|
||||||
pdfVersion: pdfVersion,
|
return new Application.Common.DTOs.PdfValidationResult(
|
||||||
hasAttachments: hasAttachments,
|
PageCount: pageCount,
|
||||||
attachmentCount: attachmentCount
|
FileSizeBytes: pdfBytes.Length,
|
||||||
|
FileSizeMB: pdfBytes.Length / 1024.0 / 1024.0,
|
||||||
|
PdfVersion: pdfVersion,
|
||||||
|
HasAttachments: hasAttachments,
|
||||||
|
AttachmentCount: attachmentCount,
|
||||||
|
IsEncrypted: isEncrypted
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -89,7 +94,7 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
|||||||
/// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param>
|
/// <param name="pdfStream">PDF content as stream (caller is responsible for disposal)</param>
|
||||||
/// <returns>PDF/A metadata including conformance level and validation errors/warnings</returns>
|
/// <returns>PDF/A metadata including conformance level and validation errors/warnings</returns>
|
||||||
/// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception>
|
/// <exception cref="BadRequestException">Thrown when stream is empty or invalid</exception>
|
||||||
public async Task<PdfAMetadata> ValidatePdfAAsync(Stream pdfStream)
|
public async Task<PdfAValidationResult> ValidatePdfAAsync(Stream pdfStream)
|
||||||
{
|
{
|
||||||
// 1. Input Validation
|
// 1. Input Validation
|
||||||
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
|
ArgumentNullException.ThrowIfNull(pdfStream, nameof(pdfStream));
|
||||||
@@ -162,18 +167,19 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
|||||||
// 8. Determine overall validity
|
// 8. Determine overall validity
|
||||||
bool isValid = errors.Count == 0;
|
bool isValid = errors.Count == 0;
|
||||||
|
|
||||||
// 9. Create and return PdfAMetadata DTO
|
// 9. Create and return PdfAValidationResult
|
||||||
return new PdfAMetadata(
|
return new PdfAValidationResult
|
||||||
isValid: isValid,
|
{
|
||||||
pdfVersion: pdfVersion,
|
IsValid = isValid,
|
||||||
pageCount: pageCount,
|
PdfVersion = pdfVersion,
|
||||||
fileSizeBytes: pdfBytes.Length,
|
PageCount = pageCount,
|
||||||
encrypted: encrypted,
|
FileSize = pdfBytes.Length,
|
||||||
pdfaVersion: pdfaVersion,
|
Encrypted = encrypted,
|
||||||
pdfaCompliant: isPdfACompliant,
|
PdfAVersion = pdfaVersion,
|
||||||
errors: errors,
|
PdfACompliant = isPdfACompliant,
|
||||||
warnings: warnings
|
Errors = errors,
|
||||||
);
|
Warnings = warnings
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
#endregion
|
#endregion
|
||||||
@@ -655,14 +661,108 @@ public class DevExpressPdfProcessor : IPdfProcessor
|
|||||||
#region Private Helpers
|
#region Private Helpers
|
||||||
|
|
||||||
/// <summary>
|
/// <summary>
|
||||||
/// Detects if PDF is encrypted by scanning for /Encrypt keyword.
|
/// Detects if a PDF is encrypted by scanning the PDF trailer/xref section for an /Encrypt dictionary reference.
|
||||||
|
///
|
||||||
|
/// Why NOT use Encoding.ASCII.GetString():
|
||||||
|
/// - Binary PDFs contain bytes 0x80–0xFF which ASCII decoding maps to '?' (0x3F), corrupting data.
|
||||||
|
/// - A string search on corrupted data can produce both false positives and false negatives.
|
||||||
|
///
|
||||||
|
/// This implementation searches the raw bytes directly, which is safe for all PDF variants
|
||||||
|
/// (binary, linearized, cross-reference stream PDFs).
|
||||||
|
///
|
||||||
|
/// According to the PDF specification (ISO 32000), the /Encrypt entry only appears in the
|
||||||
|
/// document's trailer dictionary. Searching only the last ~2KB (where the trailer lives)
|
||||||
|
/// eliminates false positives from document content that might contain the literal text "/Encrypt".
|
||||||
|
///
|
||||||
|
/// Limitation: Cross-reference stream PDFs (PDF 1.5+) embed the trailer inline in a compressed
|
||||||
|
/// stream object. In those cases the raw /Encrypt bytes may not appear in the trailer section,
|
||||||
|
/// so we fall back to searching the entire file. This is still byte-level and therefore safe.
|
||||||
/// </summary>
|
/// </summary>
|
||||||
private static bool DetectEncryption(byte[] pdfBytes)
|
private static bool DetectEncryption(byte[] pdfBytes)
|
||||||
{
|
{
|
||||||
string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes);
|
// The byte sequence we are looking for (ASCII, safe because PDF keywords are always ASCII)
|
||||||
return pdfText.Contains("/Encrypt", StringComparison.Ordinal);
|
ReadOnlySpan<byte> encryptKeyword = "/Encrypt"u8;
|
||||||
|
|
||||||
|
// Strategy 1: Search only the last 2 KB (trailer area).
|
||||||
|
// The PDF trailer dictionary is always near the end of the file and contains
|
||||||
|
// /Encrypt only when the document is encrypted. This avoids false positives
|
||||||
|
// from document content.
|
||||||
|
int trailerSearchStart = Math.Max(0, pdfBytes.Length - 2048);
|
||||||
|
ReadOnlySpan<byte> trailerRegion = pdfBytes.AsSpan(trailerSearchStart);
|
||||||
|
|
||||||
|
if (ContainsByteSequence(trailerRegion, encryptKeyword))
|
||||||
|
return true;
|
||||||
|
|
||||||
|
// Strategy 2: Some PDF generators (e.g. linearized PDFs) place the trailer
|
||||||
|
// at the beginning of the file as well. Search the first 2 KB.
|
||||||
|
int headerSearchEnd = Math.Min(2048, pdfBytes.Length);
|
||||||
|
ReadOnlySpan<byte> headerRegion = pdfBytes.AsSpan(0, headerSearchEnd);
|
||||||
|
|
||||||
|
if (ContainsByteSequence(headerRegion, encryptKeyword))
|
||||||
|
return true;
|
||||||
|
|
||||||
|
// Strategy 3: Full-file fallback for cross-reference stream PDFs (PDF 1.5+).
|
||||||
|
// In these files the xref/trailer is a compressed stream object that can appear
|
||||||
|
// anywhere. We scan the entire file but require that /Encrypt is followed by
|
||||||
|
// a space or '<' (start of dictionary), which rules out arbitrary text matches.
|
||||||
|
ReadOnlySpan<byte> fullFile = pdfBytes.AsSpan();
|
||||||
|
int idx = 0;
|
||||||
|
while (idx <= pdfBytes.Length - encryptKeyword.Length)
|
||||||
|
{
|
||||||
|
int found = IndexOf(fullFile.Slice(idx), encryptKeyword);
|
||||||
|
if (found == -1)
|
||||||
|
break;
|
||||||
|
|
||||||
|
int absolutePos = idx + found;
|
||||||
|
int afterKeyword = absolutePos + encryptKeyword.Length;
|
||||||
|
|
||||||
|
// Verify the character after /Encrypt is a PDF structural character,
|
||||||
|
// not part of a longer token (e.g. "/Encrypted" should not match).
|
||||||
|
if (afterKeyword < pdfBytes.Length)
|
||||||
|
{
|
||||||
|
byte next = pdfBytes[afterKeyword];
|
||||||
|
// Valid PDF token terminators: space (0x20), newline, tab, '<', '/', '['
|
||||||
|
if (next == 0x20 || next == 0x0A || next == 0x0D || next == 0x09 ||
|
||||||
|
next == 0x3C || next == 0x2F || next == 0x5B)
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// /Encrypt is the very last token in the file
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
idx = absolutePos + 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Returns the index of <paramref name="needle"/> within <paramref name="haystack"/>,
|
||||||
|
/// or -1 if not found. Operates entirely on raw bytes — no string allocation.
|
||||||
|
/// </summary>
|
||||||
|
private static int IndexOf(ReadOnlySpan<byte> haystack, ReadOnlySpan<byte> needle)
|
||||||
|
{
|
||||||
|
if (needle.IsEmpty) return 0;
|
||||||
|
if (haystack.Length < needle.Length) return -1;
|
||||||
|
|
||||||
|
for (int i = 0; i <= haystack.Length - needle.Length; i++)
|
||||||
|
{
|
||||||
|
if (haystack.Slice(i, needle.Length).SequenceEqual(needle))
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// <summary>
|
||||||
|
/// Returns true if <paramref name="haystack"/> contains <paramref name="needle"/> as a byte subsequence.
|
||||||
|
/// </summary>
|
||||||
|
private static bool ContainsByteSequence(ReadOnlySpan<byte> haystack, ReadOnlySpan<byte> needle)
|
||||||
|
=> IndexOf(haystack, needle) >= 0;
|
||||||
|
|
||||||
/// <summary>
|
/// <summary>
|
||||||
/// Detects PDF/A conformance level by scanning PDF metadata.
|
/// Detects PDF/A conformance level by scanning PDF metadata.
|
||||||
/// PDF/A documents contain an XMP metadata stream with pdfaid:conformance and pdfaid:part.
|
/// PDF/A documents contain an XMP metadata stream with pdfaid:conformance and pdfaid:part.
|
||||||
|
|||||||
Reference in New Issue
Block a user