Domain layer: - PdfAMetadata value object (isValid, pdfVersion, pageCount, encrypted, pdfaVersion, pdfaCompliant, errors, warnings) Infrastructure layer: - IPdfProcessor.ValidatePdfAAsync() interface method - DevExpressPdfProcessor.ValidatePdfAAsync() implementation - DetectEncryption() - scans PDF raw data for /Encrypt keyword - DetectPdfAConformance() - parses XMP metadata (pdfaid:part, pdfaid:conformance) - Validation: encrypted PDF cannot be PDF/A compliant Application layer: - ValidatePdfAQuery + ValidatePdfAQueryHandler (co-located) - ValidatePdfAQueryValidator (FluentValidation: PdfBytes XOR Base64Pdf) - PdfAValidationResult DTO - AutoMapper: PdfAMetadata -> PdfAValidationResult API layer: - PdfValidationController.ValidatePdfAFromFile() (multipart/form-data) - PdfValidationController.ValidatePdfAFromBase64() (application/json) - XML documentation with response codes Result: - POST /api/pdf/validation/validate-pdfa (both multipart and JSON) - Returns: conformance level, errors, warnings - Build: 0 errors, 4 warnings (DevExpress eval) - Tests: 20/20 passing Next: Integration tests + Swagger test case
256 lines
10 KiB
C#
256 lines
10 KiB
C#
using DevExpress.Pdf;
|
|
using DocumentOperator.Application.Common.Interfaces;
|
|
using DocumentOperator.Domain.Common.Exceptions;
|
|
using DocumentOperator.Domain.Models.ValueObjects;
|
|
|
|
namespace DocumentOperator.Infrastructure.Services.PdfProcessing;
|
|
|
|
/// <summary>
|
|
/// PDF processor implementation using DevExpress.Pdf library.
|
|
/// Handles PDF validation and metadata extraction.
|
|
/// </summary>
|
|
public class DevExpressPdfProcessor : IPdfProcessor
|
|
{
|
|
/// <summary>
|
|
/// Validates a PDF document and returns metadata.
|
|
/// </summary>
|
|
/// <param name="pdfBytes">PDF content as byte array</param>
|
|
/// <returns>PDF metadata (page count, file size, version, etc.)</returns>
|
|
/// <exception cref="PdfProcessingException">Thrown when PDF is invalid or null</exception>
|
|
public async Task<Domain.Models.ValueObjects.PdfMetadata> ValidateAsync(byte[] pdfBytes)
|
|
{
|
|
// 1. Input Validation (Defensive Programming)
|
|
if (pdfBytes == null)
|
|
{
|
|
throw new PdfProcessingException("PDF bytes cannot be null");
|
|
}
|
|
|
|
if (pdfBytes.Length == 0)
|
|
{
|
|
throw new PdfProcessingException("PDF bytes cannot be empty");
|
|
}
|
|
|
|
try
|
|
{
|
|
// 2. Load PDF with DevExpress Document API (PdfDocumentProcessor)
|
|
using var processor = new PdfDocumentProcessor();
|
|
processor.LoadDocument(new MemoryStream(pdfBytes));
|
|
|
|
// 3. Extract metadata
|
|
var document = processor.Document;
|
|
|
|
int pageCount = document.Pages.Count;
|
|
string pdfVersion = document.Version.ToString(); // z.B. "1.4", "1.7"
|
|
|
|
// Attachments (embedded files)
|
|
// DevExpress PdfDocument API doesn't expose EmbeddedFiles directly.
|
|
// We scan PDF raw data for "/EmbeddedFiles" and parse the name tree to get count.
|
|
var (hasAttachments, attachmentCount) = DetectEmbeddedFiles(pdfBytes);
|
|
|
|
// 4. Create and return PdfMetadata Value Object
|
|
return new Domain.Models.ValueObjects.PdfMetadata(
|
|
pageCount: pageCount,
|
|
fileSizeBytes: pdfBytes.Length,
|
|
pdfVersion: pdfVersion,
|
|
hasAttachments: hasAttachments,
|
|
attachmentCount: attachmentCount
|
|
);
|
|
}
|
|
catch (Exception ex) when (ex is not PdfProcessingException)
|
|
{
|
|
// Wrap DevExpress exceptions in our domain exception
|
|
throw new PdfProcessingException(
|
|
$"Failed to validate PDF: {ex.Message}",
|
|
ex);
|
|
}
|
|
}
|
|
|
|
/// <summary>
|
|
/// Validates a PDF/A document and checks conformance level.
|
|
/// </summary>
|
|
/// <param name="pdfBytes">PDF content as byte array</param>
|
|
/// <returns>PDF/A metadata including conformance level and validation errors/warnings</returns>
|
|
/// <exception cref="PdfProcessingException">Thrown when PDF is invalid or null</exception>
|
|
public async Task<PdfAMetadata> ValidatePdfAAsync(byte[] pdfBytes)
|
|
{
|
|
// 1. Input Validation
|
|
if (pdfBytes == null)
|
|
{
|
|
throw new PdfProcessingException("PDF bytes cannot be null");
|
|
}
|
|
|
|
if (pdfBytes.Length == 0)
|
|
{
|
|
throw new PdfProcessingException("PDF bytes cannot be empty");
|
|
}
|
|
|
|
try
|
|
{
|
|
// 2. Load PDF with DevExpress Document API
|
|
using var processor = new PdfDocumentProcessor();
|
|
processor.LoadDocument(new MemoryStream(pdfBytes));
|
|
|
|
var document = processor.Document;
|
|
|
|
// 3. Extract basic metadata
|
|
int pageCount = document.Pages.Count;
|
|
string pdfVersion = document.Version.ToString();
|
|
|
|
// 4. Check encryption (scan PDF raw data for /Encrypt keyword)
|
|
bool encrypted = DetectEncryption(pdfBytes);
|
|
|
|
// 5. Check PDF/A conformance (scan PDF raw data for PDF/A identifier)
|
|
var (isPdfACompliant, pdfaVersion) = DetectPdfAConformance(pdfBytes);
|
|
|
|
// 6. Collect errors and warnings
|
|
var errors = new List<string>();
|
|
var warnings = new List<string>();
|
|
|
|
// If encrypted, PDF/A compliance is not possible
|
|
if (encrypted && isPdfACompliant)
|
|
{
|
|
errors.Add("PDF/A documents cannot be encrypted");
|
|
isPdfACompliant = false;
|
|
}
|
|
|
|
// Basic PDF/A validation checks
|
|
if (isPdfACompliant)
|
|
{
|
|
// Add generic warning for manual verification
|
|
warnings.Add("Manual verification recommended: All fonts must be embedded");
|
|
warnings.Add("Manual verification recommended: No JavaScript or multimedia content");
|
|
}
|
|
|
|
// 7. Determine overall validity
|
|
bool isValid = errors.Count == 0;
|
|
|
|
// 8. Create and return PdfAMetadata
|
|
return new PdfAMetadata(
|
|
isValid: isValid,
|
|
pdfVersion: pdfVersion,
|
|
pageCount: pageCount,
|
|
fileSizeBytes: pdfBytes.Length,
|
|
encrypted: encrypted,
|
|
pdfaVersion: pdfaVersion,
|
|
pdfaCompliant: isPdfACompliant,
|
|
errors: errors,
|
|
warnings: warnings
|
|
);
|
|
}
|
|
catch (Exception ex) when (ex is not PdfProcessingException)
|
|
{
|
|
throw new PdfProcessingException(
|
|
$"Failed to validate PDF/A: {ex.Message}",
|
|
ex);
|
|
}
|
|
}
|
|
|
|
/// <summary>
|
|
/// Detects if PDF is encrypted by scanning for /Encrypt keyword.
|
|
/// </summary>
|
|
private static bool DetectEncryption(byte[] pdfBytes)
|
|
{
|
|
string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes);
|
|
return pdfText.Contains("/Encrypt", StringComparison.Ordinal);
|
|
}
|
|
|
|
/// <summary>
|
|
/// Detects PDF/A conformance level by scanning PDF metadata.
|
|
/// PDF/A documents contain an XMP metadata stream with pdfaid:conformance and pdfaid:part.
|
|
/// </summary>
|
|
private static (bool isPdfACompliant, string? pdfaVersion) DetectPdfAConformance(byte[] pdfBytes)
|
|
{
|
|
string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes);
|
|
|
|
// Look for PDF/A identifier in XMP metadata
|
|
// Example: <pdfaid:part>1</pdfaid:part><pdfaid:conformance>B</pdfaid:conformance>
|
|
if (pdfText.Contains("pdfaid:part", StringComparison.Ordinal))
|
|
{
|
|
// Try to extract part and conformance level
|
|
var partMatch = System.Text.RegularExpressions.Regex.Match(pdfText, @"pdfaid:part>(\d+)</pdfaid:part");
|
|
var conformanceMatch = System.Text.RegularExpressions.Regex.Match(pdfText, @"pdfaid:conformance>([ABU])</pdfaid:conformance");
|
|
|
|
if (partMatch.Success && conformanceMatch.Success)
|
|
{
|
|
string part = partMatch.Groups[1].Value; // "1", "2", "3"
|
|
string conformance = conformanceMatch.Groups[1].Value; // "A", "B", "U"
|
|
string pdfaVersion = $"PDF/A-{part}{conformance.ToLower()}";
|
|
return (true, pdfaVersion);
|
|
}
|
|
|
|
// Found pdfaid:part but couldn't parse details
|
|
return (true, "PDF/A (unknown level)");
|
|
}
|
|
|
|
return (false, null);
|
|
}
|
|
|
|
/// <summary>
|
|
/// Detects embedded files in PDF by scanning raw PDF data for /EmbeddedFiles keyword
|
|
/// and parsing the name tree to count attachments.
|
|
/// This is a pragmatic approach as DevExpress PdfDocument API doesn't expose EmbeddedFiles directly.
|
|
/// </summary>
|
|
/// <param name="pdfBytes">PDF raw bytes</param>
|
|
/// <returns>Tuple: (hasAttachments, attachmentCount)</returns>
|
|
private static (bool hasAttachments, int attachmentCount) DetectEmbeddedFiles(byte[] pdfBytes)
|
|
{
|
|
// PDF embedded files are declared in the document catalog:
|
|
// /Names << /EmbeddedFiles << /Names [...] >> >>
|
|
// The /Names array contains pairs: [name1, filespec1, name2, filespec2, ...]
|
|
|
|
string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes);
|
|
|
|
// Search for /EmbeddedFiles in the context of /Names dictionary
|
|
// Must appear after a /Names keyword to be valid
|
|
int searchStart = 0;
|
|
|
|
while (true)
|
|
{
|
|
// Find next occurrence of /EmbeddedFiles
|
|
int embeddedFilesIndex = pdfText.IndexOf("/EmbeddedFiles", searchStart, StringComparison.Ordinal);
|
|
|
|
if (embeddedFilesIndex == -1)
|
|
return (false, 0); // Not found
|
|
|
|
// Check if there's a /Names keyword BEFORE this /EmbeddedFiles
|
|
// within a reasonable distance (e.g., within the same PDF object, max 5000 chars back)
|
|
int contextStart = Math.Max(0, embeddedFilesIndex - 5000);
|
|
string contextBefore = pdfText.Substring(contextStart, embeddedFilesIndex - contextStart);
|
|
|
|
// Look for /Names in the context before /EmbeddedFiles
|
|
int lastNamesIndex = contextBefore.LastIndexOf("/Names", StringComparison.Ordinal);
|
|
|
|
if (lastNamesIndex != -1)
|
|
{
|
|
// Found /Names before /EmbeddedFiles - this is likely a valid embedded files declaration
|
|
// Now try to parse the /Names array
|
|
int namesArrayStart = pdfText.IndexOf("/Names", embeddedFilesIndex, StringComparison.Ordinal);
|
|
if (namesArrayStart == -1)
|
|
return (true, 0); // Has EmbeddedFiles but can't count
|
|
|
|
int arrayStart = pdfText.IndexOf('[', namesArrayStart);
|
|
if (arrayStart == -1)
|
|
return (true, 0); // Has EmbeddedFiles but can't count
|
|
|
|
int arrayEnd = pdfText.IndexOf(']', arrayStart);
|
|
if (arrayEnd == -1)
|
|
return (true, 0); // Has EmbeddedFiles but can't count
|
|
|
|
// Extract array content and count entries
|
|
string arrayContent = pdfText.Substring(arrayStart + 1, arrayEnd - arrayStart - 1);
|
|
|
|
// Count object references in array
|
|
// The /Names array contains pairs: (filename) objectReference (filename) objectReference ...
|
|
// Each object reference (pattern: "123 0 R") points to one embedded file
|
|
// So the number of object references = number of attachments
|
|
int objectCount = System.Text.RegularExpressions.Regex.Matches(arrayContent, @"\d+ \d+ R").Count;
|
|
|
|
return (true, Math.Max(1, objectCount)); // At least 1 if EmbeddedFiles found
|
|
}
|
|
|
|
// This /EmbeddedFiles was not in the right context, search for next occurrence
|
|
searchStart = embeddedFilesIndex + 1;
|
|
}
|
|
}
|
|
}
|