Files
DocumentService/DocumentOperator.Infrastructure/Services/PdfProcessing/DevExpressPdfProcessor.cs
TekH cd50d45bd5 feat: Add PDF/A validation endpoint (Feature 3)
Domain layer:

  - PdfAMetadata value object (isValid, pdfVersion, pageCount, encrypted, pdfaVersion, pdfaCompliant, errors, warnings)

Infrastructure layer:

  - IPdfProcessor.ValidatePdfAAsync() interface method

  - DevExpressPdfProcessor.ValidatePdfAAsync() implementation

  - DetectEncryption() - scans PDF raw data for /Encrypt keyword

  - DetectPdfAConformance() - parses XMP metadata (pdfaid:part, pdfaid:conformance)

  - Validation: encrypted PDF cannot be PDF/A compliant

Application layer:

  - ValidatePdfAQuery + ValidatePdfAQueryHandler (co-located)

  - ValidatePdfAQueryValidator (FluentValidation: PdfBytes XOR Base64Pdf)

  - PdfAValidationResult DTO

  - AutoMapper: PdfAMetadata -> PdfAValidationResult

API layer:

  - PdfValidationController.ValidatePdfAFromFile() (multipart/form-data)

  - PdfValidationController.ValidatePdfAFromBase64() (application/json)

  - XML documentation with response codes

Result:

  - POST /api/pdf/validation/validate-pdfa (both multipart and JSON)

  - Returns: conformance level, errors, warnings

  - Build: 0 errors, 4 warnings (DevExpress eval)

  - Tests: 20/20 passing

Next: Integration tests + Swagger test case
2026-07-09 12:59:24 +02:00

256 lines
10 KiB
C#

using DevExpress.Pdf;
using DocumentOperator.Application.Common.Interfaces;
using DocumentOperator.Domain.Common.Exceptions;
using DocumentOperator.Domain.Models.ValueObjects;
namespace DocumentOperator.Infrastructure.Services.PdfProcessing;
/// <summary>
/// PDF processor implementation using DevExpress.Pdf library.
/// Handles PDF validation and metadata extraction.
/// </summary>
public class DevExpressPdfProcessor : IPdfProcessor
{
/// <summary>
/// Validates a PDF document and returns metadata.
/// </summary>
/// <param name="pdfBytes">PDF content as byte array</param>
/// <returns>PDF metadata (page count, file size, version, etc.)</returns>
/// <exception cref="PdfProcessingException">Thrown when PDF is invalid or null</exception>
public async Task<Domain.Models.ValueObjects.PdfMetadata> ValidateAsync(byte[] pdfBytes)
{
// 1. Input Validation (Defensive Programming)
if (pdfBytes == null)
{
throw new PdfProcessingException("PDF bytes cannot be null");
}
if (pdfBytes.Length == 0)
{
throw new PdfProcessingException("PDF bytes cannot be empty");
}
try
{
// 2. Load PDF with DevExpress Document API (PdfDocumentProcessor)
using var processor = new PdfDocumentProcessor();
processor.LoadDocument(new MemoryStream(pdfBytes));
// 3. Extract metadata
var document = processor.Document;
int pageCount = document.Pages.Count;
string pdfVersion = document.Version.ToString(); // z.B. "1.4", "1.7"
// Attachments (embedded files)
// DevExpress PdfDocument API doesn't expose EmbeddedFiles directly.
// We scan PDF raw data for "/EmbeddedFiles" and parse the name tree to get count.
var (hasAttachments, attachmentCount) = DetectEmbeddedFiles(pdfBytes);
// 4. Create and return PdfMetadata Value Object
return new Domain.Models.ValueObjects.PdfMetadata(
pageCount: pageCount,
fileSizeBytes: pdfBytes.Length,
pdfVersion: pdfVersion,
hasAttachments: hasAttachments,
attachmentCount: attachmentCount
);
}
catch (Exception ex) when (ex is not PdfProcessingException)
{
// Wrap DevExpress exceptions in our domain exception
throw new PdfProcessingException(
$"Failed to validate PDF: {ex.Message}",
ex);
}
}
/// <summary>
/// Validates a PDF/A document and checks conformance level.
/// </summary>
/// <param name="pdfBytes">PDF content as byte array</param>
/// <returns>PDF/A metadata including conformance level and validation errors/warnings</returns>
/// <exception cref="PdfProcessingException">Thrown when PDF is invalid or null</exception>
public async Task<PdfAMetadata> ValidatePdfAAsync(byte[] pdfBytes)
{
// 1. Input Validation
if (pdfBytes == null)
{
throw new PdfProcessingException("PDF bytes cannot be null");
}
if (pdfBytes.Length == 0)
{
throw new PdfProcessingException("PDF bytes cannot be empty");
}
try
{
// 2. Load PDF with DevExpress Document API
using var processor = new PdfDocumentProcessor();
processor.LoadDocument(new MemoryStream(pdfBytes));
var document = processor.Document;
// 3. Extract basic metadata
int pageCount = document.Pages.Count;
string pdfVersion = document.Version.ToString();
// 4. Check encryption (scan PDF raw data for /Encrypt keyword)
bool encrypted = DetectEncryption(pdfBytes);
// 5. Check PDF/A conformance (scan PDF raw data for PDF/A identifier)
var (isPdfACompliant, pdfaVersion) = DetectPdfAConformance(pdfBytes);
// 6. Collect errors and warnings
var errors = new List<string>();
var warnings = new List<string>();
// If encrypted, PDF/A compliance is not possible
if (encrypted && isPdfACompliant)
{
errors.Add("PDF/A documents cannot be encrypted");
isPdfACompliant = false;
}
// Basic PDF/A validation checks
if (isPdfACompliant)
{
// Add generic warning for manual verification
warnings.Add("Manual verification recommended: All fonts must be embedded");
warnings.Add("Manual verification recommended: No JavaScript or multimedia content");
}
// 7. Determine overall validity
bool isValid = errors.Count == 0;
// 8. Create and return PdfAMetadata
return new PdfAMetadata(
isValid: isValid,
pdfVersion: pdfVersion,
pageCount: pageCount,
fileSizeBytes: pdfBytes.Length,
encrypted: encrypted,
pdfaVersion: pdfaVersion,
pdfaCompliant: isPdfACompliant,
errors: errors,
warnings: warnings
);
}
catch (Exception ex) when (ex is not PdfProcessingException)
{
throw new PdfProcessingException(
$"Failed to validate PDF/A: {ex.Message}",
ex);
}
}
/// <summary>
/// Detects if PDF is encrypted by scanning for /Encrypt keyword.
/// </summary>
private static bool DetectEncryption(byte[] pdfBytes)
{
string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes);
return pdfText.Contains("/Encrypt", StringComparison.Ordinal);
}
/// <summary>
/// Detects PDF/A conformance level by scanning PDF metadata.
/// PDF/A documents contain an XMP metadata stream with pdfaid:conformance and pdfaid:part.
/// </summary>
private static (bool isPdfACompliant, string? pdfaVersion) DetectPdfAConformance(byte[] pdfBytes)
{
string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes);
// Look for PDF/A identifier in XMP metadata
// Example: <pdfaid:part>1</pdfaid:part><pdfaid:conformance>B</pdfaid:conformance>
if (pdfText.Contains("pdfaid:part", StringComparison.Ordinal))
{
// Try to extract part and conformance level
var partMatch = System.Text.RegularExpressions.Regex.Match(pdfText, @"pdfaid:part>(\d+)</pdfaid:part");
var conformanceMatch = System.Text.RegularExpressions.Regex.Match(pdfText, @"pdfaid:conformance>([ABU])</pdfaid:conformance");
if (partMatch.Success && conformanceMatch.Success)
{
string part = partMatch.Groups[1].Value; // "1", "2", "3"
string conformance = conformanceMatch.Groups[1].Value; // "A", "B", "U"
string pdfaVersion = $"PDF/A-{part}{conformance.ToLower()}";
return (true, pdfaVersion);
}
// Found pdfaid:part but couldn't parse details
return (true, "PDF/A (unknown level)");
}
return (false, null);
}
/// <summary>
/// Detects embedded files in PDF by scanning raw PDF data for /EmbeddedFiles keyword
/// and parsing the name tree to count attachments.
/// This is a pragmatic approach as DevExpress PdfDocument API doesn't expose EmbeddedFiles directly.
/// </summary>
/// <param name="pdfBytes">PDF raw bytes</param>
/// <returns>Tuple: (hasAttachments, attachmentCount)</returns>
private static (bool hasAttachments, int attachmentCount) DetectEmbeddedFiles(byte[] pdfBytes)
{
// PDF embedded files are declared in the document catalog:
// /Names << /EmbeddedFiles << /Names [...] >> >>
// The /Names array contains pairs: [name1, filespec1, name2, filespec2, ...]
string pdfText = System.Text.Encoding.ASCII.GetString(pdfBytes);
// Search for /EmbeddedFiles in the context of /Names dictionary
// Must appear after a /Names keyword to be valid
int searchStart = 0;
while (true)
{
// Find next occurrence of /EmbeddedFiles
int embeddedFilesIndex = pdfText.IndexOf("/EmbeddedFiles", searchStart, StringComparison.Ordinal);
if (embeddedFilesIndex == -1)
return (false, 0); // Not found
// Check if there's a /Names keyword BEFORE this /EmbeddedFiles
// within a reasonable distance (e.g., within the same PDF object, max 5000 chars back)
int contextStart = Math.Max(0, embeddedFilesIndex - 5000);
string contextBefore = pdfText.Substring(contextStart, embeddedFilesIndex - contextStart);
// Look for /Names in the context before /EmbeddedFiles
int lastNamesIndex = contextBefore.LastIndexOf("/Names", StringComparison.Ordinal);
if (lastNamesIndex != -1)
{
// Found /Names before /EmbeddedFiles - this is likely a valid embedded files declaration
// Now try to parse the /Names array
int namesArrayStart = pdfText.IndexOf("/Names", embeddedFilesIndex, StringComparison.Ordinal);
if (namesArrayStart == -1)
return (true, 0); // Has EmbeddedFiles but can't count
int arrayStart = pdfText.IndexOf('[', namesArrayStart);
if (arrayStart == -1)
return (true, 0); // Has EmbeddedFiles but can't count
int arrayEnd = pdfText.IndexOf(']', arrayStart);
if (arrayEnd == -1)
return (true, 0); // Has EmbeddedFiles but can't count
// Extract array content and count entries
string arrayContent = pdfText.Substring(arrayStart + 1, arrayEnd - arrayStart - 1);
// Count object references in array
// The /Names array contains pairs: (filename) objectReference (filename) objectReference ...
// Each object reference (pattern: "123 0 R") points to one embedded file
// So the number of object references = number of attachments
int objectCount = System.Text.RegularExpressions.Regex.Matches(arrayContent, @"\d+ \d+ R").Count;
return (true, Math.Max(1, objectCount)); // At least 1 if EmbeddedFiles found
}
// This /EmbeddedFiles was not in the right context, search for next occurrence
searchStart = embeddedFilesIndex + 1;
}
}
}