293 lines
9.1 KiB
C#
293 lines
9.1 KiB
C#
using System;
|
|
using System.IO;
|
|
using System.Linq;
|
|
using System.Text;
|
|
using PdfiumViewer;
|
|
using System.Threading;
|
|
using xAiModels.Models;
|
|
using System.Threading.Tasks;
|
|
using xAiApi.Interfaces.Extractors;
|
|
|
|
namespace xAiApi.Providers.Extractors
|
|
{
|
|
/// <summary>
|
|
/// Extracts content from PDF files using PdfPig ...
|
|
/// </summary>
|
|
public class XPdfFileContentExtractor : IXPdfFileContentExtractor
|
|
{
|
|
/// <summary>
|
|
/// Supported MIME Types ...
|
|
/// </summary>
|
|
private static readonly string[] SupportedMimeTypes =
|
|
[
|
|
"application/pdf"
|
|
];
|
|
|
|
/// <summary>
|
|
/// Minimum Length for Fallback Using Images ...
|
|
/// </summary>
|
|
private const int MinTextLengthThreshold = 50;
|
|
|
|
/// <summary>
|
|
/// Max Pages for Falling Back Using Images ...
|
|
/// </summary>
|
|
private const int MaxPagesForImageFallback = 10;
|
|
|
|
/// <summary>
|
|
/// Default DPI for Rendering Images ...
|
|
/// </summary>
|
|
private const int RenderDpi = 150;
|
|
|
|
/// <summary>
|
|
/// Max Rendered Image Width ...
|
|
/// </summary>
|
|
private const int MaxImageWidth = 2000;
|
|
|
|
/// <summary>
|
|
/// Check if this extractor supports the specified MIME type ...
|
|
/// </summary>
|
|
public bool CanExtract(string mimeType)
|
|
{
|
|
//
|
|
return SupportedMimeTypes.Contains(
|
|
mimeType?.ToLowerInvariant() ?? string.Empty
|
|
);
|
|
}
|
|
|
|
/// <summary>
|
|
/// Extract text content from file stream ...
|
|
/// </summary>
|
|
public async Task<string> ExtractAsync(
|
|
Stream fileStream,
|
|
string mimeType,
|
|
CancellationToken cancellationToken = default
|
|
)
|
|
{
|
|
//
|
|
var result = await ExtractRichAsync(
|
|
fileStream,
|
|
"document.pdf",
|
|
mimeType,
|
|
cancellationToken
|
|
);
|
|
|
|
//
|
|
return result.Text;
|
|
}
|
|
|
|
/// <summary>
|
|
/// Extract content from stream as Rich Result ...
|
|
/// </summary>
|
|
/// <param name="fileStream"></param>
|
|
/// <param name="fileName"></param>
|
|
/// <param name="mimeType"></param>
|
|
/// <param name="cancellationToken"></param>
|
|
/// <returns></returns>
|
|
public async Task<XFileExtractionResult> ExtractRichAsync(
|
|
Stream fileStream,
|
|
string fileName,
|
|
string mimeType,
|
|
CancellationToken cancellationToken = default
|
|
)
|
|
{
|
|
//
|
|
var result = new XFileExtractionResult
|
|
{
|
|
FileName = fileName,
|
|
MimeType = mimeType
|
|
};
|
|
|
|
//
|
|
using var memoryStream = new MemoryStream();
|
|
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
|
memoryStream.Position = 0;
|
|
|
|
//
|
|
var extractedText = await ExtractTextAsync(
|
|
memoryStream,
|
|
cancellationToken
|
|
);
|
|
if (IsTextSufficient(extractedText))
|
|
{
|
|
// PDF متنی است
|
|
result.Text = extractedText;
|
|
}
|
|
else
|
|
{
|
|
// PDF اسکنشده است - Fallback به تصویر
|
|
memoryStream.Position = 0;
|
|
result = await FallbackToImageExtractionAsync(
|
|
memoryStream,
|
|
fileName,
|
|
mimeType,
|
|
cancellationToken
|
|
);
|
|
result.UsedImageFallback = true;
|
|
}
|
|
|
|
//
|
|
return result;
|
|
}
|
|
|
|
/// <summary>
|
|
/// Extract Text From PDF Files using Common way ...
|
|
/// </summary>
|
|
/// <param name="pdfStream"></param>
|
|
/// <param name="cancellationToken"></param>
|
|
/// <returns></returns>
|
|
private async Task<string> ExtractTextAsync(
|
|
Stream pdfStream,
|
|
CancellationToken cancellationToken
|
|
)
|
|
{
|
|
//
|
|
var result = await Task.Run(() =>
|
|
{
|
|
//
|
|
var sb = new StringBuilder();
|
|
using var document = UglyToad.PdfPig.PdfDocument.Open(pdfStream);
|
|
foreach (var page in document.GetPages())
|
|
{
|
|
//
|
|
var text = page.Text?.Trim();
|
|
if (!string.IsNullOrEmpty(text))
|
|
{
|
|
//
|
|
sb.AppendLine(text);
|
|
sb.AppendLine();
|
|
}
|
|
}
|
|
|
|
//
|
|
return sb.ToString();
|
|
}, cancellationToken);
|
|
|
|
//
|
|
return result;
|
|
}
|
|
|
|
/// <summary>
|
|
/// Check Extracted Content is Sufficient or not ...
|
|
/// if not try to Falling Back using Images ...
|
|
/// </summary>
|
|
private bool IsTextSufficient(string text)
|
|
{
|
|
//
|
|
if (string.IsNullOrWhiteSpace(text))
|
|
{
|
|
return false;
|
|
}
|
|
|
|
//
|
|
// حذف فاصلهها و بررسی طول
|
|
var cleanText = new string(text.Where(c => !char.IsWhiteSpace(c)).ToArray());
|
|
|
|
//
|
|
return cleanText.Length >= MinTextLengthThreshold;
|
|
}
|
|
|
|
/// <summary>
|
|
/// Fallback Converts PDF Pages to Images ...
|
|
/// </summary>
|
|
private async Task<XFileExtractionResult> FallbackToImageExtractionAsync(
|
|
Stream pdfStream,
|
|
string fileName,
|
|
string mimeType,
|
|
CancellationToken cancellationToken
|
|
)
|
|
{
|
|
//
|
|
var result = new XFileExtractionResult
|
|
{
|
|
FileName = fileName,
|
|
MimeType = mimeType,
|
|
UsedImageFallback = true
|
|
};
|
|
|
|
//
|
|
await Task.Run(() =>
|
|
{
|
|
//
|
|
try
|
|
{
|
|
//
|
|
using var document = PdfDocument.Load(pdfStream);
|
|
|
|
//
|
|
var pageCount = Math.Min(
|
|
document.PageCount,
|
|
MaxPagesForImageFallback
|
|
);
|
|
|
|
//
|
|
for (int pageIndex = 0; pageIndex < pageCount; pageIndex++)
|
|
{
|
|
//
|
|
if (cancellationToken.IsCancellationRequested)
|
|
{
|
|
break;
|
|
}
|
|
|
|
//
|
|
try
|
|
{
|
|
//
|
|
var pageSize = document.PageSizes[pageIndex];
|
|
|
|
//
|
|
var scaleFactor = RenderDpi / 72.0;
|
|
var renderWidth = (int)Math.Ceiling(pageSize.Width * scaleFactor);
|
|
var renderHeight = (int)Math.Ceiling(pageSize.Height * scaleFactor);
|
|
|
|
//
|
|
if (renderWidth > MaxImageWidth)
|
|
{
|
|
//
|
|
var ratio = (double)MaxImageWidth / renderWidth;
|
|
renderWidth = MaxImageWidth;
|
|
renderHeight = (int)(renderHeight * ratio);
|
|
}
|
|
|
|
//
|
|
using var bitmap = document.Render(
|
|
dpiX: RenderDpi,
|
|
dpiY: RenderDpi,
|
|
page: pageIndex,
|
|
width: renderWidth,
|
|
forPrinting: false,
|
|
height: renderHeight
|
|
);
|
|
|
|
//
|
|
byte[] imageBytes;
|
|
using (var memoryStream = new MemoryStream())
|
|
{
|
|
//
|
|
bitmap.Save(memoryStream, System.Drawing.Imaging.ImageFormat.Png);
|
|
imageBytes = memoryStream.ToArray();
|
|
}
|
|
|
|
//
|
|
result.Images.Add(new XExtractedImage
|
|
{
|
|
Bytes = imageBytes,
|
|
MimeType = "image/png",
|
|
PageNumber = pageIndex + 1,
|
|
Description = $"Page {pageIndex + 1} of {fileName} ({renderWidth}x{renderHeight})"
|
|
});
|
|
}
|
|
catch (Exception)
|
|
{ }
|
|
}
|
|
}
|
|
catch (Exception ex)
|
|
{
|
|
result.ErrorMessage = $"PDF image extraction failed: {ex.Message}";
|
|
}
|
|
}, cancellationToken);
|
|
|
|
//
|
|
return result;
|
|
}
|
|
}
|
|
} |