adding File Content Extraction Services ...
This commit is contained in:
@@ -0,0 +1,69 @@
|
||||
using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Text;
|
||||
using UglyToad.PdfPig;
|
||||
using System.Threading;
|
||||
using xAiApi.Interfaces;
|
||||
using System.Threading.Tasks;
|
||||
|
||||
namespace xAiApi.Providers
|
||||
{
|
||||
/// <summary>
|
||||
/// Extracts content from PDF files using PdfPig ...
|
||||
/// </summary>
|
||||
public class XPdfContentExtractor : IXFileContentExtractor
|
||||
{
|
||||
/// <summary>
|
||||
/// Supported MIME Types ...
|
||||
/// </summary>
|
||||
private static readonly string[] SupportedMimeTypes =
|
||||
[
|
||||
"application/pdf"
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Check if this extractor supports the specified MIME type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
//
|
||||
return SupportedMimeTypes.Contains(
|
||||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||||
);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract text content from file stream ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
var result = await Task.Run(() =>
|
||||
{
|
||||
//
|
||||
var sb = new StringBuilder();
|
||||
using var document = PdfDocument.Open(fileStream);
|
||||
foreach (var page in document.GetPages())
|
||||
{
|
||||
//
|
||||
var text = page.Text;
|
||||
|
||||
//
|
||||
sb.AppendLine(text);
|
||||
sb.AppendLine();
|
||||
}
|
||||
|
||||
//
|
||||
return sb.ToString();
|
||||
}, cancellationToken);
|
||||
|
||||
//
|
||||
return result;
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user