fix file content extractor mechanism ...
This commit is contained in:
@@ -0,0 +1,103 @@
|
||||
using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Threading;
|
||||
using System.Reflection;
|
||||
using xCommons.Extensions;
|
||||
using xExceptions.Constants;
|
||||
using System.Threading.Tasks;
|
||||
using System.Collections.Generic;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
|
||||
namespace xAiApi.Providers
|
||||
{
|
||||
/// <summary>
|
||||
/// Composite extractor that delegates to appropriate extractor
|
||||
/// based on MIME type ...
|
||||
/// </summary>
|
||||
public class XFileContentExtractor : IXFileContentExtractor
|
||||
{
|
||||
//
|
||||
private readonly IList<IXFileContentExtractor> extractors = [];
|
||||
|
||||
public XFileContentExtractor()
|
||||
{
|
||||
//
|
||||
var path = Path.Combine(AppContext.BaseDirectory);
|
||||
var files = Directory.GetFiles(path, "*.dll");
|
||||
foreach (var dll in files)
|
||||
{
|
||||
//
|
||||
// Retrieve DLL Assembly ...
|
||||
var assembly = Assembly.LoadFrom(dll);
|
||||
|
||||
//
|
||||
// Extract Registerar Types ...
|
||||
var extractorTypes = assembly
|
||||
.GetExportedTypes()
|
||||
.Where(t =>
|
||||
!t.IsAbstract &&
|
||||
typeof(IXFileContentExtractor).IsAssignableFrom(t));
|
||||
foreach (var extractorType in extractorTypes)
|
||||
{
|
||||
//
|
||||
// Ignore XFileContentExtractor ...
|
||||
if (extractorType.FullName.Contains(nameof(XFileContentExtractor)))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
//
|
||||
// Create Instance of Service ...
|
||||
var instance = (IXFileContentExtractor)Activator.CreateInstance(extractorType);
|
||||
if (!instance.IsNull())
|
||||
{
|
||||
//
|
||||
// Add Registerar Instance to Collection ...
|
||||
extractors.Add(instance);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Check if this extractor supports the specified MIME type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
return extractors.Any(e => e.CanExtract(mimeType));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract text content from file stream ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
var extractor = extractors
|
||||
.FirstOrDefault(e => e.CanExtract(mimeType));
|
||||
if (extractor.IsNull())
|
||||
{
|
||||
//
|
||||
XException.NotAllowed.Throw(
|
||||
$"Unsupported file type: {mimeType}"
|
||||
);
|
||||
}
|
||||
|
||||
//
|
||||
var result = await extractor
|
||||
.ExtractAsync(
|
||||
mimeType: mimeType,
|
||||
fileStream: fileStream,
|
||||
cancellationToken: cancellationToken
|
||||
);
|
||||
|
||||
//
|
||||
return result;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Text;
|
||||
using UglyToad.PdfPig;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
|
||||
namespace xAiApi.Providers
|
||||
{
|
||||
/// <summary>
|
||||
/// Extracts content from PDF files using PdfPig ...
|
||||
/// </summary>
|
||||
public class XPdfFileContentExtractor : IXPdfFileContentExtractor
|
||||
{
|
||||
/// <summary>
|
||||
/// Supported MIME Types ...
|
||||
/// </summary>
|
||||
private static readonly string[] SupportedMimeTypes =
|
||||
[
|
||||
"application/pdf"
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Check if this extractor supports the specified MIME type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
//
|
||||
return SupportedMimeTypes.Contains(
|
||||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||||
);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract text content from file stream ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
var result = await Task.Run(() =>
|
||||
{
|
||||
//
|
||||
var sb = new StringBuilder();
|
||||
using var document = PdfDocument.Open(fileStream);
|
||||
foreach (var page in document.GetPages())
|
||||
{
|
||||
//
|
||||
var text = page.Text;
|
||||
|
||||
//
|
||||
sb.AppendLine(text);
|
||||
sb.AppendLine();
|
||||
}
|
||||
|
||||
//
|
||||
return sb.ToString();
|
||||
}, cancellationToken);
|
||||
|
||||
//
|
||||
return result;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
|
||||
namespace xAiApi.Providers
|
||||
{
|
||||
/// <summary>
|
||||
/// Extracts content from plain text files (txt, md, csv, json, xml) ...
|
||||
/// </summary>
|
||||
public class XPlainTextFileContentExtractor : IXPlainTextFileContentExtractor
|
||||
{
|
||||
/// <summary>
|
||||
/// Supported MIME Types ...
|
||||
/// </summary>
|
||||
private static readonly string[] SupportedMimeTypes =
|
||||
[
|
||||
"text/plain",
|
||||
"text/markdown",
|
||||
"text/csv",
|
||||
"text/html",
|
||||
"text/xml",
|
||||
"application/json",
|
||||
"application/xml"
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Check if this extractor supports the specified MIME type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
//
|
||||
return SupportedMimeTypes.Contains(
|
||||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||||
);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract text content from file stream ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
using var reader = new StreamReader(fileStream);
|
||||
return await reader.ReadToEndAsync();
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user