adding File Content Extraction Services ...

This commit is contained in:
2026-10-01 01:16:24 +03:30
parent aac3eb92c1
commit f32972b55f
6 changed files with 225 additions and 2 deletions
@@ -0,0 +1,68 @@
using System.IO;
using System.Linq;
using System.Threading;
using xAiApi.Interfaces;
using xCommons.Extensions;
using xExceptions.Constants;
using System.Threading.Tasks;
using System.Collections.Generic;
namespace xAiApi.Providers
{
/// <summary>
/// Composite extractor that delegates to appropriate extractor
/// based on MIME type ...
/// </summary>
public class XCompositeFileContentExtractor : IXFileContentExtractor
{
//
private readonly IEnumerable<IXFileContentExtractor> extractors;
public XCompositeFileContentExtractor(
IEnumerable<IXFileContentExtractor> extractors
)
{
this.extractors = extractors;
}
/// <summary>
/// Check if this extractor supports the specified MIME type ...
/// </summary>
public bool CanExtract(string mimeType)
{
return extractors.Any(e => e.CanExtract(mimeType));
}
/// <summary>
/// Extract text content from file stream ...
/// </summary>
public async Task<string> ExtractAsync(
Stream fileStream,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var extractor = extractors
.FirstOrDefault(e => e.CanExtract(mimeType));
if (extractor.IsNull())
{
//
XException.NotAllowed.Throw(
$"Unsupported file type: {mimeType}"
);
}
//
var result = await extractor
.ExtractAsync(
mimeType: mimeType,
fileStream: fileStream,
cancellationToken: cancellationToken
);
//
return result;
}
}
}
+69
View File
@@ -0,0 +1,69 @@
using System;
using System.IO;
using System.Linq;
using System.Text;
using UglyToad.PdfPig;
using System.Threading;
using xAiApi.Interfaces;
using System.Threading.Tasks;
namespace xAiApi.Providers
{
/// <summary>
/// Extracts content from PDF files using PdfPig ...
/// </summary>
public class XPdfContentExtractor : IXFileContentExtractor
{
/// <summary>
/// Supported MIME Types ...
/// </summary>
private static readonly string[] SupportedMimeTypes =
[
"application/pdf"
];
/// <summary>
/// Check if this extractor supports the specified MIME type ...
/// </summary>
public bool CanExtract(string mimeType)
{
//
return SupportedMimeTypes.Contains(
mimeType?.ToLowerInvariant() ?? string.Empty
);
}
/// <summary>
/// Extract text content from file stream ...
/// </summary>
public async Task<string> ExtractAsync(
Stream fileStream,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var result = await Task.Run(() =>
{
//
var sb = new StringBuilder();
using var document = PdfDocument.Open(fileStream);
foreach (var page in document.GetPages())
{
//
var text = page.Text;
//
sb.AppendLine(text);
sb.AppendLine();
}
//
return sb.ToString();
}, cancellationToken);
//
return result;
}
}
}
+54
View File
@@ -0,0 +1,54 @@
using System;
using System.IO;
using System.Linq;
using System.Threading;
using xAiApi.Interfaces;
using System.Threading.Tasks;
namespace xAiApi.Providers
{
/// <summary>
/// Extracts content from plain text files (txt, md, csv, json, xml) ...
/// </summary>
public class XPlainTextContentExtractor : IXFileContentExtractor
{
/// <summary>
/// Supported MIME Types ...
/// </summary>
private static readonly string[] SupportedMimeTypes =
[
"text/plain",
"text/markdown",
"text/csv",
"text/html",
"text/xml",
"application/json",
"application/xml"
];
/// <summary>
/// Check if this extractor supports the specified MIME type ...
/// </summary>
public bool CanExtract(string mimeType)
{
//
return SupportedMimeTypes.Contains(
mimeType?.ToLowerInvariant() ?? string.Empty
);
}
/// <summary>
/// Extract text content from file stream ...
/// </summary>
public async Task<string> ExtractAsync(
Stream fileStream,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
using var reader = new StreamReader(fileStream);
return await reader.ReadToEndAsync();
}
}
}