This commit is contained in:
2026-10-03 17:42:36 +03:30
parent c30d93be0b
commit 7cad889cd0
31 changed files with 1346 additions and 212 deletions
@@ -0,0 +1,159 @@
using System;
using System.IO;
using System.Linq;
using Whisper.net;
using System.Threading;
using xAiModels.Models;
using System.Threading.Tasks;
using xAiApi.Interfaces.Extractors;
using DocumentFormat.OpenXml.EMMA;
namespace xAiApi.Providers.Extractors
{
/// <summary>
/// Extracts text content from Audio files ...
/// </summary>
public class XAudioFileContentExtractor : IXAudioFileContentExtractor
{
/// <summary>
/// Supported MIME Types ...
/// </summary>
private static readonly string[] SupportedMimeTypes =
[
"audio/mpeg",
"audio/mp3",
"audio/wav",
"audio/wave",
"audio/ogg",
"audio/m4a",
"audio/mp4",
"audio/webm"
];
private readonly string language;
private readonly string whisperModelPath;
public XAudioFileContentExtractor() : this(
language: "fa",
whisperModelPath: "Models/ggml-base.bin"
)
{ }
public XAudioFileContentExtractor(
string whisperModelPath = "Models/ggml-base.bin",
string language = "fa"
)
{
//
this.language = language;
this.whisperModelPath = whisperModelPath;
}
/// <summary>
/// Check if this extractor supports the specified MIME type ...
/// </summary>
public bool CanExtract(string mimeType)
{
//
return SupportedMimeTypes.Contains(
mimeType?.ToLowerInvariant() ?? string.Empty
);
}
/// <summary>
/// Extract text content from file stream ...
/// </summary>
public async Task<string> ExtractAsync(
Stream fileStream,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var result = await ExtractRichAsync(
fileStream,
"audio",
mimeType,
cancellationToken
);
//
return result.AudioTranscript;
}
/// <summary>
/// Extract content from stream as Rich Result ...
/// </summary>
/// <param name="fileStream"></param>
/// <param name="fileName"></param>
/// <param name="mimeType"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public async Task<XFileExtractionResult> ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var result = new XFileExtractionResult
{
FileName = fileName,
MimeType = mimeType
};
//
try
{
//
using var memoryStream = new MemoryStream();
await fileStream.CopyToAsync(memoryStream, cancellationToken);
memoryStream.Position = 0;
//
using var wavStream = await ConvertToWavAsync(
memoryStream,
cancellationToken
);
//
using var factory = WhisperFactory.FromPath(whisperModelPath);
using var processor = factory.CreateBuilder()
.WithLanguage(language)
.Build();
//
var segments = new System.Text.StringBuilder();
await foreach (var segment in processor.ProcessAsync(wavStream, cancellationToken))
{
segments.Append(segment.Text);
}
//
result.AudioTranscript = segments.ToString().Trim();
result.Text = result.AudioTranscript;
}
catch (Exception ex)
{
result.ErrorMessage = $"Audio extraction failed: {ex.Message}";
}
//
return result;
}
/// <summary>
/// Audio Format Converting ...
/// </summary>
private async Task<Stream> ConvertToWavAsync(
Stream inputStream,
CancellationToken cancellationToken
)
{
//
// TODO: Complete this ...
return await Task.FromResult(inputStream);
}
}
}
@@ -3,6 +3,7 @@ using System.IO;
using System.Linq;
using System.Text;
using System.Threading;
using xAiModels.Models;
using System.Threading.Tasks;
using xAiApi.Interfaces.Extractors;
using DocumentFormat.OpenXml.Drawing;
@@ -75,5 +76,39 @@ namespace xAiApi.Providers.Extractors
//
return stringBuilder.ToString();
}
/// <summary>
/// Extract content from stream as Rich Result ...
/// </summary>
/// <param name="fileStream"></param>
/// <param name="fileName"></param>
/// <param name="mimeType"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public async Task<XFileExtractionResult> ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var content = await ExtractAsync(
mimeType: mimeType,
fileStream: fileStream,
cancellationToken: cancellationToken
);
//
var result = new XFileExtractionResult
{
Text = content,
FileName = fileName,
MimeType = mimeType
};
//
return result;
}
}
}
@@ -5,6 +5,7 @@ using System.Linq;
using System.Text;
using ExcelDataReader;
using System.Threading;
using xAiModels.Models;
using System.Threading.Tasks;
using xAiApi.Interfaces.Extractors;
@@ -89,5 +90,39 @@ namespace xAiApi.Providers.Extractors
//
return stringBuilder.ToString();
}
/// <summary>
/// Extract content from stream as Rich Result ...
/// </summary>
/// <param name="fileStream"></param>
/// <param name="fileName"></param>
/// <param name="mimeType"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public async Task<XFileExtractionResult> ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var content = await ExtractAsync(
mimeType: mimeType,
fileStream: fileStream,
cancellationToken: cancellationToken
);
//
var result = new XFileExtractionResult
{
Text = content,
FileName = fileName,
MimeType = mimeType
};
//
return result;
}
}
}
+94 -43
View File
@@ -1,15 +1,14 @@
using System;
using System.IO;
using System.Linq;
using System.Threading;
using System.Reflection;
using xAiModels.Models;
using xCommons.Extensions;
using xExceptions.Constants;
using System.Threading.Tasks;
using System.Collections.Generic;
using xAiApi.Interfaces.Extractors;
namespace xAiApi.Providers
namespace xAiApi.Providers.Extractors
{
/// <summary>
/// Composite extractor that delegates to appropriate extractor
@@ -18,67 +17,50 @@ namespace xAiApi.Providers
public class XFileContentExtractor : IXFileContentExtractor
{
//
private readonly IList<IXFileContentExtractor> extractors = [];
private readonly IList<IXFileContentExtractorBase> _extractors = [];
public XFileContentExtractor()
public XFileContentExtractor(
IEnumerable<IXFileContentExtractorBase> availableExtractors
)
{
//
var path = Path.Combine(AppContext.BaseDirectory);
var files = Directory.GetFiles(path, "*.dll");
foreach (var dll in files)
{
//
// Retrieve DLL Assembly ...
var assembly = Assembly.LoadFrom(dll);
//
// Extract Registerar Types ...
var extractorTypes = assembly
.GetExportedTypes()
.Where(t =>
!t.IsAbstract &&
typeof(IXFileContentExtractor).IsAssignableFrom(t));
foreach (var extractorType in extractorTypes)
{
//
// Ignore XFileContentExtractor ...
if (extractorType.FullName.Contains(nameof(XFileContentExtractor)))
{
continue;
}
//
// Create Instance of Service ...
var instance = (IXFileContentExtractor)Activator.CreateInstance(extractorType);
if (!instance.IsNull())
{
//
// Add Registerar Instance to Collection ...
extractors.Add(instance);
}
}
}
_extractors = [.. availableExtractors
.Where(e => e.GetType() != typeof(XFileContentExtractor))];
}
/// <summary>
/// Check if this extractor supports the specified MIME type ...
/// </summary>
public bool CanExtract(string mimeType)
/// <param name="mimeType"></param>
/// <param name="extractors"></param>
/// <returns></returns>
public bool CanExtract(
string mimeType,
string[] extractors = null
)
{
return extractors.Any(e => e.CanExtract(mimeType));
//
return GetExtractors(extractors)
.Any(e => e.CanExtract(mimeType));
}
/// <summary>
/// Extract text content from file stream ...
/// </summary>
/// <param name="fileStream"></param>
/// <param name="mimeType"></param>
/// <param name="extractors"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public async Task<string> ExtractAsync(
Stream fileStream,
string mimeType,
string[] extractors = null,
CancellationToken cancellationToken = default
)
{
//
var extractor = extractors
var extractor = GetExtractors(extractors)
.FirstOrDefault(e => e.CanExtract(mimeType));
if (extractor.IsNull())
{
@@ -99,5 +81,74 @@ namespace xAiApi.Providers
//
return result;
}
/// <summary>
/// Extract content from stream as Rich Result ...
/// </summary>
/// <param name="fileStream"></param>
/// <param name="fileName"></param>
/// <param name="mimeType"></param>
/// <param name="extractors"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public async Task<XFileExtractionResult> ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
string[] extractors = null,
CancellationToken cancellationToken = default
)
{
//
var extractor = GetExtractors(extractors)
.FirstOrDefault(e => e.CanExtract(mimeType));
if (extractor.IsNull())
{
//
XException.NotAllowed.Throw(
$"Unsupported file type: {mimeType}"
);
}
//
var result = await extractor
.ExtractRichAsync(
mimeType: mimeType,
fileName: fileName,
fileStream: fileStream,
cancellationToken: cancellationToken
);
//
return result;
}
/// <summary>
/// Extract Provided Extractors ...
/// </summary>
/// <param name="extractors"></param>
/// <returns></returns>
private IList<IXFileContentExtractorBase> GetExtractors(string[] extractors = null)
{
//
var result = new List<IXFileContentExtractorBase>();
//
if (extractors.IsNull() || !extractors.HasChild())
{
//
result = [.. _extractors];
}
else
{
//
result = [.. _extractors
.Where(e => extractors
.Any(ex => e.GetType().FullName.Contains(ex)))];
}
//
return result;
}
}
}
@@ -28,6 +28,9 @@ namespace xAiApi.Providers.Extractors
/// </summary>
private readonly bool _enableOcrFallback;
public XImageFileContentExtractor() : this(enableOcrFallback: true)
{ }
public XImageFileContentExtractor(bool enableOcrFallback = true)
{
_enableOcrFallback = enableOcrFallback;
@@ -132,9 +135,7 @@ namespace xAiApi.Providers.Extractors
)
{
//
// TODO:
// پیاده‌سازی با Tesseract یا هر کتابخانه OCR دیگر
// در حال حاضر یک placeholder است
// TODO: Complete this ...
return await Task.FromResult(string.Empty);
}
}
@@ -2,12 +2,13 @@ using System;
using System.IO;
using System.Linq;
using System.Text;
using UglyToad.PdfPig;
using PdfiumViewer;
using System.Threading;
using xAiModels.Models;
using System.Threading.Tasks;
using xAiApi.Interfaces.Extractors;
namespace xAiApi.Providers
namespace xAiApi.Providers.Extractors
{
/// <summary>
/// Extracts content from PDF files using PdfPig ...
@@ -22,6 +23,26 @@ namespace xAiApi.Providers
"application/pdf"
];
/// <summary>
/// Minimum Length for Fallback Using Images ...
/// </summary>
private const int MinTextLengthThreshold = 50;
/// <summary>
/// Max Pages for Falling Back Using Images ...
/// </summary>
private const int MaxPagesForImageFallback = 10;
/// <summary>
/// Default DPI for Rendering Images ...
/// </summary>
private const int RenderDpi = 150;
/// <summary>
/// Max Rendered Image Width ...
/// </summary>
private const int MaxImageWidth = 2000;
/// <summary>
/// Check if this extractor supports the specified MIME type ...
/// </summary>
@@ -41,21 +62,100 @@ namespace xAiApi.Providers
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var result = await ExtractRichAsync(
fileStream,
"document.pdf",
mimeType,
cancellationToken
);
//
return result.Text;
}
/// <summary>
/// Extract content from stream as Rich Result ...
/// </summary>
/// <param name="fileStream"></param>
/// <param name="fileName"></param>
/// <param name="mimeType"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public async Task<XFileExtractionResult> ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var result = new XFileExtractionResult
{
FileName = fileName,
MimeType = mimeType
};
//
using var memoryStream = new MemoryStream();
await fileStream.CopyToAsync(memoryStream, cancellationToken);
memoryStream.Position = 0;
//
var extractedText = await ExtractTextAsync(
memoryStream,
cancellationToken
);
if (IsTextSufficient(extractedText))
{
// PDF متنی است
result.Text = extractedText;
}
else
{
// PDF اسکن‌شده است - Fallback به تصویر
memoryStream.Position = 0;
result = await FallbackToImageExtractionAsync(
memoryStream,
fileName,
mimeType,
cancellationToken
);
result.UsedImageFallback = true;
}
//
return result;
}
/// <summary>
/// Extract Text From PDF Files using Common way ...
/// </summary>
/// <param name="pdfStream"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
private async Task<string> ExtractTextAsync(
Stream pdfStream,
CancellationToken cancellationToken
)
{
//
var result = await Task.Run(() =>
{
//
var sb = new StringBuilder();
using var document = PdfDocument.Open(fileStream);
using var document = UglyToad.PdfPig.PdfDocument.Open(pdfStream);
foreach (var page in document.GetPages())
{
//
var text = page.Text;
//
sb.AppendLine(text);
sb.AppendLine();
var text = page.Text?.Trim();
if (!string.IsNullOrEmpty(text))
{
//
sb.AppendLine(text);
sb.AppendLine();
}
}
//
@@ -65,5 +165,129 @@ namespace xAiApi.Providers
//
return result;
}
/// <summary>
/// Check Extracted Content is Sufficient or not ...
/// if not try to Falling Back using Images ...
/// </summary>
private bool IsTextSufficient(string text)
{
//
if (string.IsNullOrWhiteSpace(text))
{
return false;
}
//
// حذف فاصله‌ها و بررسی طول
var cleanText = new string(text.Where(c => !char.IsWhiteSpace(c)).ToArray());
//
return cleanText.Length >= MinTextLengthThreshold;
}
/// <summary>
/// Fallback Converts PDF Pages to Images ...
/// </summary>
private async Task<XFileExtractionResult> FallbackToImageExtractionAsync(
Stream pdfStream,
string fileName,
string mimeType,
CancellationToken cancellationToken
)
{
//
var result = new XFileExtractionResult
{
FileName = fileName,
MimeType = mimeType,
UsedImageFallback = true
};
//
await Task.Run(() =>
{
//
try
{
//
using var document = PdfDocument.Load(pdfStream);
//
var pageCount = Math.Min(
document.PageCount,
MaxPagesForImageFallback
);
//
for (int pageIndex = 0; pageIndex < pageCount; pageIndex++)
{
//
if (cancellationToken.IsCancellationRequested)
{
break;
}
//
try
{
//
var pageSize = document.PageSizes[pageIndex];
//
var scaleFactor = RenderDpi / 72.0;
var renderWidth = (int)Math.Ceiling(pageSize.Width * scaleFactor);
var renderHeight = (int)Math.Ceiling(pageSize.Height * scaleFactor);
//
if (renderWidth > MaxImageWidth)
{
//
var ratio = (double)MaxImageWidth / renderWidth;
renderWidth = MaxImageWidth;
renderHeight = (int)(renderHeight * ratio);
}
//
using var bitmap = document.Render(
dpiX: RenderDpi,
dpiY: RenderDpi,
page: pageIndex,
width: renderWidth,
forPrinting: false,
height: renderHeight
);
//
byte[] imageBytes;
using (var memoryStream = new MemoryStream())
{
//
bitmap.Save(memoryStream, System.Drawing.Imaging.ImageFormat.Png);
imageBytes = memoryStream.ToArray();
}
//
result.Images.Add(new XExtractedImage
{
Bytes = imageBytes,
MimeType = "image/png",
PageNumber = pageIndex + 1,
Description = $"Page {pageIndex + 1} of {fileName} ({renderWidth}x{renderHeight})"
});
}
catch (Exception)
{ }
}
}
catch (Exception ex)
{
result.ErrorMessage = $"PDF image extraction failed: {ex.Message}";
}
}, cancellationToken);
//
return result;
}
}
}
@@ -2,10 +2,11 @@ using System;
using System.IO;
using System.Linq;
using System.Threading;
using xAiModels.Models;
using System.Threading.Tasks;
using xAiApi.Interfaces.Extractors;
namespace xAiApi.Providers
namespace xAiApi.Providers.Extractors
{
/// <summary>
/// Extracts content from plain text files (txt, md, csv, json, xml) ...
@@ -50,5 +51,39 @@ namespace xAiApi.Providers
using var reader = new StreamReader(fileStream);
return await reader.ReadToEndAsync();
}
/// <summary>
/// Extract content from stream as Rich Result ...
/// </summary>
/// <param name="fileStream"></param>
/// <param name="fileName"></param>
/// <param name="mimeType"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public async Task<XFileExtractionResult> ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var content = await ExtractAsync(
mimeType: mimeType,
fileStream: fileStream,
cancellationToken: cancellationToken
);
//
var result = new XFileExtractionResult
{
Text = content,
FileName = fileName,
MimeType = mimeType
};
//
return result;
}
}
}
@@ -0,0 +1,272 @@
using System;
using System.IO;
using System.Linq;
using System.Threading;
using xAiApi.Constants;
using xAiModels.Models;
using xAiApi.Extensions;
using xAiApi.Interfaces;
using xCommons.Extensions;
using xAiModels.Extensions;
using xAiApi.Configurations;
using xExceptions.Constants;
using System.Threading.Tasks;
using Microsoft.Extensions.AI;
using System.Collections.Generic;
using xAiApi.Interfaces.Extractors;
using System.Runtime.CompilerServices;
namespace xAiApi.Providers.Extractors
{
/// <summary>
/// Extracts content from files using Vision Model ...
/// </summary>
public class XVisionFileContentExtractor : IXVisionFileContentExtractor
{
/// <summary>
/// Supported MIME Types ...
/// </summary>
private static readonly string[] SupportedMimeTypes =
[
"image/png",
"image/jpeg",
"image/jpg",
"image/webp",
"image/bmp",
"application/pdf",
];
//
private readonly string prompt;
private readonly XAiModelDescriptor descriptor;
public XVisionFileContentExtractor(
XAiApiConfiguration configuration,
string promptName = XAiApiConstants.XAiApiContentExtractionPromptName,
string modelName = XAiApiConstants.XAiDefaultVisionModelName
)
{
//
// Retrieve Descriptoir ...
if (modelName.IsNullOrEmpty())
{
XException.InvalidArgs.Throw();
}
descriptor = configuration.GetModel(modelName);
//
// Retrieve Extraction Prompt ...
if (promptName.IsNullOrEmpty())
{
XException.InvalidArgs.Throw();
}
prompt = configuration.GetPrompt(
name: promptName,
@params: null
);
//
// Validate Requirements ...
if (prompt.IsNullOrEmpty() || descriptor.IsNullOrDefault())
{
XException.InvalidData.Throw();
}
}
/// <summary>
/// Check if this extractor supports the specified MIME type ...
/// </summary>
public bool CanExtract(string mimeType)
{
//
return SupportedMimeTypes.Contains(
mimeType?.ToLowerInvariant() ?? string.Empty
);
}
/// <summary>
/// Extract text content from file stream ...
/// </summary>
public async Task<string> ExtractAsync(
Stream fileStream,
string mimeType,
CancellationToken cancellationToken = default
)
{
var result = await ExtractRichAsync(fileStream, "file", mimeType, cancellationToken);
return result.Text;
}
/// <summary>
/// Extract content from stream as Rich Result ...
/// </summary>
/// <param name="fileStream"></param>
/// <param name="fileName"></param>
/// <param name="mimeType"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public async Task<XFileExtractionResult> ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
CancellationToken cancellationToken = default
)
{
//
var result = new XFileExtractionResult
{
FileName = fileName,
MimeType = mimeType
};
//
try
{
//
using var memoryStream = new MemoryStream();
await fileStream.CopyToAsync(memoryStream, cancellationToken);
var imageBytes = memoryStream.ToArray();
var messages = new[]
{
//
new ChatMessage(ChatRole.User, new[]
{
new DataContent(imageBytes, mimeType)
})
};
//
var extractedContent = await RequestOCRAsync(
messages: messages,
cancellationToken: cancellationToken
);
if (!extractedContent.IsNullOrEmpty())
{
result.Text = extractedContent.Trim();
}
else
{
result.ErrorMessage = "there is not any Extracted Content ...";
}
}
catch (Exception ex)
{
result.ErrorMessage = $"Error in Vision Extraction: {ex.Message} ...";
}
//
return result;
}
//
#region OCR ...
/// <summary>
/// Request for Doing OCR on Given Data Contents ...
/// </summary>
/// <param name="messages"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public virtual async Task<string> RequestOCRAsync(
IList<ChatMessage> messages,
CancellationToken cancellationToken = default
)
{
//
// Validate ...
if (!messages.HasChild())
{
XException.InvalidArgs.Throw();
}
//
using var client = descriptor.GetClient();
//
// Preparing Extraction Prompt Message ...
var pMessage = new ChatMessage(
ChatRole.System,
prompt
);
//
messages = [pMessage, .. messages];
//
var response = await client.GetResponseAsync(
messages: messages,
cancellationToken: cancellationToken
);
//
// Validate Response ...
if (!response.IsValid())
{
//
// Dispose Client ...
client.Dispose();
XException.ActionFailed.Throw();
}
//
// Retrieve Response Text ...
var result = response.Text;
//
return result;
}
/// <summary>
/// Request for Doing OCR on Given Data Contents as Stream ...
/// </summary>
/// <param name="messages"></param>
/// <param name="cancellationToken"></param>
/// <returns></returns>
public virtual async IAsyncEnumerable<string> RequestOCRAsEnumerable(
IList<ChatMessage> messages,
[EnumeratorCancellation]
CancellationToken cancellationToken = default
)
{
//
// Validate ...
if (!messages.HasChild())
{
XException.InvalidArgs.Throw();
}
//
using var client = descriptor.GetClient();
//
// Preparing Extraction Prompt Message ...
var pMessage = new ChatMessage(
ChatRole.System,
prompt
);
//
messages = [pMessage, .. messages];
//
var enumerable = client.GetStreamingResponseAsync(
options: null,
messages: messages
);
//
await foreach (var res in enumerable)
{
//
// Cancellation Token ...
if (cancellationToken.IsCancellationRequested)
{
yield break;
}
//
yield return res.Text;
}
}
#endregion
}
}