+
+
+
+
+
+
🔍 گام ۱: تحلیل وضعیت فعلی و شکافها
+
+
۱.۱ وضعیت فعلی Extractor ها:
+
+
+ | Extractor |
+ وضعیت |
+ نوع خروجی |
+ شکاف |
+
+
+ XPlainTextFileContentExtractor |
+ ✅ موجود |
+ Text |
+ - |
+
+
+ XDocxFileContentExtractor |
+ ✅ موجود |
+ Text |
+ عدم پشتیبانی از تصاویر داخل DOCX |
+
+
+ XExcelFileContentExtractor |
+ ✅ موجود |
+ Text (Tabular) |
+ - |
+
+
+ XPdfFileContentExtractor |
+ ⚠️ ناقص |
+ Text |
+ عدم Fallback برای PDF های اسکنشده |
+
+
+ | Image Extractor |
+ ❌ موجود نیست |
+ - |
+ نیاز به پیادهسازی (Vision API) |
+
+
+ | Audio Extractor |
+ ❌ موجود نیست |
+ - |
+ نیاز به پیادهسازی (Whisper) |
+
+
+
+
۱.۲ مشکل اساسی در IXFileContentExtractor فعلی:
+
+
⚠️ محدودیت طراحی فعلی:
+
Interface فعلی فقط یک string برمیگرداند. این برای متن کافی است اما برای تصویر و صوت نیاز به ساختار دادهای غنیتر داریم که بتواند چندین نوع محتوا (Text + Images + Audio) را همزمان حمل کند.
+
+
+
// ❌ Interface فعلی - محدود به string
+public interface IXFileContentExtractor
+{
+ bool CanExtract(string mimeType);
+ Task<string> ExtractAsync(Stream fileStream, string mimeType, ...);
+}
+
+
+
+
+
🎯 گام ۲: استراتژی معماری جدید
+
+
+
+
📦 خروجی چندگانه (Multi-Content)
+
+ - ایجاد
XFileExtractionResult
+ - شامل: Text, Images[], AudioTranscript
+ - سازگار با
AIContent
+
+
+
+
🖼️ پشتیبانی تصویر (Vision)
+
+ - ارسال به عنوان
DataContent
+ - سازگار با Gemini, GPT-4V, Qwen-VL
+ - OCR محلی به عنوان Fallback
+
+
+
+
🎙️ پشتیبانی صوت (Whisper)
+
+ - تبدیل Speech-to-Text
+ - استفاده از Whisper محلی یا API
+ - خروجی به صورت متن
+
+
+
+
📄 PDF Fallback هوشمند
+
+ - استخراج متن ابتدا
+ - اگر متن خالی/کمحجم → تبدیل به تصویر
+ - ارسال صفحات به عنوان Vision
+
+
+
+
+
+ ✅ مزیت کلیدی: این طراحی Backward Compatible است. کدهای موجود که از ExtractAsync استفاده میکنند، بدون تغییر کار خواهند کرد.
+
+
+
+
+
+
🔧 گام ۳: بازطراحی Interface با خروجی چندگانه
+
+
+ NEW
+ xAiApi/Models/XFileExtractionResult.cs
+
+
+
using System;
+using System.Collections.Generic;
+using System.Linq;
+using xCommons.Extensions;
+using Microsoft.Extensions.AI;
+
+namespace xAiApi.Models
+{
+ /// <summary>
+ /// نتیجه استخراج محتوا از فایل - میتواند شامل متن، تصاویر و صوت باشد ...
+ /// </summary>
+ public class XFileExtractionResult
+ {
+ /// <summary>
+ /// محتوای متنی استخراج شده ...
+ /// </summary>
+ public string Text { get; set; } = string.Empty;
+
+ /// <summary>
+ /// تصاویر استخراج شده (برای Vision Models) ...
+ /// </summary>
+ public IList<XExtractedImage> Images { get; set; } = [];
+
+ /// <summary>
+ /// متن تبدیل شده از صوت ...
+ /// </summary>
+ public string AudioTranscript { get; set; } = string.Empty;
+
+ /// <summary>
+ /// نام فایل اصلی ...
+ /// </summary>
+ public string FileName { get; set; }
+
+ /// <summary>
+ /// MIME Type فایل ...
+ /// </summary>
+ public string MimeType { get; set; }
+
+ /// <summary>
+ /// آیا استخراج با Fallback تصویری انجام شده است ...
+ /// </summary>
+ public bool UsedImageFallback { get; set; } = false;
+
+ /// <summary>
+ /// پیام خطا (در صورت بروز مشکل) ...
+ /// </summary>
+ public string ErrorMessage { get; set; }
+
+ /// <summary>
+ /// آیا نتیجه معتبر است ...
+ /// </summary>
+ public bool IsValid()
+ {
+ return !Text.IsNullOrEmpty() ||
+ Images.HasChild() ||
+ !AudioTranscript.IsNullOrEmpty();
+ }
+
+ /// <summary>
+ /// تبدیل به لیست AIContent برای ارسال به LLM ...
+ /// </summary>
+ public IList<AIContent> ToAIContents()
+ {
+ var result = new List<AIContent>();
+
+ // ۱. افزودن متن اصلی
+ if (!Text.IsNullOrEmpty())
+ {
+ result.Add(new TextContent(Text));
+ }
+
+ // ۲. افزودن تصاویر (برای Vision Models)
+ if (Images.HasChild())
+ {
+ foreach (var img in Images)
+ {
+ result.Add(new DataContent(img.Bytes, img.MimeType));
+ }
+ }
+
+ // ۳. افزودن transcript صوت
+ if (!AudioTranscript.IsNullOrEmpty())
+ {
+ result.Add(new TextContent($"[Audio Transcript]\n{AudioTranscript}"));
+ }
+
+ return result;
+ }
+ }
+
+ /// <summary>
+ /// تصویر استخراج شده از فایل ...
+ /// </summary>
+ public class XExtractedImage
+ {
+ /// <summary>
+ /// داده بایت تصویر ...
+ /// </summary>
+ public byte[] Bytes { get; set; }
+
+ /// <summary>
+ /// MIME Type تصویر ...
+ /// </summary>
+ public string MimeType { get; set; } = "image/png";
+
+ /// <summary>
+ /// شماره صفحه (برای PDF) ...
+ /// </summary>
+ public int? PageNumber { get; set; }
+
+ /// <summary>
+ /// توضیح تصویر ...
+ /// </summary>
+ public string Description { get; set; }
+ }
+}
+
+
+ MODIFY
+ xAiApi/Interfaces/Extractors/IXFileContentExtractor.cs
+
+
+
using System.IO;
+using System.Threading;
+using System.Threading.Tasks;
+using xAiApi.Models;
+
+namespace xAiApi.Interfaces.Extractors
+{
+ /// <summary>
+ /// سرویس استخراج محتوای فایل - نسخه ارتقا یافته با پشتیبانی چندوجهی ...
+ /// </summary>
+ public interface IXFileContentExtractor
+ {
+ /// <summary>
+ /// بررسی پشتیبانی از MIME Type مشخص ...
+ /// </summary>
+ bool CanExtract(string mimeType);
+
+ /// <summary>
+ /// استخراج محتوای فایل با خروجی غنی (متن + تصویر + صوت) ...
+ /// </summary>
+ Task<XFileExtractionResult> ExtractRichAsync(
+ Stream fileStream,
+ string fileName,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ );
+
+ /// <summary>
+ /// استخراج محتوای متنی (سازگار با نسخه قبل) ...
+ /// </summary>
+ Task<string> ExtractAsync(
+ Stream fileStream,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ );
+ }
+}
+
+
+
+
+
🖼️ گام ۴: پیادهسازی ImageContentExtractor
+
این Extractor تصاویر را برای مدلهای Vision آماده میکند. دو حالت پشتیبانی میشود:
+
+ - حالت ۱ (Vision API): ارسال مستقیم به عنوان
DataContent به مدلهای چندوجهی
+ - حالت ۲ (OCR Fallback): استخراج متن با Tesseract برای مدلهای متنی
+
+
+
+ NEW
+ xAiApi/Interfaces/Extractors/IXImageFileContentExtractor.cs
+
+
+
namespace xAiApi.Interfaces.Extractors
+{
+ /// <summary>
+ /// Extractor اختصاصی برای فایلهای تصویری ...
+ /// </summary>
+ public interface IXImageFileContentExtractor : IXFileContentExtractor
+ { }
+}
+
+
+ NEW
+ xAiApi/Providers/Extractors/XImageFileContentExtractor.cs
+
+
+
using System;
+using System.IO;
+using System.Linq;
+using System.Threading;
+using System.Threading.Tasks;
+using xAiApi.Models;
+using xAiApi.Interfaces.Extractors;
+
+namespace xAiApi.Providers.Extractors
+{
+ /// <summary>
+ /// استخراج محتوا از فایلهای تصویری - پشتیبانی از Vision و OCR ...
+ /// </summary>
+ public class XImageFileContentExtractor : IXImageFileContentExtractor
+ {
+ private static readonly string[] SupportedMimeTypes =
+ [
+ "image/png",
+ "image/jpeg",
+ "image/jpg",
+ "image/gif",
+ "image/webp",
+ "image/bmp"
+ ];
+
+ /// <summary>
+ /// آیا OCR فعال است (برای مدلهای غیر Vision) ...
+ /// </summary>
+ private readonly bool _enableOcrFallback;
+
+ public XImageFileContentExtractor(bool enableOcrFallback = true)
+ {
+ _enableOcrFallback = enableOcrFallback;
+ }
+
+ public bool CanExtract(string mimeType)
+ {
+ return SupportedMimeTypes.Contains(
+ mimeType?.ToLowerInvariant() ?? string.Empty
+ );
+ }
+
+ public async Task<XFileExtractionResult> ExtractRichAsync(
+ Stream fileStream,
+ string fileName,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ var result = new XFileExtractionResult
+ {
+ FileName = fileName,
+ MimeType = mimeType
+ };
+
+ // خواندن بایتهای تصویر
+ using var memoryStream = new MemoryStream();
+ await fileStream.CopyToAsync(memoryStream, cancellationToken);
+ var imageBytes = memoryStream.ToArray();
+
+ // افزودن به عنوان تصویر برای Vision Models
+ result.Images.Add(new XExtractedImage
+ {
+ Bytes = imageBytes,
+ MimeType = mimeType,
+ Description = $"Attached image: {fileName}"
+ });
+
+ // OCR Fallback برای مدلهای متنی
+ if (_enableOcrFallback)
+ {
+ try
+ {
+ var ocrText = await PerformOcrAsync(imageBytes, cancellationToken);
+ if (!string.IsNullOrWhiteSpace(ocrText))
+ {
+ result.Text = ocrText;
+ }
+ }
+ catch
+ {
+ // OCR failed, but image is still available for Vision
+ }
+ }
+
+ return result;
+ }
+
+ public async Task<string> ExtractAsync(
+ Stream fileStream,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ var result = await ExtractRichAsync(
+ fileStream,
+ "image",
+ mimeType,
+ cancellationToken
+ );
+ return result.Text;
+ }
+
+ /// <summary>
+ /// انجام OCR روی تصویر - با استفاده از Tesseract ...
+ /// </summary>
+ private async Task<string> PerformOcrAsync(
+ byte[] imageBytes,
+ CancellationToken cancellationToken
+ )
+ {
+ // پیادهسازی با Tesseract یا هر کتابخانه OCR دیگر
+ // در حال حاضر یک placeholder است
+ return await Task.FromResult(string.Empty);
+ }
+ }
+}
+
+
+
⚠️ پکیجهای NuGet پیشنهادی برای OCR:
+
+ Tesseract - کتابخانه قدرتمند و متنباز
+ Microsoft.Azure.CognitiveServices.Vision.ComputerVision - اگر از Azure استفاده میکنید
+
+
+
+
+
+
+
🎙️ گام ۵: پیادهسازی AudioContentExtractor
+
این Extractor فایلهای صوتی را با استفاده از Whisper به متن تبدیل میکند.
+
+
+ NEW
+ xAiApi/Interfaces/Extractors/IXAudioFileContentExtractor.cs
+
+
+
namespace xAiApi.Interfaces.Extractors
+{
+ /// <summary>
+ /// Extractor اختصاصی برای فایلهای صوتی ...
+ /// </summary>
+ public interface IXAudioFileContentExtractor : IXFileContentExtractor
+ { }
+}
+
+
+ NEW
+ xAiApi/Providers/Extractors/XAudioFileContentExtractor.cs
+
+
+
using System;
+using System.IO;
+using System.Linq;
+using System.Threading;
+using System.Threading.Tasks;
+using xAiApi.Models;
+using xAiApi.Interfaces.Extractors;
+// Install-Package Whisper.net
+// Install-Package Whisper.net.Runtime
+using Whisper.net;
+
+namespace xAiApi.Providers.Extractors
+{
+ /// <summary>
+ /// استخراج متن از فایلهای صوتی با استفاده از Whisper ...
+ /// </summary>
+ public class XAudioFileContentExtractor : IXAudioFileContentExtractor
+ {
+ private static readonly string[] SupportedMimeTypes =
+ [
+ "audio/mpeg",
+ "audio/mp3",
+ "audio/wav",
+ "audio/wave",
+ "audio/ogg",
+ "audio/m4a",
+ "audio/mp4",
+ "audio/webm"
+ ];
+
+ private readonly string _whisperModelPath;
+ private readonly string _language;
+
+ public XAudioFileContentExtractor(
+ string whisperModelPath = "Models/ggml-base.bin",
+ string language = "fa" // فارسی پیشفرض
+ )
+ {
+ _whisperModelPath = whisperModelPath;
+ _language = language;
+ }
+
+ public bool CanExtract(string mimeType)
+ {
+ return SupportedMimeTypes.Contains(
+ mimeType?.ToLowerInvariant() ?? string.Empty
+ );
+ }
+
+ public async Task<XFileExtractionResult> ExtractRichAsync(
+ Stream fileStream,
+ string fileName,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ var result = new XFileExtractionResult
+ {
+ FileName = fileName,
+ MimeType = mimeType
+ };
+
+ try
+ {
+ // کپی به MemoryStream برای Whisper
+ using var memoryStream = new MemoryStream();
+ await fileStream.CopyToAsync(memoryStream, cancellationToken);
+ memoryStream.Position = 0;
+
+ // تبدیل به فرمت WAV 16kHz (مورد نیاز Whisper)
+ using var wavStream = await ConvertToWavAsync(memoryStream, cancellationToken);
+
+ // انجام Speech-to-Text با Whisper
+ using var factory = WhisperFactory.FromPath(_whisperModelPath);
+ using var processor = factory.CreateBuilder()
+ .WithLanguage(_language)
+ .Build();
+
+ var segments = new System.Text.StringBuilder();
+ await foreach (var segment in processor.ProcessAsync(wavStream, cancellationToken))
+ {
+ segments.Append(segment.Text);
+ }
+
+ result.AudioTranscript = segments.ToString().Trim();
+ result.Text = result.AudioTranscript;
+ }
+ catch (Exception ex)
+ {
+ result.ErrorMessage = $"Audio extraction failed: {ex.Message}";
+ }
+
+ return result;
+ }
+
+ public async Task<string> ExtractAsync(
+ Stream fileStream,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ var result = await ExtractRichAsync(
+ fileStream,
+ "audio",
+ mimeType,
+ cancellationToken
+ );
+ return result.AudioTranscript;
+ }
+
+ /// <summary>
+ /// تبدیل فرمت صوتی به WAV 16kHz (فرمت مورد نیاز Whisper) ...
+ /// </summary>
+ private async Task<Stream> ConvertToWavAsync(
+ Stream inputStream,
+ CancellationToken cancellationToken
+ )
+ {
+ // پیادهسازی با NAudio یا ffmpeg
+ // در حال حاضر یک placeholder است
+ return await Task.FromResult(inputStream);
+ }
+ }
+}
+
+
+
💡 پکیجهای NuGet مورد نیاز:
+
+ Whisper.net - پیادهسازی C# از Whisper
+ Whisper.net.Runtime - Runtime مورد نیاز
+ NAudio - برای تبدیل فرمت صوتی
+
+
دانلود مدل Whisper: از HuggingFace مدل ggml-base.bin یا ggml-small.bin را دانلود کنید.
+
+
+
+
+
+
📄 گام ۶: ارتقای PdfExtractor با Fallback تصویری
+
این مهمترین بخش است. اگر PDF متن نداشت (اسکنشده بود)، صفحات را به تصویر تبدیل میکنیم.
+
+
+ MODIFY
+ xAiApi/Providers/Extractors/XPdfFileContentExtractor.cs
+
+
+
using System;
+using System.IO;
+using System.Linq;
+using System.Text;
+using System.Threading;
+using System.Threading.Tasks;
+using UglyToad.PdfPig;
+using UglyToad.PdfPig.Rendering;
+using xAiApi.Models;
+using xAiApi.Interfaces.Extractors;
+
+namespace xAiApi.Providers
+{
+ /// <summary>
+ /// استخراج محتوا از PDF - با Fallback هوشمند برای PDF های اسکنشده ...
+ /// </summary>
+ public class XPdfFileContentExtractor : IXPdfFileContentExtractor
+ {
+ private static readonly string[] SupportedMimeTypes = ["application/pdf"];
+
+ /// <summary>
+ /// حداقل تعداد کاراکتر برای تشخیص PDF متنی ...
+ /// </summary>
+ private const int MinTextLengthThreshold = 50;
+
+ /// <summary>
+ /// حداکثر تعداد صفحات برای تبدیل به تصویر ...
+ /// </summary>
+ private const int MaxPagesForImageFallback = 10;
+
+ public bool CanExtract(string mimeType)
+ {
+ return SupportedMimeTypes.Contains(
+ mimeType?.ToLowerInvariant() ?? string.Empty
+ );
+ }
+
+ public async Task<XFileExtractionResult> ExtractRichAsync(
+ Stream fileStream,
+ string fileName,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ var result = new XFileExtractionResult
+ {
+ FileName = fileName,
+ MimeType = mimeType
+ };
+
+ // کپی به MemoryStream
+ using var memoryStream = new MemoryStream();
+ await fileStream.CopyToAsync(memoryStream, cancellationToken);
+ memoryStream.Position = 0;
+
+ // مرحله ۱: تلاش برای استخراج متن
+ var extractedText = await ExtractTextAsync(memoryStream, cancellationToken);
+
+ // مرحله ۲: بررسی کیفیت متن استخراج شده
+ if (IsTextSufficient(extractedText))
+ {
+ // PDF متنی است
+ result.Text = extractedText;
+ }
+ else
+ {
+ // PDF اسکنشده است - Fallback به تصویر
+ memoryStream.Position = 0;
+ result = await FallbackToImageExtractionAsync(
+ memoryStream,
+ fileName,
+ mimeType,
+ cancellationToken
+ );
+ result.UsedImageFallback = true;
+ }
+
+ return result;
+ }
+
+ public async Task<string> ExtractAsync(
+ Stream fileStream,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ var result = await ExtractRichAsync(
+ fileStream,
+ "document.pdf",
+ mimeType,
+ cancellationToken
+ );
+ return result.Text;
+ }
+
+ /// <summary>
+ /// استخراج متن از PDF ...
+ /// </summary>
+ private async Task<string> ExtractTextAsync(
+ Stream pdfStream,
+ CancellationToken cancellationToken
+ )
+ {
+ return await Task.Run(() =>
+ {
+ var sb = new StringBuilder();
+ using var document = PdfDocument.Open(pdfStream);
+
+ foreach (var page in document.GetPages())
+ {
+ var text = page.Text?.Trim();
+ if (!string.IsNullOrEmpty(text))
+ {
+ sb.AppendLine(text);
+ sb.AppendLine();
+ }
+ }
+
+ return sb.ToString();
+ }, cancellationToken);
+ }
+
+ /// <summary>
+ /// بررسی آیا متن استخراج شده کافی است ...
+ /// </summary>
+ private bool IsTextSufficient(string text)
+ {
+ if (string.IsNullOrWhiteSpace(text))
+ {
+ return false;
+ }
+
+ // حذف فاصلهها و بررسی طول
+ var cleanText = new string(text.Where(c => !char.IsWhiteSpace(c)).ToArray());
+ return cleanText.Length >= MinTextLengthThreshold;
+ }
+
+ /// <summary>
+ /// Fallback: تبدیل صفحات PDF به تصویر ...
+ /// </summary>
+ private async Task<XFileExtractionResult> FallbackToImageExtractionAsync(
+ Stream pdfStream,
+ string fileName,
+ string mimeType,
+ CancellationToken cancellationToken
+ )
+ {
+ var result = new XFileExtractionResult
+ {
+ FileName = fileName,
+ MimeType = mimeType,
+ UsedImageFallback = true
+ };
+
+ await Task.Run(() =>
+ {
+ using var document = PdfDocument.Open(pdfStream);
+ using var renderer = new PdfPigRenderer(document);
+
+ var pageCount = Math.Min(
+ document.NumberOfPages,
+ MaxPagesForImageFallback
+ );
+
+ for (int i = 0; i < pageCount; i++)
+ {
+ var page = document.GetPage(i);
+ var bitmap = renderer.Render(page);
+
+ // تبدیل Bitmap به PNG
+ using var memoryStream = new MemoryStream();
+ bitmap.Save(memoryStream, System.Drawing.Imaging.ImageFormat.Png);
+ var imageBytes = memoryStream.ToArray();
+
+ result.Images.Add(new XExtractedImage
+ {
+ Bytes = imageBytes,
+ MimeType = "image/png",
+ PageNumber = i + 1,
+ Description = $"Page {i + 1} of {fileName}"
+ });
+ }
+ }, cancellationToken);
+
+ return result;
+ }
+ }
+}
+
+
+
⚠️ پکیجهای NuGet اضافی برای PDF به تصویر:
+
+ UglyToad.PdfPig - قبلاً نصب شده
+ System.Drawing.Common - برای کار با Bitmap
+
+
نکته: PdfPigRenderer در نسخههای جدید PdfPig وجود دارد. اگر نبود، از Docnet یا PdfiumViewer استفاده کنید.
+
+
+
+
+
+
🔗 گام ۷: یکپارچهسازی در XAIServiceBase
+
+
+ MODIFY
+ xAiApi/Providers/XAIServiceBase.cs
+
+
+
اصلاح متد AskAsync - بخش پردازش فایلها:
+
+
// در متد AskAsync، بعد از آپلود فایلها و AddReference:
+
+// ✅ استخراج محتوای فایلها با خروجی غنی
+IList<AIContent> fileContents = null;
+if (uploadedFiles.HasChild())
+{
+ fileContents = new List<AIContent>();
+
+ foreach (var uploadedFile in uploadedFiles)
+ {
+ try
+ {
+ // دریافت Stream فایل از FileProvider
+ var fileDescriptor = await fileProvider.GetFileDescriptor(
+ id: uploadedFile.Id,
+ cancellationToken: cancellationToken
+ );
+
+ if (fileDescriptor == null || fileDescriptor.Stream == null)
+ {
+ continue;
+ }
+
+ // استخراج محتوا با استفاده از Composite Extractor
+ var extractionResult = await fileContentExtractor.ExtractRichAsync(
+ fileStream: fileDescriptor.Stream,
+ fileName: uploadedFile.FileName ?? uploadedFile.Name,
+ mimeType: fileDescriptor.MIMEType,
+ cancellationToken: cancellationToken
+ );
+
+ if (extractionResult.IsValid())
+ {
+ // تبدیل به AIContent
+ var contents = extractionResult.ToAIContents();
+ fileContents.AddRange(contents);
+
+ // ذخیره Metadata برای ردیابی
+ logger.LogInformation(
+ "File {FileName} extracted successfully. Text: {HasText}, Images: {ImageCount}, Fallback: {UsedFallback}",
+ uploadedFile.FileName,
+ !string.IsNullOrEmpty(extractionResult.Text),
+ extractionResult.Images.Count,
+ extractionResult.UsedImageFallback
+ );
+ }
+
+ fileDescriptor.Stream.Dispose();
+ }
+ catch (Exception ex)
+ {
+ logger.LogError(ex, "Failed to extract content from file: {FileName}", uploadedFile.FileName);
+ }
+ }
+}
+
+// تبدیل به ChatMessage با محتوای فایلها
+var promptChatMessage = promptMessage.ToChatMessages(fileContents);
+
+
+ MODIFY
+ xAiModels/Extensions/XAiModelsExtensions.cs
+
+
+
اصلاح متد ToChatMessages برای پشتیبانی از محتوای چندگانه:
+
+
public static ChatMessage ToChatMessages(
+ this XAiMessageDto source,
+ IList<AIContent> additionalContents = null
+)
+{
+ ChatMessage result = null;
+
+ if (!source.IsNullOrDefault())
+ {
+ var contents = new List<AIContent>();
+
+ // ۱. افزودن متن اصلی پیام
+ if (!string.IsNullOrWhiteSpace(source.Content))
+ {
+ contents.Add(new TextContent(source.Content));
+ }
+
+ // ۲. ✅ افزودن محتوای فایلهای ضمیمه (متن + تصویر + صوت)
+ if (additionalContents != null && additionalContents.HasChild())
+ {
+ contents.AddRange(additionalContents);
+ }
+
+ result = new ChatMessage
+ {
+ AuthorName =
+ source.Role == XAiChatRole.User &&
+ !source.Owner.IsNullOrDefault()
+ ? source.Owner.GetFullname()
+ : string.Empty,
+ Role = source.Role.ToChatRole(),
+ MessageId = source.Id.ToString(),
+ Contents = contents
+ };
+ }
+
+ return result;
+}
+
+
+
+
+
🔌 گام ۸: ثبت در DI و پیکربندی
+
+
+ MODIFY
+ xAiApi/Startup.cs
+
+
+
public void ConfigureServices(IServiceCollection services)
+{
+ // ... (سایر ثبتها)
+
+ // ✅ ثبت Extractor های فایل
+ services.AddSingleton<IXFileContentExtractor, XPlainTextFileContentExtractor>();
+ services.AddSingleton<IXFileContentExtractor, XDocxFileContentExtractor>();
+ services.AddSingleton<IXFileContentExtractor, XExcelFileContentExtractor>();
+
+ // ✅ Extractor های جدید
+ services.AddSingleton<IXFileContentExtractor>(sp =>
+ new XImageFileContentExtractor(enableOcrFallback: true));
+ services.AddSingleton<IXFileContentExtractor>(sp =>
+ new XAudioFileContentExtractor(
+ whisperModelPath: "Models/ggml-base.bin",
+ language: "fa"
+ ));
+ services.AddSingleton<IXFileContentExtractor, XPdfFileContentExtractor>();
+
+ // ✅ ثبت Composite Extractor
+ services.AddSingleton<IXFileContentExtractor, XFileContentExtractor>();
+
+ // ... (بقیه کدها)
+}
+
+
+
💡 پیکربندی در appsettings.json:
+
{
+ "AiApiConfiguration": {
+ "FileExtraction": {
+ "EnableOcrFallback": true,
+ "WhisperModelPath": "Models/ggml-base.bin",
+ "WhisperLanguage": "fa",
+ "MaxPdfPagesForImageFallback": 10,
+ "MinTextLengthThreshold": 50
+ }
+ }
+}
+
+
+
+
+
+
🔄 گام ۹: جریان کامل پردازش
+
+
+ 📤 آپلود فایل
+ →
+ 💾 ذخیره در FileService
+ →
+ 🔍 تشخیص MIME Type
+ →
+ 🎯 انتخاب Extractor مناسب
+ →
+ 📊 استخراج محتوا (Text/Image/Audio)
+ →
+ 🔄 Fallback در صورت نیاز
+ →
+ 🎨 تبدیل به AIContent
+ →
+ 🤖 ارسال به LLM
+
+
+
مثالهای کاربردی:
+
+
+ | نوع فایل |
+ استراتژی پردازش |
+ خروجی |
+
+
+ | 📄 PDF متنی |
+ استخراج متن با PdfPig |
+ Text |
+
+
+ | 📄 PDF اسکنشده |
+ تبدیل صفحات به PNG |
+ Images[] (Vision) |
+
+
+ | 🖼️ تصویر (PNG/JPG) |
+ ارسال مستقیم + OCR |
+ Image + Text (OCR) |
+
+
+ | 🎙️ صوت (MP3/WAV) |
+ Whisper Speech-to-Text |
+ AudioTranscript |
+
+
+ | 📝 Word (DOCX) |
+ استخراج متن با OpenXml |
+ Text |
+
+
+ | 📊 Excel (XLSX) |
+ استخراج جدول با ExcelDataReader |
+ Text (Tabular) |
+
+
+
+
+
+
+
📋 گام ۱۰: خلاصه تغییرات و پکیجها
+
+
فایلهای جدید:
+
+
+ | فایل |
+ توضیح |
+
+
+ xAiApi/Models/XFileExtractionResult.cs |
+ مدل خروجی غنی با پشتیبانی از Text + Images + Audio |
+
+
+ xAiApi/Interfaces/Extractors/IXImageFileContentExtractor.cs |
+ Interface برای Image Extractor |
+
+
+ xAiApi/Interfaces/Extractors/IXAudioFileContentExtractor.cs |
+ Interface برای Audio Extractor |
+
+
+ xAiApi/Providers/Extractors/XImageFileContentExtractor.cs |
+ پیادهسازی Image Extractor با Vision + OCR |
+
+
+ xAiApi/Providers/Extractors/XAudioFileContentExtractor.cs |
+ پیادهسازی Audio Extractor با Whisper |
+
+
+
+
فایلهای اصلاحشده:
+
+
+ | فایل |
+ تغییر |
+
+
+ xAiApi/Interfaces/Extractors/IXFileContentExtractor.cs |
+ افزودن متد ExtractRichAsync |
+
+
+ xAiApi/Providers/Extractors/XPdfFileContentExtractor.cs |
+ افزودن Fallback تصویری برای PDF های اسکنشده |
+
+
+ xAiApi/Providers/XAIServiceBase.cs |
+ استفاده از ExtractRichAsync و پردازش چندوجهی |
+
+
+ xAiModels/Extensions/XAiModelsExtensions.cs |
+ پشتیبانی از additionalContents در ToChatMessages |
+
+
+ xAiApi/Startup.cs |
+ ثبت Extractor های جدید در DI |
+
+
+
+
پکیجهای NuGet مورد نیاز:
+
+
+ | پکیج |
+ کاربرد |
+ وضعیت |
+
+
+ UglyToad.PdfPig |
+ استخراج متن و تصویر از PDF |
+ ✅ نصب شده |
+
+
+ DocumentFormat.OpenXml |
+ استخراج متن از DOCX |
+ ✅ نصب شده |
+
+
+ ExcelDataReader |
+ استخراج متن از Excel |
+ ✅ نصب شده |
+
+
+ System.Drawing.Common |
+ کار با Bitmap برای PDF به تصویر |
+ ❌ نیاز به نصب |
+
+
+ Whisper.net |
+ Speech-to-Text محلی |
+ ❌ نیاز به نصب |
+
+
+ Whisper.net.Runtime |
+ Runtime برای Whisper |
+ ❌ نیاز به نصب |
+
+
+ NAudio |
+ تبدیل فرمت صوتی |
+ ❌ نیاز به نصب |
+
+
+ Tesseract (اختیاری) |
+ OCR برای تصاویر |
+ ⚠️ اختیاری |
+
+
+
+
+
✅ دستاوردهای این ارتقا:
+
+ - 🎨 پشتیبانی چندوجهی کامل: متن، تصویر، صوت
+ - 📄 PDF هوشمند: تشخیص خودکار PDF متنی/اسکنشده
+ - 🔄 Fallback خودکار: تبدیل به تصویر در صورت عدم وجود متن
+ - 🎙️ Speech-to-Text محلی: بدون نیاز به API خارجی
+ - 🖼️ Vision API Ready: سازگار با Gemini, GPT-4V, Qwen-VL
+ - 🔌 Backward Compatible: کدهای موجود بدون تغییر کار میکنند
+
+
+
+
+