diff --git a/Interfaces/Extractors/IXDocxFileContentExtractor.cs b/Interfaces/Extractors/IXDocxFileContentExtractor.cs
new file mode 100644
index 0000000..3ec8aaf
--- /dev/null
+++ b/Interfaces/Extractors/IXDocxFileContentExtractor.cs
@@ -0,0 +1,5 @@
+namespace xAiApi.Interfaces.Extractors
+{
+ public interface IXDocxFileContentExtractor : IXFileContentExtractor
+ { }
+}
\ No newline at end of file
diff --git a/Interfaces/Extractors/IXExcelFileContentExtractor.cs b/Interfaces/Extractors/IXExcelFileContentExtractor.cs
new file mode 100644
index 0000000..9711643
--- /dev/null
+++ b/Interfaces/Extractors/IXExcelFileContentExtractor.cs
@@ -0,0 +1,5 @@
+namespace xAiApi.Interfaces.Extractors
+{
+ public interface IXExcelFileContentExtractor : IXFileContentExtractor
+ { }
+}
\ No newline at end of file
diff --git a/Interfaces/Extractors/IXFileContentExtractor.cs b/Interfaces/Extractors/IXFileContentExtractor.cs
index 4194aa3..ccb62db 100644
--- a/Interfaces/Extractors/IXFileContentExtractor.cs
+++ b/Interfaces/Extractors/IXFileContentExtractor.cs
@@ -1,6 +1,7 @@
using System.IO;
using System.Threading;
using System.Threading.Tasks;
+using xAiModels.Models;
namespace xAiApi.Interfaces.Extractors
{
@@ -22,5 +23,20 @@ namespace xAiApi.Interfaces.Extractors
string mimeType,
CancellationToken cancellationToken = default
);
+
+ ///
+ /// Extract content from stream as Rich Result ...
+ ///
+ ///
+ ///
+ ///
+ ///
+ ///
+ Task ExtractRichAsync(
+ Stream fileStream,
+ string fileName,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ );
}
}
\ No newline at end of file
diff --git a/Interfaces/Extractors/IXImageFileContentExtractor.cs b/Interfaces/Extractors/IXImageFileContentExtractor.cs
new file mode 100644
index 0000000..2844cea
--- /dev/null
+++ b/Interfaces/Extractors/IXImageFileContentExtractor.cs
@@ -0,0 +1,5 @@
+namespace xAiApi.Interfaces.Extractors
+{
+ public interface IXImageFileContentExtractor : IXFileContentExtractor
+ { }
+}
\ No newline at end of file
diff --git a/Providers/Extractors/XDocxFileContentExtractor.cs b/Providers/Extractors/XDocxFileContentExtractor.cs
new file mode 100644
index 0000000..a67b138
--- /dev/null
+++ b/Providers/Extractors/XDocxFileContentExtractor.cs
@@ -0,0 +1,79 @@
+using System;
+using System.IO;
+using System.Linq;
+using System.Text;
+using System.Threading;
+using System.Threading.Tasks;
+using xAiApi.Interfaces.Extractors;
+using DocumentFormat.OpenXml.Drawing;
+using DocumentFormat.OpenXml.Packaging;
+
+namespace xAiApi.Providers.Extractors
+{
+ ///
+ /// Extracts text content from DOCX files using DocumentFormat.OpenXml ...
+ ///
+ public class XDocxFileContentExtractor : IXDocxFileContentExtractor
+ {
+ ///
+ /// Supported MIME Types ...
+ ///
+ private static readonly string[] SupportedMimeTypes =
+ [
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document"
+ ];
+
+ ///
+ /// Check if this extractor supports the specified MIME type ...
+ ///
+ public bool CanExtract(string mimeType)
+ {
+ //
+ return SupportedMimeTypes.Contains(
+ mimeType?.ToLowerInvariant() ?? string.Empty
+ );
+ }
+
+ ///
+ /// Extract text content from file stream ...
+ ///
+ public async Task ExtractAsync(
+ Stream fileStream,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ //
+ using var memoryStream = new MemoryStream();
+ await fileStream.CopyToAsync(memoryStream, cancellationToken);
+ memoryStream.Position = 0;
+
+ //
+ var stringBuilder = new StringBuilder();
+
+ //
+ using (var wordDocument = WordprocessingDocument.Open(memoryStream, false))
+ {
+ //
+ var body = wordDocument.MainDocumentPart?.Document?.Body;
+ if (body != null)
+ {
+ //
+ var paragraphs = body.Elements();
+ foreach (var paragraph in paragraphs)
+ {
+ //
+ var text = paragraph.InnerText?.Trim();
+ if (!string.IsNullOrEmpty(text))
+ {
+ stringBuilder.AppendLine(text);
+ }
+ }
+ }
+ }
+
+ //
+ return stringBuilder.ToString();
+ }
+ }
+}
\ No newline at end of file
diff --git a/Providers/Extractors/XExcelFileContentExtractor.cs b/Providers/Extractors/XExcelFileContentExtractor.cs
new file mode 100644
index 0000000..76326d4
--- /dev/null
+++ b/Providers/Extractors/XExcelFileContentExtractor.cs
@@ -0,0 +1,93 @@
+using System;
+using System.IO;
+using System.Data;
+using System.Linq;
+using System.Text;
+using ExcelDataReader;
+using System.Threading;
+using System.Threading.Tasks;
+using xAiApi.Interfaces.Extractors;
+
+namespace xAiApi.Providers.Extractors
+{
+ public class XExcelFileContentExtractor : IXExcelFileContentExtractor
+ {
+ ///
+ /// Supported MIME Types ...
+ ///
+ private static readonly string[] SupportedMimeTypes =
+ [
+ "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", // .xlsx
+ "application/vnd.ms-excel" // .xls
+ ];
+
+ ///
+ /// Check if this extractor supports the specified MIME type ...
+ ///
+ public bool CanExtract(string mimeType)
+ {
+ //
+ return SupportedMimeTypes.Contains(
+ mimeType?.ToLowerInvariant() ?? string.Empty
+ );
+ }
+
+ ///
+ /// Extract text content from file stream ...
+ ///
+ public async Task ExtractAsync(
+ Stream fileStream,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ //
+ Encoding.RegisterProvider(CodePagesEncodingProvider.Instance);
+
+ //
+ using var memoryStream = new MemoryStream();
+ await fileStream.CopyToAsync(memoryStream, cancellationToken);
+ memoryStream.Position = 0;
+
+ //
+ var stringBuilder = new StringBuilder();
+
+ //
+ using (var reader = ExcelReaderFactory.CreateReader(memoryStream))
+ {
+ //
+ var result = reader.AsDataSet();
+ foreach (DataTable table in result.Tables)
+ {
+ //
+ stringBuilder.AppendLine($"--- Sheet: {table.TableName} ---");
+
+ //
+ var headers = new string[table.Columns.Count];
+ for (int i = 0; i < table.Columns.Count; i++)
+ {
+ headers[i] = table.Columns[i].ColumnName;
+ }
+ stringBuilder.AppendLine(string.Join(" | ", headers));
+ stringBuilder.AppendLine(new string('-', 50));
+
+ //
+ foreach (DataRow row in table.Rows)
+ {
+ //
+ var rowValues = new string[row.ItemArray.Length];
+ for (int i = 0; i < row.ItemArray.Length; i++)
+ {
+ rowValues[i] = row.ItemArray[i]?.ToString()?.Trim() ?? string.Empty;
+ }
+ stringBuilder.AppendLine(string.Join(" | ", rowValues));
+ }
+ stringBuilder.AppendLine();
+ }
+ }
+
+ //
+ return stringBuilder.ToString();
+ }
+ }
+}
\ No newline at end of file
diff --git a/Providers/Extractors/XImageFileContentExtractor.cs b/Providers/Extractors/XImageFileContentExtractor.cs
new file mode 100644
index 0000000..4b53475
--- /dev/null
+++ b/Providers/Extractors/XImageFileContentExtractor.cs
@@ -0,0 +1,141 @@
+using System;
+using System.IO;
+using System.Linq;
+using System.Threading;
+using xAiModels.Models;
+using System.Threading.Tasks;
+using xAiApi.Interfaces.Extractors;
+
+namespace xAiApi.Providers.Extractors
+{
+ public class XImageFileContentExtractor : IXImageFileContentExtractor
+ {
+ ///
+ /// Supported MIME Types ...
+ ///
+ private static readonly string[] SupportedMimeTypes =
+ [
+ "image/png",
+ "image/jpeg",
+ "image/jpg",
+ "image/gif",
+ "image/webp",
+ "image/bmp"
+ ];
+
+ ///
+ /// آیا OCR فعال است (برای مدلهای غیر Vision) ...
+ ///
+ private readonly bool _enableOcrFallback;
+
+ public XImageFileContentExtractor(bool enableOcrFallback = true)
+ {
+ _enableOcrFallback = enableOcrFallback;
+ }
+
+ ///
+ /// Check if this extractor supports the specified MIME type ...
+ ///
+ public bool CanExtract(string mimeType)
+ {
+ //
+ return SupportedMimeTypes.Contains(
+ mimeType?.ToLowerInvariant() ?? string.Empty
+ );
+ }
+
+ ///
+ /// Extract text content from file stream ...
+ ///
+ public async Task ExtractAsync(
+ Stream fileStream,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ //
+ var result = await ExtractRichAsync(
+ fileStream,
+ "image",
+ mimeType,
+ cancellationToken
+ );
+
+ //
+ return result.Text;
+ }
+
+ ///
+ /// Extract content from stream as Rich Result ...
+ ///
+ ///
+ ///
+ ///
+ ///
+ ///
+ public async Task ExtractRichAsync(
+ Stream fileStream,
+ string fileName,
+ string mimeType,
+ CancellationToken cancellationToken = default
+ )
+ {
+ //
+ var result = new XFileExtractionResult
+ {
+ FileName = fileName,
+ MimeType = mimeType
+ };
+
+ //
+ using var memoryStream = new MemoryStream();
+ await fileStream.CopyToAsync(memoryStream, cancellationToken);
+ var imageBytes = memoryStream.ToArray();
+
+ //
+ result.Images.Add(new XExtractedImage
+ {
+ Bytes = imageBytes,
+ MimeType = mimeType,
+ Description = $"Attached image: {fileName}"
+ });
+
+ // OCR Fallback برای مدلهای متنی
+ if (_enableOcrFallback)
+ {
+ //
+ try
+ {
+ //
+ var ocrText = await PerformOcrAsync(imageBytes, cancellationToken);
+ if (!string.IsNullOrWhiteSpace(ocrText))
+ {
+ result.Text = ocrText;
+ }
+ }
+ catch
+ {
+ // OCR failed, but image is still available for Vision
+ }
+ }
+
+ //
+ return result;
+ }
+
+ ///
+ /// Do OCR on Image for Text Extraction ...
+ ///
+ private async Task PerformOcrAsync(
+ byte[] imageBytes,
+ CancellationToken cancellationToken
+ )
+ {
+ //
+ // TODO:
+ // پیادهسازی با Tesseract یا هر کتابخانه OCR دیگر
+ // در حال حاضر یک placeholder است
+ return await Task.FromResult(string.Empty);
+ }
+ }
+}
\ No newline at end of file
diff --git a/xAiApi.csproj b/xAiApi.csproj
index cd82fb1..24aa676 100644
--- a/xAiApi.csproj
+++ b/xAiApi.csproj
@@ -46,6 +46,9 @@
+
+
+