Compare commits
1
Commits
4521a53734
...
c30d93be0b
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c30d93be0b |
@@ -0,0 +1,5 @@
|
||||
namespace xAiApi.Interfaces.Extractors
|
||||
{
|
||||
public interface IXDocxFileContentExtractor : IXFileContentExtractor
|
||||
{ }
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
namespace xAiApi.Interfaces.Extractors
|
||||
{
|
||||
public interface IXExcelFileContentExtractor : IXFileContentExtractor
|
||||
{ }
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
using System.IO;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using xAiModels.Models;
|
||||
|
||||
namespace xAiApi.Interfaces.Extractors
|
||||
{
|
||||
@@ -22,5 +23,20 @@ namespace xAiApi.Interfaces.Extractors
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
);
|
||||
|
||||
/// <summary>
|
||||
/// Extract content from stream as Rich Result ...
|
||||
/// </summary>
|
||||
/// <param name="fileStream"></param>
|
||||
/// <param name="fileName"></param>
|
||||
/// <param name="mimeType"></param>
|
||||
/// <param name="cancellationToken"></param>
|
||||
/// <returns></returns>
|
||||
Task<XFileExtractionResult> ExtractRichAsync(
|
||||
Stream fileStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
namespace xAiApi.Interfaces.Extractors
|
||||
{
|
||||
public interface IXImageFileContentExtractor : IXFileContentExtractor
|
||||
{ }
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Text;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
using DocumentFormat.OpenXml.Drawing;
|
||||
using DocumentFormat.OpenXml.Packaging;
|
||||
|
||||
namespace xAiApi.Providers.Extractors
|
||||
{
|
||||
/// <summary>
|
||||
/// Extracts text content from DOCX files using DocumentFormat.OpenXml ...
|
||||
/// </summary>
|
||||
public class XDocxFileContentExtractor : IXDocxFileContentExtractor
|
||||
{
|
||||
/// <summary>
|
||||
/// Supported MIME Types ...
|
||||
/// </summary>
|
||||
private static readonly string[] SupportedMimeTypes =
|
||||
[
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document"
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Check if this extractor supports the specified MIME type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
//
|
||||
return SupportedMimeTypes.Contains(
|
||||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||||
);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract text content from file stream ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
using var memoryStream = new MemoryStream();
|
||||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
memoryStream.Position = 0;
|
||||
|
||||
//
|
||||
var stringBuilder = new StringBuilder();
|
||||
|
||||
//
|
||||
using (var wordDocument = WordprocessingDocument.Open(memoryStream, false))
|
||||
{
|
||||
//
|
||||
var body = wordDocument.MainDocumentPart?.Document?.Body;
|
||||
if (body != null)
|
||||
{
|
||||
//
|
||||
var paragraphs = body.Elements<Paragraph>();
|
||||
foreach (var paragraph in paragraphs)
|
||||
{
|
||||
//
|
||||
var text = paragraph.InnerText?.Trim();
|
||||
if (!string.IsNullOrEmpty(text))
|
||||
{
|
||||
stringBuilder.AppendLine(text);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
//
|
||||
return stringBuilder.ToString();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,93 @@
|
||||
using System;
|
||||
using System.IO;
|
||||
using System.Data;
|
||||
using System.Linq;
|
||||
using System.Text;
|
||||
using ExcelDataReader;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
|
||||
namespace xAiApi.Providers.Extractors
|
||||
{
|
||||
public class XExcelFileContentExtractor : IXExcelFileContentExtractor
|
||||
{
|
||||
/// <summary>
|
||||
/// Supported MIME Types ...
|
||||
/// </summary>
|
||||
private static readonly string[] SupportedMimeTypes =
|
||||
[
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", // .xlsx
|
||||
"application/vnd.ms-excel" // .xls
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// Check if this extractor supports the specified MIME type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
//
|
||||
return SupportedMimeTypes.Contains(
|
||||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||||
);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract text content from file stream ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
Encoding.RegisterProvider(CodePagesEncodingProvider.Instance);
|
||||
|
||||
//
|
||||
using var memoryStream = new MemoryStream();
|
||||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
memoryStream.Position = 0;
|
||||
|
||||
//
|
||||
var stringBuilder = new StringBuilder();
|
||||
|
||||
//
|
||||
using (var reader = ExcelReaderFactory.CreateReader(memoryStream))
|
||||
{
|
||||
//
|
||||
var result = reader.AsDataSet();
|
||||
foreach (DataTable table in result.Tables)
|
||||
{
|
||||
//
|
||||
stringBuilder.AppendLine($"--- Sheet: {table.TableName} ---");
|
||||
|
||||
//
|
||||
var headers = new string[table.Columns.Count];
|
||||
for (int i = 0; i < table.Columns.Count; i++)
|
||||
{
|
||||
headers[i] = table.Columns[i].ColumnName;
|
||||
}
|
||||
stringBuilder.AppendLine(string.Join(" | ", headers));
|
||||
stringBuilder.AppendLine(new string('-', 50));
|
||||
|
||||
//
|
||||
foreach (DataRow row in table.Rows)
|
||||
{
|
||||
//
|
||||
var rowValues = new string[row.ItemArray.Length];
|
||||
for (int i = 0; i < row.ItemArray.Length; i++)
|
||||
{
|
||||
rowValues[i] = row.ItemArray[i]?.ToString()?.Trim() ?? string.Empty;
|
||||
}
|
||||
stringBuilder.AppendLine(string.Join(" | ", rowValues));
|
||||
}
|
||||
stringBuilder.AppendLine();
|
||||
}
|
||||
}
|
||||
|
||||
//
|
||||
return stringBuilder.ToString();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,141 @@
|
||||
using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Threading;
|
||||
using xAiModels.Models;
|
||||
using System.Threading.Tasks;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
|
||||
namespace xAiApi.Providers.Extractors
|
||||
{
|
||||
public class XImageFileContentExtractor : IXImageFileContentExtractor
|
||||
{
|
||||
/// <summary>
|
||||
/// Supported MIME Types ...
|
||||
/// </summary>
|
||||
private static readonly string[] SupportedMimeTypes =
|
||||
[
|
||||
"image/png",
|
||||
"image/jpeg",
|
||||
"image/jpg",
|
||||
"image/gif",
|
||||
"image/webp",
|
||||
"image/bmp"
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// آیا OCR فعال است (برای مدلهای غیر Vision) ...
|
||||
/// </summary>
|
||||
private readonly bool _enableOcrFallback;
|
||||
|
||||
public XImageFileContentExtractor(bool enableOcrFallback = true)
|
||||
{
|
||||
_enableOcrFallback = enableOcrFallback;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Check if this extractor supports the specified MIME type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
//
|
||||
return SupportedMimeTypes.Contains(
|
||||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||||
);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract text content from file stream ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
var result = await ExtractRichAsync(
|
||||
fileStream,
|
||||
"image",
|
||||
mimeType,
|
||||
cancellationToken
|
||||
);
|
||||
|
||||
//
|
||||
return result.Text;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract content from stream as Rich Result ...
|
||||
/// </summary>
|
||||
/// <param name="fileStream"></param>
|
||||
/// <param name="fileName"></param>
|
||||
/// <param name="mimeType"></param>
|
||||
/// <param name="cancellationToken"></param>
|
||||
/// <returns></returns>
|
||||
public async Task<XFileExtractionResult> ExtractRichAsync(
|
||||
Stream fileStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
var result = new XFileExtractionResult
|
||||
{
|
||||
FileName = fileName,
|
||||
MimeType = mimeType
|
||||
};
|
||||
|
||||
//
|
||||
using var memoryStream = new MemoryStream();
|
||||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
var imageBytes = memoryStream.ToArray();
|
||||
|
||||
//
|
||||
result.Images.Add(new XExtractedImage
|
||||
{
|
||||
Bytes = imageBytes,
|
||||
MimeType = mimeType,
|
||||
Description = $"Attached image: {fileName}"
|
||||
});
|
||||
|
||||
// OCR Fallback برای مدلهای متنی
|
||||
if (_enableOcrFallback)
|
||||
{
|
||||
//
|
||||
try
|
||||
{
|
||||
//
|
||||
var ocrText = await PerformOcrAsync(imageBytes, cancellationToken);
|
||||
if (!string.IsNullOrWhiteSpace(ocrText))
|
||||
{
|
||||
result.Text = ocrText;
|
||||
}
|
||||
}
|
||||
catch
|
||||
{
|
||||
// OCR failed, but image is still available for Vision
|
||||
}
|
||||
}
|
||||
|
||||
//
|
||||
return result;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Do OCR on Image for Text Extraction ...
|
||||
/// </summary>
|
||||
private async Task<string> PerformOcrAsync(
|
||||
byte[] imageBytes,
|
||||
CancellationToken cancellationToken
|
||||
)
|
||||
{
|
||||
//
|
||||
// TODO:
|
||||
// پیادهسازی با Tesseract یا هر کتابخانه OCR دیگر
|
||||
// در حال حاضر یک placeholder است
|
||||
return await Task.FromResult(string.Empty);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -46,6 +46,9 @@
|
||||
|
||||
<!-- Dependencies -->
|
||||
<ItemGroup>
|
||||
<PackageReference Include="DocumentFormat.OpenXml" Version="3.5.1" />
|
||||
<PackageReference Include="ExcelDataReader" Version="3.9.0" />
|
||||
<PackageReference Include="ExcelDataReader.DataSet" Version="3.9.0" />
|
||||
<PackageReference Include="OllamaSharp" Version="5.4.25" />
|
||||
<PackageReference Include="UglyToad.PdfPig" Version="1.7.0-custom-5" />
|
||||
<PackageReference Include="Microsoft.AspNetCore.Mvc.NewtonsoftJson" Version="3.1.11" />
|
||||
|
||||
Reference in New Issue
Block a user