using System; using System.IO; using System.Linq; using System.Text; using System.Threading; using xAiModels.Models; using System.Threading.Tasks; using xAiApi.Interfaces.Extractors; using DocumentFormat.OpenXml.Drawing; using DocumentFormat.OpenXml.Packaging; namespace xAiApi.Providers.Extractors { /// /// Extracts text content from DOCX files using DocumentFormat.OpenXml ... /// public class XDocxFileContentExtractor : IXDocxFileContentExtractor { /// /// Supported MIME Types ... /// private static readonly string[] SupportedMimeTypes = [ "application/vnd.openxmlformats-officedocument.wordprocessingml.document" ]; /// /// Check if this extractor supports the specified MIME type ... /// public bool CanExtract(string mimeType) { // return SupportedMimeTypes.Contains( mimeType?.ToLowerInvariant() ?? string.Empty ); } /// /// Extract text content from file stream ... /// public async Task ExtractAsync( Stream fileStream, string mimeType, CancellationToken cancellationToken = default ) { // using var memoryStream = new MemoryStream(); await fileStream.CopyToAsync(memoryStream, cancellationToken); memoryStream.Position = 0; // var stringBuilder = new StringBuilder(); // using (var wordDocument = WordprocessingDocument.Open(memoryStream, false)) { // var body = wordDocument.MainDocumentPart?.Document?.Body; if (body != null) { // var paragraphs = body.Elements(); foreach (var paragraph in paragraphs) { // var text = paragraph.InnerText?.Trim(); if (!string.IsNullOrEmpty(text)) { stringBuilder.AppendLine(text); } } } } // return stringBuilder.ToString(); } /// /// Extract content from stream as Rich Result ... /// /// /// /// /// /// public async Task ExtractRichAsync( Stream fileStream, string fileName, string mimeType, CancellationToken cancellationToken = default ) { // var content = await ExtractAsync( mimeType: mimeType, fileStream: fileStream, cancellationToken: cancellationToken ); // var result = new XFileExtractionResult { Text = content, FileName = fileName, MimeType = mimeType }; // return result; } } }