1346 lines
52 KiB
HTML
1346 lines
52 KiB
HTML
<!DOCTYPE html>
|
||
<html lang="fa" dir="rtl">
|
||
<head>
|
||
<meta charset="UTF-8">
|
||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||
<title>ارتقای سرویس استخراج محتوا - پشتیبانی از تصویر، صوت و PDF Fallback - فن آوران ساحر علم</title>
|
||
<style>
|
||
:root {
|
||
--primary: #1e3a8a;
|
||
--secondary: #3b82f6;
|
||
--accent: #f59e0b;
|
||
--success: #10b981;
|
||
--danger: #ef4444;
|
||
--warning: #f97316;
|
||
--purple: #8b5cf6;
|
||
--bg-light: #f8fafc;
|
||
--bg-code: #1e293b;
|
||
--text-dark: #0f172a;
|
||
--text-muted: #64748b;
|
||
--border: #e2e8f0;
|
||
}
|
||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||
body {
|
||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||
color: var(--text-dark);
|
||
line-height: 1.8;
|
||
padding: 20px;
|
||
}
|
||
.container {
|
||
max-width: 1200px;
|
||
margin: 0 auto;
|
||
background: white;
|
||
border-radius: 16px;
|
||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||
overflow: hidden;
|
||
}
|
||
.header {
|
||
background: linear-gradient(135deg, var(--primary) 0%, var(--purple) 100%);
|
||
color: white;
|
||
padding: 40px;
|
||
text-align: center;
|
||
}
|
||
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
|
||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||
.meta-bar {
|
||
display: flex;
|
||
justify-content: space-between;
|
||
background: var(--bg-light);
|
||
padding: 15px 30px;
|
||
border-bottom: 2px solid var(--border);
|
||
flex-wrap: wrap;
|
||
gap: 15px;
|
||
}
|
||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||
.meta-item strong { color: var(--primary); }
|
||
.content { padding: 40px; }
|
||
.section {
|
||
margin-bottom: 35px;
|
||
padding: 25px;
|
||
background: var(--bg-light);
|
||
border-radius: 12px;
|
||
border-right: 5px solid var(--secondary);
|
||
}
|
||
.section h2 {
|
||
color: var(--primary);
|
||
font-size: 1.5em;
|
||
margin-bottom: 20px;
|
||
padding-bottom: 10px;
|
||
border-bottom: 2px solid var(--border);
|
||
}
|
||
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
|
||
.arch-grid {
|
||
display: grid;
|
||
grid-template-columns: repeat(auto-fit, minmax(280px, 1fr));
|
||
gap: 20px;
|
||
margin: 20px 0;
|
||
}
|
||
.arch-card {
|
||
background: white;
|
||
padding: 20px;
|
||
border-radius: 10px;
|
||
box-shadow: 0 4px 12px rgba(0,0,0,0.08);
|
||
border-top: 4px solid var(--secondary);
|
||
}
|
||
.arch-card h4 { color: var(--primary); margin-bottom: 12px; font-size: 1.1em; }
|
||
.arch-card ul { list-style: none; padding-right: 0; }
|
||
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; font-size: 0.95em; }
|
||
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
|
||
pre {
|
||
background: var(--bg-code);
|
||
color: #e2e8f0;
|
||
padding: 18px;
|
||
border-radius: 8px;
|
||
overflow-x: auto;
|
||
direction: ltr;
|
||
text-align: left;
|
||
font-family: 'Consolas', monospace;
|
||
font-size: 0.83em;
|
||
margin: 15px 0;
|
||
border-right: 4px solid var(--accent);
|
||
}
|
||
code {
|
||
background: #fef3c7;
|
||
color: #92400e;
|
||
padding: 2px 8px;
|
||
border-radius: 4px;
|
||
font-family: 'Consolas', monospace;
|
||
font-size: 0.9em;
|
||
direction: ltr;
|
||
display: inline-block;
|
||
}
|
||
table {
|
||
width: 100%;
|
||
border-collapse: collapse;
|
||
margin: 15px 0;
|
||
background: white;
|
||
border-radius: 8px;
|
||
overflow: hidden;
|
||
}
|
||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
|
||
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
|
||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||
.highlight { background: linear-gradient(120deg, #fef3c7 0%, #fef3c7 100%); padding: 2px 6px; border-radius: 4px; font-weight: bold; }
|
||
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
|
||
.toc h3 { color: var(--primary); margin-bottom: 15px; }
|
||
.toc ol { padding-right: 25px; }
|
||
.toc li { padding: 6px 0; }
|
||
.toc a { color: var(--secondary); text-decoration: none; }
|
||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||
.badge-critical { background: var(--danger); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||
.flow-diagram { background: white; padding: 25px; border-radius: 10px; margin: 20px 0; text-align: center; }
|
||
.flow-step { display: inline-block; background: var(--secondary); color: white; padding: 10px 18px; border-radius: 8px; margin: 5px; font-size: 0.88em; }
|
||
.flow-arrow { display: inline-block; color: var(--accent); font-size: 1.5em; margin: 0 8px; vertical-align: middle; }
|
||
.diff-old { background: #fee2e2; color: #991b1b; padding: 2px 4px; border-radius: 3px; text-decoration: line-through; }
|
||
.diff-new { background: #d1fae5; color: #065f46; padding: 2px 4px; border-radius: 3px; }
|
||
</style>
|
||
</head>
|
||
<body>
|
||
<div class="container">
|
||
|
||
<div class="header">
|
||
<h1>🎨 ارتقای سرویس استخراج محتوا</h1>
|
||
<div class="subtitle">پشتیبانی از تصویر، صوت و Fallback هوشمند برای PDF های اسکنشده</div>
|
||
</div>
|
||
|
||
<div class="meta-bar">
|
||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
|
||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||
</div>
|
||
|
||
<div class="content">
|
||
|
||
<div class="toc">
|
||
<h3>📑 فهرست مطالب</h3>
|
||
<ol>
|
||
<li><a href="#analysis">تحلیل وضعیت فعلی و شکافها</a></li>
|
||
<li><a href="#strategy">استراتژی معماری جدید</a></li>
|
||
<li><a href="#step1">گام ۱: بازطراحی Interface با خروجی چندگانه</a></li>
|
||
<li><a href="#step2">گام ۲: پیادهسازی ImageContentExtractor</a></li>
|
||
<li><a href="#step3">گام ۳: پیادهسازی AudioContentExtractor</a></li>
|
||
<li><a href="#step4">گام ۴: ارتقای PdfExtractor با Fallback تصویری</a></li>
|
||
<li><a href="#step5">گام ۵: یکپارچهسازی در XAIServiceBase</a></li>
|
||
<li><a href="#step6">گام ۶: ثبت در DI و پیکربندی</a></li>
|
||
<li><a href="#flow">جریان کامل پردازش</a></li>
|
||
<li><a href="#summary">خلاصه تغییرات و پکیجها</a></li>
|
||
</ol>
|
||
</div>
|
||
|
||
<!-- Section 1: Analysis -->
|
||
<div class="section" id="analysis">
|
||
<h2>🔍 گام ۱: تحلیل وضعیت فعلی و شکافها</h2>
|
||
|
||
<h3>۱.۱ وضعیت فعلی Extractor ها:</h3>
|
||
<table>
|
||
<tr>
|
||
<th>Extractor</th>
|
||
<th>وضعیت</th>
|
||
<th>نوع خروجی</th>
|
||
<th>شکاف</th>
|
||
</tr>
|
||
<tr>
|
||
<td><code>XPlainTextFileContentExtractor</code></td>
|
||
<td>✅ موجود</td>
|
||
<td>Text</td>
|
||
<td>-</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>XDocxFileContentExtractor</code></td>
|
||
<td>✅ موجود</td>
|
||
<td>Text</td>
|
||
<td>عدم پشتیبانی از تصاویر داخل DOCX</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>XExcelFileContentExtractor</code></td>
|
||
<td>✅ موجود</td>
|
||
<td>Text (Tabular)</td>
|
||
<td>-</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>XPdfFileContentExtractor</code></td>
|
||
<td>⚠️ ناقص</td>
|
||
<td>Text</td>
|
||
<td><span class="highlight">عدم Fallback برای PDF های اسکنشده</span></td>
|
||
</tr>
|
||
<tr>
|
||
<td>Image Extractor</td>
|
||
<td>❌ موجود نیست</td>
|
||
<td>-</td>
|
||
<td>نیاز به پیادهسازی (Vision API)</td>
|
||
</tr>
|
||
<tr>
|
||
<td>Audio Extractor</td>
|
||
<td>❌ موجود نیست</td>
|
||
<td>-</td>
|
||
<td>نیاز به پیادهسازی (Whisper)</td>
|
||
</tr>
|
||
</table>
|
||
|
||
<h3>۱.۲ مشکل اساسی در <code>IXFileContentExtractor</code> فعلی:</h3>
|
||
<div class="alert alert-danger">
|
||
<strong>⚠️ محدودیت طراحی فعلی:</strong>
|
||
<p>Interface فعلی فقط یک <code>string</code> برمیگرداند. این برای متن کافی است اما برای <strong>تصویر</strong> و <strong>صوت</strong> نیاز به ساختار دادهای غنیتر داریم که بتواند چندین نوع محتوا (Text + Images + Audio) را همزمان حمل کند.</p>
|
||
</div>
|
||
|
||
<pre>// ❌ Interface فعلی - محدود به string
|
||
public interface IXFileContentExtractor
|
||
{
|
||
bool CanExtract(string mimeType);
|
||
Task<string> ExtractAsync(Stream fileStream, string mimeType, ...);
|
||
}</pre>
|
||
</div>
|
||
|
||
<!-- Section 2: Strategy -->
|
||
<div class="section" id="strategy">
|
||
<h2>🎯 گام ۲: استراتژی معماری جدید</h2>
|
||
|
||
<div class="arch-grid">
|
||
<div class="arch-card">
|
||
<h4>📦 خروجی چندگانه (Multi-Content)</h4>
|
||
<ul>
|
||
<li>ایجاد <code>XFileExtractionResult</code></li>
|
||
<li>شامل: Text, Images[], AudioTranscript</li>
|
||
<li>سازگار با <code>AIContent</code></li>
|
||
</ul>
|
||
</div>
|
||
<div class="arch-card">
|
||
<h4>🖼️ پشتیبانی تصویر (Vision)</h4>
|
||
<ul>
|
||
<li>ارسال به عنوان <code>DataContent</code></li>
|
||
<li>سازگار با Gemini, GPT-4V, Qwen-VL</li>
|
||
<li>OCR محلی به عنوان Fallback</li>
|
||
</ul>
|
||
</div>
|
||
<div class="arch-card">
|
||
<h4>🎙️ پشتیبانی صوت (Whisper)</h4>
|
||
<ul>
|
||
<li>تبدیل Speech-to-Text</li>
|
||
<li>استفاده از Whisper محلی یا API</li>
|
||
<li>خروجی به صورت متن</li>
|
||
</ul>
|
||
</div>
|
||
<div class="arch-card">
|
||
<h4>📄 PDF Fallback هوشمند</h4>
|
||
<ul>
|
||
<li>استخراج متن ابتدا</li>
|
||
<li>اگر متن خالی/کمحجم → تبدیل به تصویر</li>
|
||
<li>ارسال صفحات به عنوان Vision</li>
|
||
</ul>
|
||
</div>
|
||
</div>
|
||
|
||
<div class="alert alert-success">
|
||
<strong>✅ مزیت کلیدی:</strong> این طراحی <strong>Backward Compatible</strong> است. کدهای موجود که از <code>ExtractAsync</code> استفاده میکنند، بدون تغییر کار خواهند کرد.
|
||
</div>
|
||
</div>
|
||
|
||
<!-- Step 1: Redesign Interface -->
|
||
<div class="section" id="step1">
|
||
<h2>🔧 گام ۳: بازطراحی Interface با خروجی چندگانه</h2>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-new">NEW</span>
|
||
<span class="path">xAiApi/Models/XFileExtractionResult.cs</span>
|
||
</div>
|
||
|
||
<pre>using System;
|
||
using System.Collections.Generic;
|
||
using System.Linq;
|
||
using xCommons.Extensions;
|
||
using Microsoft.Extensions.AI;
|
||
|
||
namespace xAiApi.Models
|
||
{
|
||
/// <summary>
|
||
/// نتیجه استخراج محتوا از فایل - میتواند شامل متن، تصاویر و صوت باشد ...
|
||
/// </summary>
|
||
public class XFileExtractionResult
|
||
{
|
||
/// <summary>
|
||
/// محتوای متنی استخراج شده ...
|
||
/// </summary>
|
||
public string Text { get; set; } = string.Empty;
|
||
|
||
/// <summary>
|
||
/// تصاویر استخراج شده (برای Vision Models) ...
|
||
/// </summary>
|
||
public IList<XExtractedImage> Images { get; set; } = [];
|
||
|
||
/// <summary>
|
||
/// متن تبدیل شده از صوت ...
|
||
/// </summary>
|
||
public string AudioTranscript { get; set; } = string.Empty;
|
||
|
||
/// <summary>
|
||
/// نام فایل اصلی ...
|
||
/// </summary>
|
||
public string FileName { get; set; }
|
||
|
||
/// <summary>
|
||
/// MIME Type فایل ...
|
||
/// </summary>
|
||
public string MimeType { get; set; }
|
||
|
||
/// <summary>
|
||
/// آیا استخراج با Fallback تصویری انجام شده است ...
|
||
/// </summary>
|
||
public bool UsedImageFallback { get; set; } = false;
|
||
|
||
/// <summary>
|
||
/// پیام خطا (در صورت بروز مشکل) ...
|
||
/// </summary>
|
||
public string ErrorMessage { get; set; }
|
||
|
||
/// <summary>
|
||
/// آیا نتیجه معتبر است ...
|
||
/// </summary>
|
||
public bool IsValid()
|
||
{
|
||
return !Text.IsNullOrEmpty() ||
|
||
Images.HasChild() ||
|
||
!AudioTranscript.IsNullOrEmpty();
|
||
}
|
||
|
||
/// <summary>
|
||
/// تبدیل به لیست AIContent برای ارسال به LLM ...
|
||
/// </summary>
|
||
public IList<AIContent> ToAIContents()
|
||
{
|
||
var result = new List<AIContent>();
|
||
|
||
// ۱. افزودن متن اصلی
|
||
if (!Text.IsNullOrEmpty())
|
||
{
|
||
result.Add(new TextContent(Text));
|
||
}
|
||
|
||
// ۲. افزودن تصاویر (برای Vision Models)
|
||
if (Images.HasChild())
|
||
{
|
||
foreach (var img in Images)
|
||
{
|
||
result.Add(new DataContent(img.Bytes, img.MimeType));
|
||
}
|
||
}
|
||
|
||
// ۳. افزودن transcript صوت
|
||
if (!AudioTranscript.IsNullOrEmpty())
|
||
{
|
||
result.Add(new TextContent($"[Audio Transcript]\n{AudioTranscript}"));
|
||
}
|
||
|
||
return result;
|
||
}
|
||
}
|
||
|
||
/// <summary>
|
||
/// تصویر استخراج شده از فایل ...
|
||
/// </summary>
|
||
public class XExtractedImage
|
||
{
|
||
/// <summary>
|
||
/// داده بایت تصویر ...
|
||
/// </summary>
|
||
public byte[] Bytes { get; set; }
|
||
|
||
/// <summary>
|
||
/// MIME Type تصویر ...
|
||
/// </summary>
|
||
public string MimeType { get; set; } = "image/png";
|
||
|
||
/// <summary>
|
||
/// شماره صفحه (برای PDF) ...
|
||
/// </summary>
|
||
public int? PageNumber { get; set; }
|
||
|
||
/// <summary>
|
||
/// توضیح تصویر ...
|
||
/// </summary>
|
||
public string Description { get; set; }
|
||
}
|
||
}</pre>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-modify">MODIFY</span>
|
||
<span class="path">xAiApi/Interfaces/Extractors/IXFileContentExtractor.cs</span>
|
||
</div>
|
||
|
||
<pre>using System.IO;
|
||
using System.Threading;
|
||
using System.Threading.Tasks;
|
||
using xAiApi.Models;
|
||
|
||
namespace xAiApi.Interfaces.Extractors
|
||
{
|
||
/// <summary>
|
||
/// سرویس استخراج محتوای فایل - نسخه ارتقا یافته با پشتیبانی چندوجهی ...
|
||
/// </summary>
|
||
public interface IXFileContentExtractor
|
||
{
|
||
/// <summary>
|
||
/// بررسی پشتیبانی از MIME Type مشخص ...
|
||
/// </summary>
|
||
bool CanExtract(string mimeType);
|
||
|
||
/// <summary>
|
||
/// استخراج محتوای فایل با خروجی غنی (متن + تصویر + صوت) ...
|
||
/// </summary>
|
||
Task<XFileExtractionResult> ExtractRichAsync(
|
||
Stream fileStream,
|
||
string fileName,
|
||
string mimeType,
|
||
CancellationToken cancellationToken = default
|
||
);
|
||
|
||
/// <summary>
|
||
/// استخراج محتوای متنی (سازگار با نسخه قبل) ...
|
||
/// </summary>
|
||
Task<string> ExtractAsync(
|
||
Stream fileStream,
|
||
string mimeType,
|
||
CancellationToken cancellationToken = default
|
||
);
|
||
}
|
||
}</pre>
|
||
</div>
|
||
|
||
<!-- Step 2: Image Extractor -->
|
||
<div class="section" id="step2">
|
||
<h2>🖼️ گام ۴: پیادهسازی ImageContentExtractor</h2>
|
||
<p>این Extractor تصاویر را برای مدلهای Vision آماده میکند. دو حالت پشتیبانی میشود:</p>
|
||
<ul style="padding-right: 25px; margin: 15px 0;">
|
||
<li><strong>حالت ۱ (Vision API):</strong> ارسال مستقیم به عنوان <code>DataContent</code> به مدلهای چندوجهی</li>
|
||
<li><strong>حالت ۲ (OCR Fallback):</strong> استخراج متن با Tesseract برای مدلهای متنی</li>
|
||
</ul>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-new">NEW</span>
|
||
<span class="path">xAiApi/Interfaces/Extractors/IXImageFileContentExtractor.cs</span>
|
||
</div>
|
||
|
||
<pre>namespace xAiApi.Interfaces.Extractors
|
||
{
|
||
/// <summary>
|
||
/// Extractor اختصاصی برای فایلهای تصویری ...
|
||
/// </summary>
|
||
public interface IXImageFileContentExtractor : IXFileContentExtractor
|
||
{ }
|
||
}</pre>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-new">NEW</span>
|
||
<span class="path">xAiApi/Providers/Extractors/XImageFileContentExtractor.cs</span>
|
||
</div>
|
||
|
||
<pre>using System;
|
||
using System.IO;
|
||
using System.Linq;
|
||
using System.Threading;
|
||
using System.Threading.Tasks;
|
||
using xAiApi.Models;
|
||
using xAiApi.Interfaces.Extractors;
|
||
|
||
namespace xAiApi.Providers.Extractors
|
||
{
|
||
/// <summary>
|
||
/// استخراج محتوا از فایلهای تصویری - پشتیبانی از Vision و OCR ...
|
||
/// </summary>
|
||
public class XImageFileContentExtractor : IXImageFileContentExtractor
|
||
{
|
||
private static readonly string[] SupportedMimeTypes =
|
||
[
|
||
"image/png",
|
||
"image/jpeg",
|
||
"image/jpg",
|
||
"image/gif",
|
||
"image/webp",
|
||
"image/bmp"
|
||
];
|
||
|
||
/// <summary>
|
||
/// آیا OCR فعال است (برای مدلهای غیر Vision) ...
|
||
/// </summary>
|
||
private readonly bool _enableOcrFallback;
|
||
|
||
public XImageFileContentExtractor(bool enableOcrFallback = true)
|
||
{
|
||
_enableOcrFallback = enableOcrFallback;
|
||
}
|
||
|
||
public bool CanExtract(string mimeType)
|
||
{
|
||
return SupportedMimeTypes.Contains(
|
||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||
);
|
||
}
|
||
|
||
public async Task<XFileExtractionResult> ExtractRichAsync(
|
||
Stream fileStream,
|
||
string fileName,
|
||
string mimeType,
|
||
CancellationToken cancellationToken = default
|
||
)
|
||
{
|
||
var result = new XFileExtractionResult
|
||
{
|
||
FileName = fileName,
|
||
MimeType = mimeType
|
||
};
|
||
|
||
// خواندن بایتهای تصویر
|
||
using var memoryStream = new MemoryStream();
|
||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||
var imageBytes = memoryStream.ToArray();
|
||
|
||
// افزودن به عنوان تصویر برای Vision Models
|
||
result.Images.Add(new XExtractedImage
|
||
{
|
||
Bytes = imageBytes,
|
||
MimeType = mimeType,
|
||
Description = $"Attached image: {fileName}"
|
||
});
|
||
|
||
// OCR Fallback برای مدلهای متنی
|
||
if (_enableOcrFallback)
|
||
{
|
||
try
|
||
{
|
||
var ocrText = await PerformOcrAsync(imageBytes, cancellationToken);
|
||
if (!string.IsNullOrWhiteSpace(ocrText))
|
||
{
|
||
result.Text = ocrText;
|
||
}
|
||
}
|
||
catch
|
||
{
|
||
// OCR failed, but image is still available for Vision
|
||
}
|
||
}
|
||
|
||
return result;
|
||
}
|
||
|
||
public async Task<string> ExtractAsync(
|
||
Stream fileStream,
|
||
string mimeType,
|
||
CancellationToken cancellationToken = default
|
||
)
|
||
{
|
||
var result = await ExtractRichAsync(
|
||
fileStream,
|
||
"image",
|
||
mimeType,
|
||
cancellationToken
|
||
);
|
||
return result.Text;
|
||
}
|
||
|
||
/// <summary>
|
||
/// انجام OCR روی تصویر - با استفاده از Tesseract ...
|
||
/// </summary>
|
||
private async Task<string> PerformOcrAsync(
|
||
byte[] imageBytes,
|
||
CancellationToken cancellationToken
|
||
)
|
||
{
|
||
// پیادهسازی با Tesseract یا هر کتابخانه OCR دیگر
|
||
// در حال حاضر یک placeholder است
|
||
return await Task.FromResult(string.Empty);
|
||
}
|
||
}
|
||
}</pre>
|
||
|
||
<div class="alert alert-warning">
|
||
<strong>⚠️ پکیجهای NuGet پیشنهادی برای OCR:</strong>
|
||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||
<li><code>Tesseract</code> - کتابخانه قدرتمند و متنباز</li>
|
||
<li><code>Microsoft.Azure.CognitiveServices.Vision.ComputerVision</code> - اگر از Azure استفاده میکنید</li>
|
||
</ul>
|
||
</div>
|
||
</div>
|
||
|
||
<!-- Step 3: Audio Extractor -->
|
||
<div class="section" id="step3">
|
||
<h2>🎙️ گام ۵: پیادهسازی AudioContentExtractor</h2>
|
||
<p>این Extractor فایلهای صوتی را با استفاده از <strong>Whisper</strong> به متن تبدیل میکند.</p>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-new">NEW</span>
|
||
<span class="path">xAiApi/Interfaces/Extractors/IXAudioFileContentExtractor.cs</span>
|
||
</div>
|
||
|
||
<pre>namespace xAiApi.Interfaces.Extractors
|
||
{
|
||
/// <summary>
|
||
/// Extractor اختصاصی برای فایلهای صوتی ...
|
||
/// </summary>
|
||
public interface IXAudioFileContentExtractor : IXFileContentExtractor
|
||
{ }
|
||
}</pre>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-new">NEW</span>
|
||
<span class="path">xAiApi/Providers/Extractors/XAudioFileContentExtractor.cs</span>
|
||
</div>
|
||
|
||
<pre>using System;
|
||
using System.IO;
|
||
using System.Linq;
|
||
using System.Threading;
|
||
using System.Threading.Tasks;
|
||
using xAiApi.Models;
|
||
using xAiApi.Interfaces.Extractors;
|
||
// Install-Package Whisper.net
|
||
// Install-Package Whisper.net.Runtime
|
||
using Whisper.net;
|
||
|
||
namespace xAiApi.Providers.Extractors
|
||
{
|
||
/// <summary>
|
||
/// استخراج متن از فایلهای صوتی با استفاده از Whisper ...
|
||
/// </summary>
|
||
public class XAudioFileContentExtractor : IXAudioFileContentExtractor
|
||
{
|
||
private static readonly string[] SupportedMimeTypes =
|
||
[
|
||
"audio/mpeg",
|
||
"audio/mp3",
|
||
"audio/wav",
|
||
"audio/wave",
|
||
"audio/ogg",
|
||
"audio/m4a",
|
||
"audio/mp4",
|
||
"audio/webm"
|
||
];
|
||
|
||
private readonly string _whisperModelPath;
|
||
private readonly string _language;
|
||
|
||
public XAudioFileContentExtractor(
|
||
string whisperModelPath = "Models/ggml-base.bin",
|
||
string language = "fa" // فارسی پیشفرض
|
||
)
|
||
{
|
||
_whisperModelPath = whisperModelPath;
|
||
_language = language;
|
||
}
|
||
|
||
public bool CanExtract(string mimeType)
|
||
{
|
||
return SupportedMimeTypes.Contains(
|
||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||
);
|
||
}
|
||
|
||
public async Task<XFileExtractionResult> ExtractRichAsync(
|
||
Stream fileStream,
|
||
string fileName,
|
||
string mimeType,
|
||
CancellationToken cancellationToken = default
|
||
)
|
||
{
|
||
var result = new XFileExtractionResult
|
||
{
|
||
FileName = fileName,
|
||
MimeType = mimeType
|
||
};
|
||
|
||
try
|
||
{
|
||
// کپی به MemoryStream برای Whisper
|
||
using var memoryStream = new MemoryStream();
|
||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||
memoryStream.Position = 0;
|
||
|
||
// تبدیل به فرمت WAV 16kHz (مورد نیاز Whisper)
|
||
using var wavStream = await ConvertToWavAsync(memoryStream, cancellationToken);
|
||
|
||
// انجام Speech-to-Text با Whisper
|
||
using var factory = WhisperFactory.FromPath(_whisperModelPath);
|
||
using var processor = factory.CreateBuilder()
|
||
.WithLanguage(_language)
|
||
.Build();
|
||
|
||
var segments = new System.Text.StringBuilder();
|
||
await foreach (var segment in processor.ProcessAsync(wavStream, cancellationToken))
|
||
{
|
||
segments.Append(segment.Text);
|
||
}
|
||
|
||
result.AudioTranscript = segments.ToString().Trim();
|
||
result.Text = result.AudioTranscript;
|
||
}
|
||
catch (Exception ex)
|
||
{
|
||
result.ErrorMessage = $"Audio extraction failed: {ex.Message}";
|
||
}
|
||
|
||
return result;
|
||
}
|
||
|
||
public async Task<string> ExtractAsync(
|
||
Stream fileStream,
|
||
string mimeType,
|
||
CancellationToken cancellationToken = default
|
||
)
|
||
{
|
||
var result = await ExtractRichAsync(
|
||
fileStream,
|
||
"audio",
|
||
mimeType,
|
||
cancellationToken
|
||
);
|
||
return result.AudioTranscript;
|
||
}
|
||
|
||
/// <summary>
|
||
/// تبدیل فرمت صوتی به WAV 16kHz (فرمت مورد نیاز Whisper) ...
|
||
/// </summary>
|
||
private async Task<Stream> ConvertToWavAsync(
|
||
Stream inputStream,
|
||
CancellationToken cancellationToken
|
||
)
|
||
{
|
||
// پیادهسازی با NAudio یا ffmpeg
|
||
// در حال حاضر یک placeholder است
|
||
return await Task.FromResult(inputStream);
|
||
}
|
||
}
|
||
}</pre>
|
||
|
||
<div class="alert alert-info">
|
||
<strong>💡 پکیجهای NuGet مورد نیاز:</strong>
|
||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||
<li><code>Whisper.net</code> - پیادهسازی C# از Whisper</li>
|
||
<li><code>Whisper.net.Runtime</code> - Runtime مورد نیاز</li>
|
||
<li><code>NAudio</code> - برای تبدیل فرمت صوتی</li>
|
||
</ul>
|
||
<p style="margin-top: 10px;"><strong>دانلود مدل Whisper:</strong> از <a href="https://huggingface.co/ggerganov/whisper.cpp/tree/main" target="_blank">HuggingFace</a> مدل <code>ggml-base.bin</code> یا <code>ggml-small.bin</code> را دانلود کنید.</p>
|
||
</div>
|
||
</div>
|
||
|
||
<!-- Step 4: PDF Fallback -->
|
||
<div class="section" id="step4">
|
||
<h2>📄 گام ۶: ارتقای PdfExtractor با Fallback تصویری</h2>
|
||
<p>این مهمترین بخش است. اگر PDF متن نداشت (اسکنشده بود)، صفحات را به تصویر تبدیل میکنیم.</p>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-modify">MODIFY</span>
|
||
<span class="path">xAiApi/Providers/Extractors/XPdfFileContentExtractor.cs</span>
|
||
</div>
|
||
|
||
<pre>using System;
|
||
using System.IO;
|
||
using System.Linq;
|
||
using System.Text;
|
||
using System.Threading;
|
||
using System.Threading.Tasks;
|
||
using UglyToad.PdfPig;
|
||
using UglyToad.PdfPig.Rendering;
|
||
using xAiApi.Models;
|
||
using xAiApi.Interfaces.Extractors;
|
||
|
||
namespace xAiApi.Providers
|
||
{
|
||
/// <summary>
|
||
/// استخراج محتوا از PDF - با Fallback هوشمند برای PDF های اسکنشده ...
|
||
/// </summary>
|
||
public class XPdfFileContentExtractor : IXPdfFileContentExtractor
|
||
{
|
||
private static readonly string[] SupportedMimeTypes = ["application/pdf"];
|
||
|
||
/// <summary>
|
||
/// حداقل تعداد کاراکتر برای تشخیص PDF متنی ...
|
||
/// </summary>
|
||
private const int MinTextLengthThreshold = 50;
|
||
|
||
/// <summary>
|
||
/// حداکثر تعداد صفحات برای تبدیل به تصویر ...
|
||
/// </summary>
|
||
private const int MaxPagesForImageFallback = 10;
|
||
|
||
public bool CanExtract(string mimeType)
|
||
{
|
||
return SupportedMimeTypes.Contains(
|
||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||
);
|
||
}
|
||
|
||
public async Task<XFileExtractionResult> ExtractRichAsync(
|
||
Stream fileStream,
|
||
string fileName,
|
||
string mimeType,
|
||
CancellationToken cancellationToken = default
|
||
)
|
||
{
|
||
var result = new XFileExtractionResult
|
||
{
|
||
FileName = fileName,
|
||
MimeType = mimeType
|
||
};
|
||
|
||
// کپی به MemoryStream
|
||
using var memoryStream = new MemoryStream();
|
||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||
memoryStream.Position = 0;
|
||
|
||
// مرحله ۱: تلاش برای استخراج متن
|
||
var extractedText = await ExtractTextAsync(memoryStream, cancellationToken);
|
||
|
||
// مرحله ۲: بررسی کیفیت متن استخراج شده
|
||
if (IsTextSufficient(extractedText))
|
||
{
|
||
// PDF متنی است
|
||
result.Text = extractedText;
|
||
}
|
||
else
|
||
{
|
||
// PDF اسکنشده است - Fallback به تصویر
|
||
memoryStream.Position = 0;
|
||
result = await FallbackToImageExtractionAsync(
|
||
memoryStream,
|
||
fileName,
|
||
mimeType,
|
||
cancellationToken
|
||
);
|
||
result.UsedImageFallback = true;
|
||
}
|
||
|
||
return result;
|
||
}
|
||
|
||
public async Task<string> ExtractAsync(
|
||
Stream fileStream,
|
||
string mimeType,
|
||
CancellationToken cancellationToken = default
|
||
)
|
||
{
|
||
var result = await ExtractRichAsync(
|
||
fileStream,
|
||
"document.pdf",
|
||
mimeType,
|
||
cancellationToken
|
||
);
|
||
return result.Text;
|
||
}
|
||
|
||
/// <summary>
|
||
/// استخراج متن از PDF ...
|
||
/// </summary>
|
||
private async Task<string> ExtractTextAsync(
|
||
Stream pdfStream,
|
||
CancellationToken cancellationToken
|
||
)
|
||
{
|
||
return await Task.Run(() =>
|
||
{
|
||
var sb = new StringBuilder();
|
||
using var document = PdfDocument.Open(pdfStream);
|
||
|
||
foreach (var page in document.GetPages())
|
||
{
|
||
var text = page.Text?.Trim();
|
||
if (!string.IsNullOrEmpty(text))
|
||
{
|
||
sb.AppendLine(text);
|
||
sb.AppendLine();
|
||
}
|
||
}
|
||
|
||
return sb.ToString();
|
||
}, cancellationToken);
|
||
}
|
||
|
||
/// <summary>
|
||
/// بررسی آیا متن استخراج شده کافی است ...
|
||
/// </summary>
|
||
private bool IsTextSufficient(string text)
|
||
{
|
||
if (string.IsNullOrWhiteSpace(text))
|
||
{
|
||
return false;
|
||
}
|
||
|
||
// حذف فاصلهها و بررسی طول
|
||
var cleanText = new string(text.Where(c => !char.IsWhiteSpace(c)).ToArray());
|
||
return cleanText.Length >= MinTextLengthThreshold;
|
||
}
|
||
|
||
/// <summary>
|
||
/// Fallback: تبدیل صفحات PDF به تصویر ...
|
||
/// </summary>
|
||
private async Task<XFileExtractionResult> FallbackToImageExtractionAsync(
|
||
Stream pdfStream,
|
||
string fileName,
|
||
string mimeType,
|
||
CancellationToken cancellationToken
|
||
)
|
||
{
|
||
var result = new XFileExtractionResult
|
||
{
|
||
FileName = fileName,
|
||
MimeType = mimeType,
|
||
UsedImageFallback = true
|
||
};
|
||
|
||
await Task.Run(() =>
|
||
{
|
||
using var document = PdfDocument.Open(pdfStream);
|
||
using var renderer = new PdfPigRenderer(document);
|
||
|
||
var pageCount = Math.Min(
|
||
document.NumberOfPages,
|
||
MaxPagesForImageFallback
|
||
);
|
||
|
||
for (int i = 0; i < pageCount; i++)
|
||
{
|
||
var page = document.GetPage(i);
|
||
var bitmap = renderer.Render(page);
|
||
|
||
// تبدیل Bitmap به PNG
|
||
using var memoryStream = new MemoryStream();
|
||
bitmap.Save(memoryStream, System.Drawing.Imaging.ImageFormat.Png);
|
||
var imageBytes = memoryStream.ToArray();
|
||
|
||
result.Images.Add(new XExtractedImage
|
||
{
|
||
Bytes = imageBytes,
|
||
MimeType = "image/png",
|
||
PageNumber = i + 1,
|
||
Description = $"Page {i + 1} of {fileName}"
|
||
});
|
||
}
|
||
}, cancellationToken);
|
||
|
||
return result;
|
||
}
|
||
}
|
||
}</pre>
|
||
|
||
<div class="alert alert-warning">
|
||
<strong>⚠️ پکیجهای NuGet اضافی برای PDF به تصویر:</strong>
|
||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||
<li><code>UglyToad.PdfPig</code> - قبلاً نصب شده</li>
|
||
<li><code>System.Drawing.Common</code> - برای کار با Bitmap</li>
|
||
</ul>
|
||
<p style="margin-top: 10px;"><strong>نکته:</strong> <code>PdfPigRenderer</code> در نسخههای جدید PdfPig وجود دارد. اگر نبود، از <code>Docnet</code> یا <code>PdfiumViewer</code> استفاده کنید.</p>
|
||
</div>
|
||
</div>
|
||
|
||
<!-- Step 5: Integration -->
|
||
<div class="section" id="step5">
|
||
<h2>🔗 گام ۷: یکپارچهسازی در XAIServiceBase</h2>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-modify">MODIFY</span>
|
||
<span class="path">xAiApi/Providers/XAIServiceBase.cs</span>
|
||
</div>
|
||
|
||
<p><strong>اصلاح متد <code>AskAsync</code> - بخش پردازش فایلها:</strong></p>
|
||
|
||
<pre>// در متد AskAsync، بعد از آپلود فایلها و AddReference:
|
||
|
||
// ✅ استخراج محتوای فایلها با خروجی غنی
|
||
IList<AIContent> fileContents = null;
|
||
if (uploadedFiles.HasChild())
|
||
{
|
||
fileContents = new List<AIContent>();
|
||
|
||
foreach (var uploadedFile in uploadedFiles)
|
||
{
|
||
try
|
||
{
|
||
// دریافت Stream فایل از FileProvider
|
||
var fileDescriptor = await fileProvider.GetFileDescriptor(
|
||
id: uploadedFile.Id,
|
||
cancellationToken: cancellationToken
|
||
);
|
||
|
||
if (fileDescriptor == null || fileDescriptor.Stream == null)
|
||
{
|
||
continue;
|
||
}
|
||
|
||
// استخراج محتوا با استفاده از Composite Extractor
|
||
var extractionResult = await fileContentExtractor.ExtractRichAsync(
|
||
fileStream: fileDescriptor.Stream,
|
||
fileName: uploadedFile.FileName ?? uploadedFile.Name,
|
||
mimeType: fileDescriptor.MIMEType,
|
||
cancellationToken: cancellationToken
|
||
);
|
||
|
||
if (extractionResult.IsValid())
|
||
{
|
||
// تبدیل به AIContent
|
||
var contents = extractionResult.ToAIContents();
|
||
fileContents.AddRange(contents);
|
||
|
||
// ذخیره Metadata برای ردیابی
|
||
logger.LogInformation(
|
||
"File {FileName} extracted successfully. Text: {HasText}, Images: {ImageCount}, Fallback: {UsedFallback}",
|
||
uploadedFile.FileName,
|
||
!string.IsNullOrEmpty(extractionResult.Text),
|
||
extractionResult.Images.Count,
|
||
extractionResult.UsedImageFallback
|
||
);
|
||
}
|
||
|
||
fileDescriptor.Stream.Dispose();
|
||
}
|
||
catch (Exception ex)
|
||
{
|
||
logger.LogError(ex, "Failed to extract content from file: {FileName}", uploadedFile.FileName);
|
||
}
|
||
}
|
||
}
|
||
|
||
// تبدیل به ChatMessage با محتوای فایلها
|
||
var promptChatMessage = promptMessage.ToChatMessages(fileContents);</pre>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-modify">MODIFY</span>
|
||
<span class="path">xAiModels/Extensions/XAiModelsExtensions.cs</span>
|
||
</div>
|
||
|
||
<p><strong>اصلاح متد <code>ToChatMessages</code> برای پشتیبانی از محتوای چندگانه:</strong></p>
|
||
|
||
<pre>public static ChatMessage ToChatMessages(
|
||
this XAiMessageDto source,
|
||
IList<AIContent> additionalContents = null
|
||
)
|
||
{
|
||
ChatMessage result = null;
|
||
|
||
if (!source.IsNullOrDefault())
|
||
{
|
||
var contents = new List<AIContent>();
|
||
|
||
// ۱. افزودن متن اصلی پیام
|
||
if (!string.IsNullOrWhiteSpace(source.Content))
|
||
{
|
||
contents.Add(new TextContent(source.Content));
|
||
}
|
||
|
||
// ۲. ✅ افزودن محتوای فایلهای ضمیمه (متن + تصویر + صوت)
|
||
if (additionalContents != null && additionalContents.HasChild())
|
||
{
|
||
contents.AddRange(additionalContents);
|
||
}
|
||
|
||
result = new ChatMessage
|
||
{
|
||
AuthorName =
|
||
source.Role == XAiChatRole.User &&
|
||
!source.Owner.IsNullOrDefault()
|
||
? source.Owner.GetFullname()
|
||
: string.Empty,
|
||
Role = source.Role.ToChatRole(),
|
||
MessageId = source.Id.ToString(),
|
||
Contents = contents
|
||
};
|
||
}
|
||
|
||
return result;
|
||
}</pre>
|
||
</div>
|
||
|
||
<!-- Step 6: DI Registration -->
|
||
<div class="section" id="step6">
|
||
<h2>🔌 گام ۸: ثبت در DI و پیکربندی</h2>
|
||
|
||
<div class="file-change">
|
||
<span class="badge-modify">MODIFY</span>
|
||
<span class="path">xAiApi/Startup.cs</span>
|
||
</div>
|
||
|
||
<pre>public void ConfigureServices(IServiceCollection services)
|
||
{
|
||
// ... (سایر ثبتها)
|
||
|
||
// ✅ ثبت Extractor های فایل
|
||
services.AddSingleton<IXFileContentExtractor, XPlainTextFileContentExtractor>();
|
||
services.AddSingleton<IXFileContentExtractor, XDocxFileContentExtractor>();
|
||
services.AddSingleton<IXFileContentExtractor, XExcelFileContentExtractor>();
|
||
|
||
// ✅ Extractor های جدید
|
||
services.AddSingleton<IXFileContentExtractor>(sp =>
|
||
new XImageFileContentExtractor(enableOcrFallback: true));
|
||
services.AddSingleton<IXFileContentExtractor>(sp =>
|
||
new XAudioFileContentExtractor(
|
||
whisperModelPath: "Models/ggml-base.bin",
|
||
language: "fa"
|
||
));
|
||
services.AddSingleton<IXFileContentExtractor, XPdfFileContentExtractor>();
|
||
|
||
// ✅ ثبت Composite Extractor
|
||
services.AddSingleton<IXFileContentExtractor, XFileContentExtractor>();
|
||
|
||
// ... (بقیه کدها)
|
||
}</pre>
|
||
|
||
<div class="alert alert-info">
|
||
<strong>💡 پیکربندی در <code>appsettings.json</code>:</strong>
|
||
<pre>{
|
||
"AiApiConfiguration": {
|
||
"FileExtraction": {
|
||
"EnableOcrFallback": true,
|
||
"WhisperModelPath": "Models/ggml-base.bin",
|
||
"WhisperLanguage": "fa",
|
||
"MaxPdfPagesForImageFallback": 10,
|
||
"MinTextLengthThreshold": 50
|
||
}
|
||
}
|
||
}</pre>
|
||
</div>
|
||
</div>
|
||
|
||
<!-- Flow -->
|
||
<div class="section" id="flow">
|
||
<h2>🔄 گام ۹: جریان کامل پردازش</h2>
|
||
|
||
<div class="flow-diagram">
|
||
<span class="flow-step">📤 آپلود فایل</span>
|
||
<span class="flow-arrow">→</span>
|
||
<span class="flow-step">💾 ذخیره در FileService</span>
|
||
<span class="flow-arrow">→</span>
|
||
<span class="flow-step">🔍 تشخیص MIME Type</span>
|
||
<span class="flow-arrow">→</span>
|
||
<span class="flow-step">🎯 انتخاب Extractor مناسب</span>
|
||
<span class="flow-arrow">→</span>
|
||
<span class="flow-step">📊 استخراج محتوا (Text/Image/Audio)</span>
|
||
<span class="flow-arrow">→</span>
|
||
<span class="flow-step">🔄 Fallback در صورت نیاز</span>
|
||
<span class="flow-arrow">→</span>
|
||
<span class="flow-step">🎨 تبدیل به AIContent</span>
|
||
<span class="flow-arrow">→</span>
|
||
<span class="flow-step">🤖 ارسال به LLM</span>
|
||
</div>
|
||
|
||
<h3>مثالهای کاربردی:</h3>
|
||
<table>
|
||
<tr>
|
||
<th>نوع فایل</th>
|
||
<th>استراتژی پردازش</th>
|
||
<th>خروجی</th>
|
||
</tr>
|
||
<tr>
|
||
<td>📄 PDF متنی</td>
|
||
<td>استخراج متن با PdfPig</td>
|
||
<td>Text</td>
|
||
</tr>
|
||
<tr>
|
||
<td>📄 PDF اسکنشده</td>
|
||
<td>تبدیل صفحات به PNG</td>
|
||
<td>Images[] (Vision)</td>
|
||
</tr>
|
||
<tr>
|
||
<td>🖼️ تصویر (PNG/JPG)</td>
|
||
<td>ارسال مستقیم + OCR</td>
|
||
<td>Image + Text (OCR)</td>
|
||
</tr>
|
||
<tr>
|
||
<td>🎙️ صوت (MP3/WAV)</td>
|
||
<td>Whisper Speech-to-Text</td>
|
||
<td>AudioTranscript</td>
|
||
</tr>
|
||
<tr>
|
||
<td>📝 Word (DOCX)</td>
|
||
<td>استخراج متن با OpenXml</td>
|
||
<td>Text</td>
|
||
</tr>
|
||
<tr>
|
||
<td>📊 Excel (XLSX)</td>
|
||
<td>استخراج جدول با ExcelDataReader</td>
|
||
<td>Text (Tabular)</td>
|
||
</tr>
|
||
</table>
|
||
</div>
|
||
|
||
<!-- Summary -->
|
||
<div class="section" id="summary">
|
||
<h2>📋 گام ۱۰: خلاصه تغییرات و پکیجها</h2>
|
||
|
||
<h3>فایلهای جدید:</h3>
|
||
<table>
|
||
<tr>
|
||
<th>فایل</th>
|
||
<th>توضیح</th>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiApi/Models/XFileExtractionResult.cs</code></td>
|
||
<td>مدل خروجی غنی با پشتیبانی از Text + Images + Audio</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiApi/Interfaces/Extractors/IXImageFileContentExtractor.cs</code></td>
|
||
<td>Interface برای Image Extractor</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiApi/Interfaces/Extractors/IXAudioFileContentExtractor.cs</code></td>
|
||
<td>Interface برای Audio Extractor</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiApi/Providers/Extractors/XImageFileContentExtractor.cs</code></td>
|
||
<td>پیادهسازی Image Extractor با Vision + OCR</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiApi/Providers/Extractors/XAudioFileContentExtractor.cs</code></td>
|
||
<td>پیادهسازی Audio Extractor با Whisper</td>
|
||
</tr>
|
||
</table>
|
||
|
||
<h3>فایلهای اصلاحشده:</h3>
|
||
<table>
|
||
<tr>
|
||
<th>فایل</th>
|
||
<th>تغییر</th>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiApi/Interfaces/Extractors/IXFileContentExtractor.cs</code></td>
|
||
<td>افزودن متد <code>ExtractRichAsync</code></td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiApi/Providers/Extractors/XPdfFileContentExtractor.cs</code></td>
|
||
<td>افزودن Fallback تصویری برای PDF های اسکنشده</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiApi/Providers/XAIServiceBase.cs</code></td>
|
||
<td>استفاده از <code>ExtractRichAsync</code> و پردازش چندوجهی</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiModels/Extensions/XAiModelsExtensions.cs</code></td>
|
||
<td>پشتیبانی از <code>additionalContents</code> در <code>ToChatMessages</code></td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>xAiApi/Startup.cs</code></td>
|
||
<td>ثبت Extractor های جدید در DI</td>
|
||
</tr>
|
||
</table>
|
||
|
||
<h3>پکیجهای NuGet مورد نیاز:</h3>
|
||
<table>
|
||
<tr>
|
||
<th>پکیج</th>
|
||
<th>کاربرد</th>
|
||
<th>وضعیت</th>
|
||
</tr>
|
||
<tr>
|
||
<td><code>UglyToad.PdfPig</code></td>
|
||
<td>استخراج متن و تصویر از PDF</td>
|
||
<td>✅ نصب شده</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>DocumentFormat.OpenXml</code></td>
|
||
<td>استخراج متن از DOCX</td>
|
||
<td>✅ نصب شده</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>ExcelDataReader</code></td>
|
||
<td>استخراج متن از Excel</td>
|
||
<td>✅ نصب شده</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>System.Drawing.Common</code></td>
|
||
<td>کار با Bitmap برای PDF به تصویر</td>
|
||
<td>❌ نیاز به نصب</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>Whisper.net</code></td>
|
||
<td>Speech-to-Text محلی</td>
|
||
<td>❌ نیاز به نصب</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>Whisper.net.Runtime</code></td>
|
||
<td>Runtime برای Whisper</td>
|
||
<td>❌ نیاز به نصب</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>NAudio</code></td>
|
||
<td>تبدیل فرمت صوتی</td>
|
||
<td>❌ نیاز به نصب</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code>Tesseract</code> (اختیاری)</td>
|
||
<td>OCR برای تصاویر</td>
|
||
<td>⚠️ اختیاری</td>
|
||
</tr>
|
||
</table>
|
||
|
||
<div class="alert alert-success">
|
||
<strong>✅ دستاوردهای این ارتقا:</strong>
|
||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||
<li>🎨 <strong>پشتیبانی چندوجهی کامل:</strong> متن، تصویر، صوت</li>
|
||
<li>📄 <strong>PDF هوشمند:</strong> تشخیص خودکار PDF متنی/اسکنشده</li>
|
||
<li>🔄 <strong>Fallback خودکار:</strong> تبدیل به تصویر در صورت عدم وجود متن</li>
|
||
<li>🎙️ <strong>Speech-to-Text محلی:</strong> بدون نیاز به API خارجی</li>
|
||
<li>🖼️ <strong>Vision API Ready:</strong> سازگار با Gemini, GPT-4V, Qwen-VL</li>
|
||
<li>🔌 <strong>Backward Compatible:</strong> کدهای موجود بدون تغییر کار میکنند</li>
|
||
</ul>
|
||
</div>
|
||
</div>
|
||
|
||
</div>
|
||
|
||
<div class="footer">
|
||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
|
||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||
🎨 ارتقای سرویس استخراج محتوا - پشتیبانی از تصویر، صوت و PDF Fallback
|
||
</p>
|
||
</div>
|
||
|
||
</div>
|
||
</body>
|
||
</html> |