last ...
This commit is contained in:
@@ -0,0 +1,466 @@
|
|||||||
|
<!DOCTYPE html>
|
||||||
|
<html lang="fa" dir="rtl">
|
||||||
|
<head>
|
||||||
|
<meta charset="UTF-8">
|
||||||
|
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||||
|
<title>افزودن پشتیبانی DOCX و Excel به FileContentExtractor - فن آوران ساحر علم</title>
|
||||||
|
<style>
|
||||||
|
:root {
|
||||||
|
--primary: #1e3a8a;
|
||||||
|
--secondary: #3b82f6;
|
||||||
|
--accent: #f59e0b;
|
||||||
|
--success: #10b981;
|
||||||
|
--danger: #ef4444;
|
||||||
|
--warning: #f97316;
|
||||||
|
--bg-light: #f8fafc;
|
||||||
|
--bg-code: #1e293b;
|
||||||
|
--text-dark: #0f172a;
|
||||||
|
--text-muted: #64748b;
|
||||||
|
--border: #e2e8f0;
|
||||||
|
}
|
||||||
|
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||||
|
body {
|
||||||
|
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||||
|
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||||
|
color: var(--text-dark);
|
||||||
|
line-height: 1.8;
|
||||||
|
padding: 20px;
|
||||||
|
}
|
||||||
|
.container {
|
||||||
|
max-width: 1200px;
|
||||||
|
margin: 0 auto;
|
||||||
|
background: white;
|
||||||
|
border-radius: 16px;
|
||||||
|
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||||
|
overflow: hidden;
|
||||||
|
}
|
||||||
|
.header {
|
||||||
|
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
|
||||||
|
color: white;
|
||||||
|
padding: 40px;
|
||||||
|
text-align: center;
|
||||||
|
position: relative;
|
||||||
|
}
|
||||||
|
.header h1 { font-size: 2.2em; margin-bottom: 10px; position: relative; }
|
||||||
|
.header .subtitle { font-size: 1.1em; opacity: 0.95; position: relative; }
|
||||||
|
.meta-bar {
|
||||||
|
display: flex;
|
||||||
|
justify-content: space-between;
|
||||||
|
background: var(--bg-light);
|
||||||
|
padding: 15px 30px;
|
||||||
|
border-bottom: 2px solid var(--border);
|
||||||
|
flex-wrap: wrap;
|
||||||
|
gap: 15px;
|
||||||
|
}
|
||||||
|
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||||
|
.meta-item strong { color: var(--primary); }
|
||||||
|
.content { padding: 40px; }
|
||||||
|
.section {
|
||||||
|
margin-bottom: 35px;
|
||||||
|
padding: 25px;
|
||||||
|
background: var(--bg-light);
|
||||||
|
border-radius: 12px;
|
||||||
|
border-right: 5px solid var(--secondary);
|
||||||
|
}
|
||||||
|
.section h2 {
|
||||||
|
color: var(--primary);
|
||||||
|
font-size: 1.6em;
|
||||||
|
margin-bottom: 20px;
|
||||||
|
padding-bottom: 10px;
|
||||||
|
border-bottom: 2px solid var(--border);
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 10px;
|
||||||
|
}
|
||||||
|
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
|
||||||
|
.step {
|
||||||
|
background: white;
|
||||||
|
padding: 20px;
|
||||||
|
margin: 15px 0;
|
||||||
|
border-radius: 10px;
|
||||||
|
box-shadow: 0 2px 8px rgba(0,0,0,0.05);
|
||||||
|
border-right: 4px solid var(--accent);
|
||||||
|
}
|
||||||
|
.step-number {
|
||||||
|
display: inline-block;
|
||||||
|
background: var(--accent);
|
||||||
|
color: white;
|
||||||
|
width: 32px;
|
||||||
|
height: 32px;
|
||||||
|
border-radius: 50%;
|
||||||
|
text-align: center;
|
||||||
|
line-height: 32px;
|
||||||
|
font-weight: bold;
|
||||||
|
margin-left: 10px;
|
||||||
|
}
|
||||||
|
.arch-grid {
|
||||||
|
display: grid;
|
||||||
|
grid-template-columns: repeat(auto-fit, minmax(280px, 1fr));
|
||||||
|
gap: 20px;
|
||||||
|
margin: 20px 0;
|
||||||
|
}
|
||||||
|
.arch-card {
|
||||||
|
background: white;
|
||||||
|
padding: 20px;
|
||||||
|
border-radius: 10px;
|
||||||
|
box-shadow: 0 4px 12px rgba(0,0,0,0.08);
|
||||||
|
border-top: 4px solid var(--secondary);
|
||||||
|
transition: transform 0.3s;
|
||||||
|
}
|
||||||
|
.arch-card:hover { transform: translateY(-5px); }
|
||||||
|
.arch-card h4 { color: var(--primary); margin-bottom: 12px; font-size: 1.15em; }
|
||||||
|
.arch-card ul { list-style: none; padding-right: 0; }
|
||||||
|
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; font-size: 0.95em; }
|
||||||
|
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
|
||||||
|
.tech-badge {
|
||||||
|
display: inline-block;
|
||||||
|
background: var(--secondary);
|
||||||
|
color: white;
|
||||||
|
padding: 4px 12px;
|
||||||
|
border-radius: 20px;
|
||||||
|
font-size: 0.85em;
|
||||||
|
margin: 3px;
|
||||||
|
}
|
||||||
|
.tech-badge.primary { background: var(--primary); }
|
||||||
|
.tech-badge.success { background: var(--success); }
|
||||||
|
.tech-badge.accent { background: var(--accent); }
|
||||||
|
pre {
|
||||||
|
background: var(--bg-code);
|
||||||
|
color: #e2e8f0;
|
||||||
|
padding: 18px;
|
||||||
|
border-radius: 8px;
|
||||||
|
overflow-x: auto;
|
||||||
|
direction: ltr;
|
||||||
|
text-align: left;
|
||||||
|
font-family: 'Consolas', 'Courier New', monospace;
|
||||||
|
font-size: 0.85em;
|
||||||
|
margin: 15px 0;
|
||||||
|
border-right: 4px solid var(--accent);
|
||||||
|
}
|
||||||
|
code {
|
||||||
|
background: #fef3c7;
|
||||||
|
color: #92400e;
|
||||||
|
padding: 2px 8px;
|
||||||
|
border-radius: 4px;
|
||||||
|
font-family: 'Consolas', monospace;
|
||||||
|
font-size: 0.9em;
|
||||||
|
direction: ltr;
|
||||||
|
display: inline-block;
|
||||||
|
}
|
||||||
|
table {
|
||||||
|
width: 100%;
|
||||||
|
border-collapse: collapse;
|
||||||
|
margin: 15px 0;
|
||||||
|
background: white;
|
||||||
|
border-radius: 8px;
|
||||||
|
overflow: hidden;
|
||||||
|
box-shadow: 0 2px 8px rgba(0,0,0,0.05);
|
||||||
|
}
|
||||||
|
th { background: var(--primary); color: white; padding: 12px; text-align: right; font-weight: bold; }
|
||||||
|
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||||
|
tr:hover { background: var(--bg-light); }
|
||||||
|
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||||
|
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||||
|
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||||
|
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
|
||||||
|
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; margin-top: 40px; }
|
||||||
|
.footer p { margin: 5px 0; }
|
||||||
|
.highlight { background: linear-gradient(120deg, #fef3c7 0%, #fef3c7 100%); padding: 2px 6px; border-radius: 4px; font-weight: bold; }
|
||||||
|
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
|
||||||
|
.toc h3 { color: var(--primary); margin-bottom: 15px; }
|
||||||
|
.toc ol { padding-right: 25px; }
|
||||||
|
.toc li { padding: 6px 0; }
|
||||||
|
.toc a { color: var(--secondary); text-decoration: none; transition: color 0.2s; }
|
||||||
|
.toc a:hover { color: var(--primary); text-decoration: underline; }
|
||||||
|
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||||
|
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||||
|
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||||
|
</style>
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
<div class="container">
|
||||||
|
|
||||||
|
<div class="header">
|
||||||
|
<h1>📄 افزودن پشتیبانی DOCX و Excel به FileContentExtractor</h1>
|
||||||
|
<div class="subtitle">تکمیل زنجیره استخراج محتوا برای فرمتهای آفیس در معماری xAiApi</div>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<div class="meta-bar">
|
||||||
|
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||||
|
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||||
|
<div class="meta-item">📅 <strong>تاریخ تهیه مستند:</strong> چهارشنبه ۸ مهر ۱۴۰۵</div>
|
||||||
|
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi / xAiService</div>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<div class="content">
|
||||||
|
|
||||||
|
<div class="toc">
|
||||||
|
<h3>📑 فهرست مطالب</h3>
|
||||||
|
<ol>
|
||||||
|
<li><a href="#step1">گام ۱: نصب پکیجهای NuGet مورد نیاز</a></li>
|
||||||
|
<li><a href="#step2">گام ۲: پیادهسازی DocxContentExtractor</a></li>
|
||||||
|
<li><a href="#step3">گام ۳: پیادهسازی ExcelContentExtractor</a></li>
|
||||||
|
<li><a href="#step4">گام ۴: ثبت سرویسها در Dependency Injection</a></li>
|
||||||
|
<li><a href="#step5">گام ۵: نکات کلیدی و ملاحظات عملکردی</a></li>
|
||||||
|
</ol>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<!-- Section 1: NuGet -->
|
||||||
|
<div class="section" id="step1">
|
||||||
|
<h2>📦 گام ۱: نصب پکیجهای NuGet مورد نیاز</h2>
|
||||||
|
<p>برای استخراج متن از فایلهای Office بدون نیاز به نصب نرمافزار Microsoft Office روی سرور، از کتابخانههای سبک و قدرتمند زیر استفاده میکنیم:</p>
|
||||||
|
|
||||||
|
<div class="step">
|
||||||
|
<span class="step-number">۱</span>
|
||||||
|
<strong>دستورات نصب در Package Manager Console (پروژه xAiService یا xAiApi):</strong>
|
||||||
|
<pre># برای پردازش فایلهای Word (DOCX)
|
||||||
|
Install-Package DocumentFormat.OpenXml
|
||||||
|
|
||||||
|
# برای پردازش فایلهای Excel (XLSX و XLS)
|
||||||
|
Install-Package ExcelDataReader
|
||||||
|
Install-Package ExcelDataReader.DataSet</pre>
|
||||||
|
</div>
|
||||||
|
<div class="alert alert-info">
|
||||||
|
<strong>💡 چرا این پکیجها؟</strong><br>
|
||||||
|
<code>DocumentFormat.OpenXml</code> استاندارد رسمی مایکروسافت است و بسیار پایدار میباشد. <code>ExcelDataReader</code> فوقالعاده سریع است و نیاز به COM Interop ندارد، که آن را برای محیطهای سروری (Server-side) ایدهآل میکند.
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<!-- Section 2: DOCX -->
|
||||||
|
<div class="section" id="step2">
|
||||||
|
<h2>📝 گام ۲: پیادهسازی DocxContentExtractor</h2>
|
||||||
|
<p>این کلاس مسئول استخراج متن از فایلهای <code>.docx</code> است. نکته مهم در اینجا این است که استریم ورودی را به <code>MemoryStream</code> تبدیل میکنیم، زیرا <code>OpenXml</code> به یک استریم قابل جستجو (Seekable) نیاز دارد.</p>
|
||||||
|
|
||||||
|
<div class="file-change">
|
||||||
|
<span class="badge-new">NEW</span>
|
||||||
|
<span class="path">xAiService/Providers/DocxContentExtractor.cs</span>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<pre>using System;
|
||||||
|
using System.IO;
|
||||||
|
using System.Linq;
|
||||||
|
using System.Text;
|
||||||
|
using System.Threading;
|
||||||
|
using System.Threading.Tasks;
|
||||||
|
using xAiService.Interfaces;
|
||||||
|
using DocumentFormat.OpenXml.Packaging;
|
||||||
|
using DocumentFormat.OpenXml.Wordprocessing;
|
||||||
|
|
||||||
|
namespace xAiService.Providers
|
||||||
|
{
|
||||||
|
/// <summary>
|
||||||
|
/// Extracts text content from DOCX files using DocumentFormat.OpenXml ...
|
||||||
|
/// </summary>
|
||||||
|
public class DocxContentExtractor : IFileContentExtractor
|
||||||
|
{
|
||||||
|
private const string DocxMimeType = "application/vnd.openxmlformats-officedocument.wordprocessingml.document";
|
||||||
|
|
||||||
|
public bool CanExtract(string mimeType)
|
||||||
|
{
|
||||||
|
return string.Equals(mimeType, DocxMimeType, StringComparison.OrdinalIgnoreCase);
|
||||||
|
}
|
||||||
|
|
||||||
|
public async Task<string> ExtractAsync(
|
||||||
|
Stream fileStream,
|
||||||
|
string mimeType,
|
||||||
|
CancellationToken cancellationToken = default
|
||||||
|
)
|
||||||
|
{
|
||||||
|
// نکته حیاتی: OpenXml به استریم قابل جستجو (Seekable) نیاز دارد.
|
||||||
|
// استریم IFormFile ممکن است Seekable نباشد، بنابراین آن را در حافظه کپی میکنیم.
|
||||||
|
using var memoryStream = new MemoryStream();
|
||||||
|
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||||||
|
memoryStream.Position = 0;
|
||||||
|
|
||||||
|
var stringBuilder = new StringBuilder();
|
||||||
|
|
||||||
|
// باز کردن سند Word به حالت فقط خواندنی
|
||||||
|
using (var wordDocument = WordprocessingDocument.Open(memoryStream, false))
|
||||||
|
{
|
||||||
|
var body = wordDocument.MainDocumentPart?.Document?.Body;
|
||||||
|
if (body != null)
|
||||||
|
{
|
||||||
|
// استخراج پاراگرافها و joining آنها با خط جدید برای حفظ ساختار برای LLM
|
||||||
|
var paragraphs = body.Elements<Paragraph>();
|
||||||
|
foreach (var paragraph in paragraphs)
|
||||||
|
{
|
||||||
|
var text = paragraph.InnerText?.Trim();
|
||||||
|
if (!string.IsNullOrEmpty(text))
|
||||||
|
{
|
||||||
|
stringBuilder.AppendLine(text);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return stringBuilder.ToString();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}</pre>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<!-- Section 3: Excel -->
|
||||||
|
<div class="section" id="step3">
|
||||||
|
<h2>📊 گام ۳: پیادهسازی ExcelContentExtractor</h2>
|
||||||
|
<p>این کلاس فایلهای <code>.xlsx</code> و <code>.xls</code> را خوانده و محتوای آن را به فرمت متنی ساختاریافته (شبیه CSV) تبدیل میکند تا مدل زبانی (LLM) بتواند به راحتی روابط سطرها و ستونها را درک کند.</p>
|
||||||
|
|
||||||
|
<div class="file-change">
|
||||||
|
<span class="badge-new">NEW</span>
|
||||||
|
<span class="path">xAiService/Providers/ExcelContentExtractor.cs</span>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<pre>using System;
|
||||||
|
using System.Data;
|
||||||
|
using System.IO;
|
||||||
|
using System.Text;
|
||||||
|
using System.Threading;
|
||||||
|
using System.Threading.Tasks;
|
||||||
|
using xAiService.Interfaces;
|
||||||
|
using ExcelDataReader;
|
||||||
|
|
||||||
|
namespace xAiService.Providers
|
||||||
|
{
|
||||||
|
/// <summary>
|
||||||
|
/// Extracts tabular content from Excel files (XLSX, XLS) using ExcelDataReader ...
|
||||||
|
/// </summary>
|
||||||
|
public class ExcelContentExtractor : IFileContentExtractor
|
||||||
|
{
|
||||||
|
private static readonly string[] SupportedMimeTypes = new[]
|
||||||
|
{
|
||||||
|
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", // .xlsx
|
||||||
|
"application/vnd.ms-excel" // .xls
|
||||||
|
};
|
||||||
|
|
||||||
|
public bool CanExtract(string mimeType)
|
||||||
|
{
|
||||||
|
return Array.Exists(SupportedMimeTypes, type =>
|
||||||
|
string.Equals(type, mimeType, StringComparison.OrdinalIgnoreCase)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
public async Task<string> ExtractAsync(
|
||||||
|
Stream fileStream,
|
||||||
|
string mimeType,
|
||||||
|
CancellationToken cancellationToken = default
|
||||||
|
)
|
||||||
|
{
|
||||||
|
// ثبت Encoding Provider برای پشتیبانی از فرمتهای قدیمیتر Excel
|
||||||
|
System.Text.Encoding.RegisterProvider(System.Text.CodePagesEncodingProvider.Instance);
|
||||||
|
|
||||||
|
// کپی در MemoryStream برای اطمینان از Seekable بودن
|
||||||
|
using var memoryStream = new MemoryStream();
|
||||||
|
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||||||
|
memoryStream.Position = 0;
|
||||||
|
|
||||||
|
var stringBuilder = new StringBuilder();
|
||||||
|
|
||||||
|
// ایجاد Reader به صورت خودکار بر اساس فرمت فایل
|
||||||
|
using (var reader = ExcelReaderFactory.CreateReader(memoryStream))
|
||||||
|
{
|
||||||
|
var result = reader.AsDataSet();
|
||||||
|
|
||||||
|
foreach (DataTable table in result.Tables)
|
||||||
|
{
|
||||||
|
stringBuilder.AppendLine($"--- Sheet: {table.TableName} ---");
|
||||||
|
|
||||||
|
// افزودن هدر ستونها
|
||||||
|
var headers = new string[table.Columns.Count];
|
||||||
|
for (int i = 0; i < table.Columns.Count; i++)
|
||||||
|
{
|
||||||
|
headers[i] = table.Columns[i].ColumnName;
|
||||||
|
}
|
||||||
|
stringBuilder.AppendLine(string.Join(" | ", headers));
|
||||||
|
stringBuilder.AppendLine(new string('-', 50));
|
||||||
|
|
||||||
|
// افزودن دادههای سطرها
|
||||||
|
foreach (DataRow row in table.Rows)
|
||||||
|
{
|
||||||
|
var rowValues = new string[row.ItemArray.Length];
|
||||||
|
for (int i = 0; i < row.ItemArray.Length; i++)
|
||||||
|
{
|
||||||
|
rowValues[i] = row.ItemArray[i]?.ToString()?.Trim() ?? string.Empty;
|
||||||
|
}
|
||||||
|
stringBuilder.AppendLine(string.Join(" | ", rowValues));
|
||||||
|
}
|
||||||
|
stringBuilder.AppendLine();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return stringBuilder.ToString();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}</pre>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<!-- Section 4: DI -->
|
||||||
|
<div class="section" id="step4">
|
||||||
|
<h2>🔗 گام ۴: ثبت سرویسها در Dependency Injection</h2>
|
||||||
|
<p>اکنون باید Extractor های جدید را به کانتینر DI اضافه کنیم. <code>CompositeFileContentExtractor</code> که قبلاً طراحی شده بود، به صورت خودکار تمام پیادهسازیهای <code>IFileContentExtractor</code> را از طریق تزریق <code>IEnumerable<IFileContentExtractor></code> دریافت کرده و مدیریت میکند.</p>
|
||||||
|
|
||||||
|
<div class="file-change">
|
||||||
|
<span class="badge-new">MODIFY</span>
|
||||||
|
<span class="path">xAiApi/Startup.cs (متد ConfigureServices)</span>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<pre>public void ConfigureServices(IServiceCollection services)
|
||||||
|
{
|
||||||
|
// ... (سایر ثبتهای سرویس)
|
||||||
|
|
||||||
|
// ✅ ثبت Extractor های فایل (الگوی Composite)
|
||||||
|
services.AddSingleton<IFileContentExtractor, PlainTextContentExtractor>();
|
||||||
|
services.AddSingleton<IFileContentExtractor, PdfContentExtractor>();
|
||||||
|
|
||||||
|
// ✅ افزودن Extractor های جدید آفیس
|
||||||
|
services.AddSingleton<IFileContentExtractor, DocxContentExtractor>();
|
||||||
|
services.AddSingleton<IFileContentExtractor, ExcelContentExtractor>();
|
||||||
|
|
||||||
|
// ✅ ثبت Composite Extractor (این کلاس لیست بالا را در Constructor دریافت میکند)
|
||||||
|
services.AddSingleton<IFileContentExtractor, CompositeFileContentExtractor>();
|
||||||
|
|
||||||
|
// ... (سایر ثبتهای سرویس)
|
||||||
|
}</pre>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<!-- Section 5: Considerations -->
|
||||||
|
<div class="section" id="step5">
|
||||||
|
<h2>⚠️ گام ۵: نکات کلیدی و ملاحظات عملکردی</h2>
|
||||||
|
|
||||||
|
<div class="arch-grid">
|
||||||
|
<div class="arch-card">
|
||||||
|
<h4>🧠 محدودیت Context Window مدل</h4>
|
||||||
|
<p>فایلهای Excel بزرگ میتوانند هزاران خط متن تولید کنند. قبل از ارسال به LLM، حتماً طول رشته استخراج شده را بررسی کنید و در صورت نیاز، آن را خلاصه (Chunk) کنید یا فقط چند سطر اول را ارسال نمایید.</p>
|
||||||
|
</div>
|
||||||
|
<div class="arch-card">
|
||||||
|
<h4>💾 مدیریت حافظه (Memory)</h4>
|
||||||
|
<p>استفاده از <code>MemoryStream</code> برای فایلهای بسیار بزرگ (مثلاً بالای 50 مگابایت) ممکن است باعث <code>OutOfMemoryException</code> شود. در <code>appsettings.json</code> حداکثر حجم فایل آپلودی را محدود کنید.</p>
|
||||||
|
</div>
|
||||||
|
<div class="arch-card">
|
||||||
|
<h4>🔒 امنیت و اعتبارسنجی</h4>
|
||||||
|
<p>کتابخانه <code>ExcelDataReader</code> در برابر فایلهای مخرب مقاوم است، اما همیشه قبل از پردازش، نوع فایل (MIME Type) و پسوند آن را در لایه Controller اعتبارسنجی کنید.</p>
|
||||||
|
</div>
|
||||||
|
<div class="arch-card">
|
||||||
|
<h4>🌐 پشتیبانی از Encoding</h4>
|
||||||
|
<p>خط <code>Encoding.RegisterProvider(CodePagesEncodingProvider.Instance)</code> در Extractor اکسل حیاتی است. بدون آن، خواندن فایلهای <code>.xls</code> قدیمی با کاراکترهای فارسی با خطا مواجه میشود.</p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<div class="alert alert-success">
|
||||||
|
<strong>✅ نتیجهگیری:</strong><br>
|
||||||
|
با افزودن این دو کلاس، معماری <code>FileContentExtractor</code> شما اکنون به طور کامل از فرمتهای متنی، PDF، Word و Excel پشتیبانی میکند. این تغییرات کاملاً با اصل Open/Closed (باز برای توسعه، بسته برای تغییر) در SOLID همخوانی دارد، زیرا برای افزودن فرمت جدید (مثلاً PowerPoint در آینده)، فقط کافی است یک کلاس جدید اضافه و در DI ثبت کنید، بدون اینکه کدهای موجود را تغییر دهید.
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<div class="footer">
|
||||||
|
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||||
|
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||||
|
<p><strong>📅 تاریخ تهیه مستند:</strong> چهارشنبه ۸ مهر ۱۴۰۵</p>
|
||||||
|
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||||
|
📄 مستند فنی افزودن پشتیبانی Office Files به xAiApi - تمامی حقوق محفوظ است
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
</div>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+1
-1
Submodule Modules/xAiModels updated: 755eafa7d2...26df2c7ebc
+1
-1
Submodule xAiApi updated: 4521a53734...c30d93be0b
Reference in New Issue
Block a user