Last ...
This commit is contained in:
@@ -24,4 +24,17 @@ sc delete XLLMService
|
||||
sc delete XLLMEmbdService
|
||||
|
||||
https://docs.mozilla.ai/llamafile/using-llamafile/api
|
||||
https://github.com/mozilla-ai/llamafile
|
||||
https://github.com/mozilla-ai/llamafile
|
||||
|
||||
-------------------
|
||||
|
||||
Uncesored Qwen 3.5 Coder GGUF:
|
||||
|
||||
https://huggingface.co/DavidAU/Qwen3.5-9B-The-Defiant-Fable-Uncensored-Heretic-NEO-IMATRIX-MAX-MTP-GGUF
|
||||
|
||||
-------------------
|
||||
|
||||
Download these files and put them inside xAiApi/Models:
|
||||
|
||||
https://huggingface.co/ggerganov/whisper.cpp/blob/main/ggml-base.bin
|
||||
https://huggingface.co/ggerganov/whisper.cpp/blob/main/ggml-small.bin
|
||||
@@ -514,6 +514,33 @@
|
||||
"id": "171f43f3-ae73-49b7-9239-9f3fb695cc2b",
|
||||
"title": "InProgress",
|
||||
"cards": [
|
||||
{
|
||||
"id": "9757ac0e-2aff-44a1-a0db-7ee4468def1b",
|
||||
"listId": "171f43f3-ae73-49b7-9239-9f3fb695cc2b",
|
||||
"title": "Detect a Very Good GGUF Model for Performing OCR on Bynary Contents ...",
|
||||
"description": "",
|
||||
"labels": [],
|
||||
"checkboxes": [],
|
||||
"comments": []
|
||||
},
|
||||
{
|
||||
"id": "5cccaa60-c985-4a99-93eb-adb38a1f79db",
|
||||
"listId": "171f43f3-ae73-49b7-9239-9f3fb695cc2b",
|
||||
"title": "Complete OCR Preparation ...",
|
||||
"description": "",
|
||||
"labels": [],
|
||||
"checkboxes": [],
|
||||
"comments": []
|
||||
},
|
||||
{
|
||||
"id": "38a94972-d86a-49a3-8917-9b2d958df660",
|
||||
"listId": "171f43f3-ae73-49b7-9239-9f3fb695cc2b",
|
||||
"title": "Complete Audio Transcript ...",
|
||||
"description": "",
|
||||
"labels": [],
|
||||
"checkboxes": [],
|
||||
"comments": []
|
||||
},
|
||||
{
|
||||
"id": "5d103817-846f-477f-a381-50e86c3cc7d4",
|
||||
"listId": "171f43f3-ae73-49b7-9239-9f3fb695cc2b",
|
||||
@@ -569,7 +596,7 @@
|
||||
{
|
||||
"id": "07602ca6-38b8-4d19-b0f6-8730e11de1df",
|
||||
"title": "Add Support for Files Attachment in AI Asking ",
|
||||
"checked": false
|
||||
"checked": true
|
||||
},
|
||||
{
|
||||
"id": "149c4343-16ab-4ac8-899b-23d202000b85",
|
||||
|
||||
@@ -0,0 +1,441 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fa" dir="rtl">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>رفع خطای وابستگی چرخشی (Circular Dependency) در سرویس OCR - فن آوران ساحر علم</title>
|
||||
<style>
|
||||
:root {
|
||||
--primary: #1e3a8a;
|
||||
--secondary: #3b82f6;
|
||||
--accent: #f59e0b;
|
||||
--success: #10b981;
|
||||
--danger: #ef4444;
|
||||
--warning: #f97316;
|
||||
--bg-light: #f8fafc;
|
||||
--bg-code: #1e293b;
|
||||
--text-dark: #0f172a;
|
||||
--text-muted: #64748b;
|
||||
--border: #e2e8f0;
|
||||
}
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||
color: var(--text-dark);
|
||||
line-height: 1.8;
|
||||
padding: 20px;
|
||||
}
|
||||
.container {
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
background: white;
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||
overflow: hidden;
|
||||
}
|
||||
.header {
|
||||
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
|
||||
color: white;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
}
|
||||
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
|
||||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||||
.meta-bar {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
background: var(--bg-light);
|
||||
padding: 15px 30px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
flex-wrap: wrap;
|
||||
gap: 15px;
|
||||
}
|
||||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||
.meta-item strong { color: var(--primary); }
|
||||
.content { padding: 40px; }
|
||||
.section {
|
||||
margin-bottom: 35px;
|
||||
padding: 25px;
|
||||
background: var(--bg-light);
|
||||
border-radius: 12px;
|
||||
border-right: 5px solid var(--secondary);
|
||||
}
|
||||
.section h2 {
|
||||
color: var(--primary);
|
||||
font-size: 1.5em;
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
}
|
||||
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
|
||||
pre {
|
||||
background: var(--bg-code);
|
||||
color: #e2e8f0;
|
||||
padding: 18px;
|
||||
border-radius: 8px;
|
||||
overflow-x: auto;
|
||||
direction: ltr;
|
||||
text-align: left;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.85em;
|
||||
margin: 15px 0;
|
||||
border-right: 4px solid var(--accent);
|
||||
}
|
||||
code {
|
||||
background: #fef3c7;
|
||||
color: #92400e;
|
||||
padding: 2px 8px;
|
||||
border-radius: 4px;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.9em;
|
||||
direction: ltr;
|
||||
display: inline-block;
|
||||
}
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
margin: 15px 0;
|
||||
background: white;
|
||||
border-radius: 8px;
|
||||
overflow: hidden;
|
||||
}
|
||||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
|
||||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||||
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
|
||||
.toc h3 { color: var(--primary); margin-bottom: 15px; }
|
||||
.toc ol { padding-right: 25px; }
|
||||
.toc li { padding: 6px 0; }
|
||||
.toc a { color: var(--secondary); text-decoration: none; }
|
||||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
|
||||
<div class="header">
|
||||
<h1>🔗 رفع خطای وابستگی چرخشی (Circular Dependency)</h1>
|
||||
<div class="subtitle">تحلیل و رفع چرخه وابستگی بین IXFileContentExtractor و IXDefaultAIOCRService</div>
|
||||
</div>
|
||||
|
||||
<div class="meta-bar">
|
||||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
|
||||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||||
</div>
|
||||
|
||||
<div class="content">
|
||||
|
||||
<div class="toc">
|
||||
<h3>📑 فهرست مطالب</h3>
|
||||
<ol>
|
||||
<li><a href="#analysis">تحلیل ریشه خطا</a></li>
|
||||
<li><a href="#solution">استراتژی رفع خطا</a></li>
|
||||
<li><a href="#step1">گام ۱: اصلاح Interface سرویس OCR</a></li>
|
||||
<li><a href="#step2">گام ۲: بازنویسی مستقل کلاس XDefaultAIOCRService</a></li>
|
||||
<li><a href="#step3">گام ۳: بررسی ثبت در Dependency Injection</a></li>
|
||||
</ol>
|
||||
</div>
|
||||
|
||||
<!-- Section 1: Analysis -->
|
||||
<div class="section" id="analysis">
|
||||
<h2>🔍 گام ۱: تحلیل ریشه خطا</h2>
|
||||
<p>پیغام خطای DI به وضوح یک <strong>وابستگی چرخشی (Circular Dependency)</strong> را نشان میدهد:</p>
|
||||
<div class="alert alert-danger">
|
||||
<code>IXFileContentExtractor</code> ➔ <code>XVisionFileContentExtractor</code> ➔ <code>IXDefaultAIOCRService</code> ➔ <code>XDefaultAIOCRService</code> ➔ <code>XAIServiceBase</code> ➔ <code>IXFileContentExtractor</code>
|
||||
</div>
|
||||
<p><strong>چرا این اتفاق افتاد؟</strong></p>
|
||||
<ul style="padding-right: 25px; margin-top: 15px;">
|
||||
<li>کلاس <code>XVisionFileContentExtractor</code> برای انجام OCR به <code>IXDefaultAIOCRService</code> نیاز دارد.</li>
|
||||
<li>کلاس <code>XDefaultAIOCRService</code> از <code>XAIServiceBase</code> ارثبری کرده است.</li>
|
||||
<li>سازنده (Constructor) کلاس <code>XAIServiceBase</code> به <code>IXFileContentExtractor</code> نیاز دارد.</li>
|
||||
</ul>
|
||||
<p>این زنجیره باعث میشود کانتینر DI در یک حلقه بینهایت گیر کند و نتواند هیچیک از این سرویسها را مقداردهی اولیه نماید.</p>
|
||||
</div>
|
||||
|
||||
<!-- Section 2: Solution -->
|
||||
<div class="section" id="solution">
|
||||
<h2>💡 گام ۲: استراتژی رفع خطا</h2>
|
||||
<p>برای شکستن این چرخه، باید <strong>وابستگی <code>XDefaultAIOCRService</code> به <code>XAIServiceBase</code> را حذف کنیم</strong>. </p>
|
||||
<p>سرویس OCR فقط نیاز به برقراری ارتباط با مدل زبانی (LLM) برای استخراج متن از تصویر دارد و به هیچوجه به <code>IXFileContentExtractor</code>، <code>IXFileProvider</code> یا <code>IXAiDataProvider</code> نیاز ندارد. بنابراین، با حذف ارثبری از <code>XAIServiceBase</code> و پیادهسازی مستقیم منطق ساخت <code>IChatClient</code> درون همین کلاس، چرخه وابستگی کاملاً شکسته میشود.</p>
|
||||
</div>
|
||||
|
||||
<!-- Section 3: Step 1 -->
|
||||
<div class="section" id="step1">
|
||||
<h2>🛠️ گام ۳: اصلاح Interface سرویس OCR</h2>
|
||||
<p>ابتدا باید ارثبری غیرضروری <code>IXAiServiceBase</code> را از اینترفیس حذف کنیم، زیرا این سرویس فقط وظیفه OCR را بر عهده دارد و نیازی به متدهای مدیریت پروژه، مکالمه و پیام ندارد.</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Interfaces/IXDefaultAIOCRService.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using Microsoft.Extensions.AI;
|
||||
using System.Collections.Generic;
|
||||
|
||||
namespace xAiApi.Interfaces
|
||||
{
|
||||
/// <summary>
|
||||
/// سرویس اختصاصی برای انجام OCR بر روی محتوای تصویری ...
|
||||
/// </summary>
|
||||
public interface IXDefaultAIOCRService
|
||||
{
|
||||
/// <summary>
|
||||
/// درخواست انجام OCR بر روی محتوای داده شده ...
|
||||
/// </summary>
|
||||
Task<string> RequestOCRAsync(
|
||||
IList<ChatMessage> messages,
|
||||
CancellationToken cancellationToken = default
|
||||
);
|
||||
|
||||
/// <summary>
|
||||
/// درخواست انجام OCR بر روی محتوای داده شده به صورت Stream ...
|
||||
/// </summary>
|
||||
IAsyncEnumerable<string> RequestOCRAsEnumerable(
|
||||
IList<ChatMessage> messages,
|
||||
CancellationToken cancellationToken = default
|
||||
);
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 4: Step 2 -->
|
||||
<div class="section" id="step2">
|
||||
<h2>⚙️ گام ۴: بازنویسی مستقل کلاس XDefaultAIOCRService</h2>
|
||||
<p>اکنون کلاس پیادهسازی را بازنویسی میکنیم تا دیگر از <code>XAIServiceBase</code> ارثبری نکند. منطق ساخت <code>IChatClient</code> (که قبلاً در کلاس پایه بود) به صورت مستقیم و تمیز درون این کلاس قرار میگیرد.</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Providers/XDefaultAIOCRService.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>using System;
|
||||
using System.Linq;
|
||||
using System.Net.Http;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using System.Collections.Generic;
|
||||
using System.Runtime.CompilerServices;
|
||||
using Microsoft.Extensions.AI;
|
||||
using Microsoft.Extensions.Logging;
|
||||
using OpenAI;
|
||||
using OllamaSharp;
|
||||
using System.ClientModel;
|
||||
using System.ClientModel.Primitives;
|
||||
using xAiApi.Configurations;
|
||||
using xAiApi.Constants;
|
||||
using xAiApi.Extensions;
|
||||
using xAiModels.Constants;
|
||||
using xAiModels.Extensions;
|
||||
using xCommons.Extensions;
|
||||
using xExceptions.Constants;
|
||||
|
||||
namespace xAiApi.Providers
|
||||
{
|
||||
/// <summary>
|
||||
/// پیادهسازی مستقل سرویس OCR بدون وابستگی به XAIServiceBase ...
|
||||
/// </summary>
|
||||
public class XDefaultAIOCRService : IXDefaultAIOCRService
|
||||
{
|
||||
private readonly string prompt;
|
||||
private readonly XAiModelDescriptor descriptor;
|
||||
private readonly ILogger<XDefaultAIOCRService> logger;
|
||||
|
||||
public XDefaultAIOCRService(
|
||||
ILogger<XDefaultAIOCRService> logger,
|
||||
XAiApiConfiguration configuration,
|
||||
string model = null,
|
||||
string prompt = null
|
||||
)
|
||||
{
|
||||
this.logger = logger;
|
||||
|
||||
// ۱. دریافت پیکربندی مدل
|
||||
if (model.IsNullOrEmpty())
|
||||
{
|
||||
model = XAiApiConstants.XAiDefaultVisionModelName;
|
||||
}
|
||||
|
||||
descriptor = configuration.GetModel(model);
|
||||
if (descriptor.IsNullOrDefault())
|
||||
{
|
||||
XException.InvalidData.Throw("Invalid OCR Model Configuration");
|
||||
}
|
||||
|
||||
// ۲. دریافت پرامپت استخراج
|
||||
if (prompt.IsNullOrEmpty())
|
||||
{
|
||||
prompt = configuration.GetPrompt(
|
||||
name: XAiApiConstants.XAiApiContentExtractionPromptName,
|
||||
@params: null
|
||||
);
|
||||
}
|
||||
|
||||
this.prompt = prompt;
|
||||
if (this.prompt.IsNullOrEmpty())
|
||||
{
|
||||
XException.InvalidArgs.Throw("OCR Prompt cannot be empty");
|
||||
}
|
||||
}
|
||||
|
||||
public virtual async Task<string> RequestOCRAsync(
|
||||
IList<ChatMessage> messages,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
if (!messages.HasChild())
|
||||
{
|
||||
XException.InvalidArgs.Throw();
|
||||
}
|
||||
|
||||
using var client = GetClient();
|
||||
|
||||
var pMessage = new ChatMessage(ChatRole.System, prompt);
|
||||
messages = [pMessage, .. messages];
|
||||
|
||||
var response = await client.GetResponseAsync(
|
||||
messages: messages,
|
||||
cancellationToken: cancellationToken
|
||||
);
|
||||
|
||||
if (!response.IsValid())
|
||||
{
|
||||
XException.ActionFailed.Throw();
|
||||
}
|
||||
|
||||
return response.Text;
|
||||
}
|
||||
|
||||
public virtual async IAsyncEnumerable<string> RequestOCRAsEnumerable(
|
||||
IList<ChatMessage> messages,
|
||||
[EnumeratorCancellation] CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
if (!messages.HasChild())
|
||||
{
|
||||
XException.InvalidArgs.Throw();
|
||||
}
|
||||
|
||||
using var client = GetClient();
|
||||
|
||||
var pMessage = new ChatMessage(ChatRole.System, prompt);
|
||||
messages = [pMessage, .. messages];
|
||||
|
||||
var enumerable = client.GetStreamingResponseAsync(
|
||||
options: null,
|
||||
messages: messages
|
||||
);
|
||||
|
||||
await foreach (var res in enumerable)
|
||||
{
|
||||
if (cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
yield break;
|
||||
}
|
||||
yield return res.Text;
|
||||
}
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// ساخت کلاینت ارتباط با LLM (جایگزین متد GetClient در XAIServiceBase) ...
|
||||
/// </summary>
|
||||
private IChatClient GetClient()
|
||||
{
|
||||
var model = descriptor.LLM;
|
||||
var apiKey = descriptor.ApiKey;
|
||||
var url = new Uri(descriptor.Url);
|
||||
var httpClient = new HttpClient { BaseAddress = url, Timeout = Timeout.InfiniteTimeSpan };
|
||||
|
||||
IChatClient result = null;
|
||||
switch (descriptor.Provider)
|
||||
{
|
||||
case XAiModelProviderType.Ollama:
|
||||
var ollamaClient = new OllamaApiClient(httpClient, model);
|
||||
result = new ChatClientBuilder(ollamaClient).UseFunctionInvocation().Build();
|
||||
break;
|
||||
case XAiModelProviderType.OpenAI:
|
||||
var openAiClient = new OpenAIClient(
|
||||
new ApiKeyCredential(apiKey.IsNullOrEmpty() ? XAiApiConstants.XOpenAINoKey : apiKey),
|
||||
new OpenAIClientOptions
|
||||
{
|
||||
Endpoint = url,
|
||||
Transport = new HttpClientPipelineTransport(httpClient)
|
||||
}
|
||||
);
|
||||
result = new ChatClientBuilder(openAiClient.GetChatClient(model).AsIChatClient()).UseFunctionInvocation().Build();
|
||||
break;
|
||||
default:
|
||||
XException.InvalidData.Throw($"Unsupported provider: {descriptor.Provider}");
|
||||
break;
|
||||
}
|
||||
|
||||
if (result == null)
|
||||
{
|
||||
XException.InvalidData.Throw("Failed to create Chat Client");
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 5: Step 3 -->
|
||||
<div class="section" id="step3">
|
||||
<h2>🔌 گام ۵: بررسی ثبت در Dependency Injection</h2>
|
||||
<p>با توجه به تغییرات فوق، ثبت سرویس در فایل <code>Startup.cs</code> بدون هیچ تغییری به درستی کار خواهد کرد، زیرا امضای Constructor اکنون سادهتر شده و فقط به <code>ILogger</code> و <code>XAiApiConfiguration</code> وابسته است که هر دو از قبل در DI ثبت شدهاند.</p>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ تأییدیه:</strong><br>
|
||||
خط <code>services.AddScoped<IXDefaultAIOCRService, XDefaultAIOCRService>();</code> در <code>Startup.cs</code> کاملاً معتبر است و دیگر باعث ایجاد چرخه وابستگی نمیشود.
|
||||
</div>
|
||||
|
||||
<h3>خلاصه دستاوردهای این اصلاح:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>مزیت</th>
|
||||
<th>توضیح</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>🚫 حذف Circular Dependency</td>
|
||||
<td>چرخه معیوب بین Extractor و سرویس OCR کاملاً شکسته شد.</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>⚡ افزایش عملکرد (Performance)</td>
|
||||
<td>سرویس OCR دیگر بار اضافی مقداردهی اولیه DataProvider و FileProvider را تحمل نمیکند.</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>🎯 اصل تکوظیفهای (SRP)</td>
|
||||
<td>کلاس <code>XDefaultAIOCRService</code> اکنون فقط و فقط مسئول ارتباط با مدل برای OCR است.</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>🧪 قابلیت تستپذیری (Testability)</td>
|
||||
<td>تزریق وابستگیهای کمتر، نوشتن Unit Test برای این سرویس را بسیار سادهتر میکند.</td>
|
||||
</tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div class="footer">
|
||||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
|
||||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||
🔗 مستند فنی رفع خطای وابستگی چرخشی در xAiApi - تمامی حقوق محفوظ است
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,616 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fa" dir="rtl">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>رفع خطای Circular Dependency در XDefaultAIOCRService - فن آوران ساحر علم</title>
|
||||
<style>
|
||||
:root {
|
||||
--primary: #1e3a8a;
|
||||
--secondary: #3b82f6;
|
||||
--accent: #f59e0b;
|
||||
--success: #10b981;
|
||||
--danger: #ef4444;
|
||||
--warning: #f97316;
|
||||
--bg-light: #f8fafc;
|
||||
--bg-code: #1e293b;
|
||||
--text-dark: #0f172a;
|
||||
--text-muted: #64748b;
|
||||
--border: #e2e8f0;
|
||||
}
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||
color: var(--text-dark);
|
||||
line-height: 1.8;
|
||||
padding: 20px;
|
||||
}
|
||||
.container {
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
background: white;
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||
overflow: hidden;
|
||||
}
|
||||
.header {
|
||||
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
|
||||
color: white;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
}
|
||||
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
|
||||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||||
.meta-bar {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
background: var(--bg-light);
|
||||
padding: 15px 30px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
flex-wrap: wrap;
|
||||
gap: 15px;
|
||||
}
|
||||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||
.meta-item strong { color: var(--primary); }
|
||||
.content { padding: 40px; }
|
||||
.section {
|
||||
margin-bottom: 35px;
|
||||
padding: 25px;
|
||||
background: var(--bg-light);
|
||||
border-radius: 12px;
|
||||
border-right: 5px solid var(--secondary);
|
||||
}
|
||||
.section h2 {
|
||||
color: var(--primary);
|
||||
font-size: 1.5em;
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
}
|
||||
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
|
||||
pre {
|
||||
background: var(--bg-code);
|
||||
color: #e2e8f0;
|
||||
padding: 18px;
|
||||
border-radius: 8px;
|
||||
overflow-x: auto;
|
||||
direction: ltr;
|
||||
text-align: left;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.85em;
|
||||
margin: 15px 0;
|
||||
border-right: 4px solid var(--accent);
|
||||
}
|
||||
code {
|
||||
background: #fef3c7;
|
||||
color: #92400e;
|
||||
padding: 2px 8px;
|
||||
border-radius: 4px;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.9em;
|
||||
direction: ltr;
|
||||
display: inline-block;
|
||||
}
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
margin: 15px 0;
|
||||
background: white;
|
||||
border-radius: 8px;
|
||||
overflow: hidden;
|
||||
}
|
||||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||
tr:hover { background: var(--bg-light); }
|
||||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
|
||||
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
|
||||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||||
.footer p { margin: 5px 0; }
|
||||
.highlight { background: linear-gradient(120deg, #fef3c7 0%, #fef3c7 100%); padding: 2px 6px; border-radius: 4px; font-weight: bold; }
|
||||
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
|
||||
.toc h3 { color: var(--primary); margin-bottom: 15px; }
|
||||
.toc ol { padding-right: 25px; }
|
||||
.toc li { padding: 6px 0; }
|
||||
.toc a { color: var(--secondary); text-decoration: none; }
|
||||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.badge-remove { background: var(--danger); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.arch-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(280px, 1fr)); gap: 20px; margin: 20px 0; }
|
||||
.arch-card { background: white; padding: 20px; border-radius: 10px; box-shadow: 0 4px 12px rgba(0,0,0,0.08); border-top: 4px solid var(--secondary); }
|
||||
.arch-card h4 { color: var(--primary); margin-bottom: 12px; }
|
||||
.arch-card ul { list-style: none; padding-right: 0; }
|
||||
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; }
|
||||
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
|
||||
.diff-old { background: #fee2e2; color: #991b1b; padding: 2px 4px; border-radius: 3px; text-decoration: line-through; }
|
||||
.diff-new { background: #d1fae5; color: #065f46; padding: 2px 4px; border-radius: 3px; }
|
||||
.chain-diagram { background: white; padding: 25px; border-radius: 10px; margin: 20px 0; text-align: center; }
|
||||
.chain-step { display: inline-block; background: var(--danger); color: white; padding: 10px 18px; border-radius: 8px; margin: 5px; font-size: 0.88em; }
|
||||
.chain-step.fixed { background: var(--success); }
|
||||
.chain-arrow { display: inline-block; color: var(--accent); font-size: 1.5em; margin: 0 8px; vertical-align: middle; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
|
||||
<div class="header">
|
||||
<h1>🔗 رفع خطای Circular Dependency</h1>
|
||||
<div class="subtitle">تحلیل و رفع حلقه وابستگی بین XVisionFileContentExtractor و XDefaultAIOCRService</div>
|
||||
</div>
|
||||
|
||||
<div class="meta-bar">
|
||||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
|
||||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||||
</div>
|
||||
|
||||
<div class="content">
|
||||
|
||||
<div class="toc">
|
||||
<h3>📑 فهرست مطالب</h3>
|
||||
<ol>
|
||||
<li><a href="#analysis">تحلیل دقیق ریشه خطا</a></li>
|
||||
<li><a href="#root-cause">گام ۱: شناسایی علت اصلی</a></li>
|
||||
<li><a href="#solution">گام ۲: راهحل اصلاحی</a></li>
|
||||
<li><a href="#step1">گام ۳: اصلاح XDefaultAIOCRService</a></li>
|
||||
<li><a href="#step2">گام ۴: بررسی Startup.cs</a></li>
|
||||
<li><a href="#verification">گام ۵: تأیید رفع خطا</a></li>
|
||||
</ol>
|
||||
</div>
|
||||
|
||||
<!-- Section 1: Analysis -->
|
||||
<div class="section" id="analysis">
|
||||
<h2>🔍 گام ۱: تحلیل دقیق ریشه خطا</h2>
|
||||
<p>پیغام خطای DI به وضوح یک <strong>حلقه وابستگی (Circular Dependency)</strong> را نشان میدهد:</p>
|
||||
|
||||
<div class="chain-diagram">
|
||||
<span class="chain-step">IXFileContentExtractor</span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step">XVisionFileContentExtractor</span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step">IXDefaultAIOCRService</span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step">XDefaultAIOCRService</span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step">IXFileContentExtractor ❌</span>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-danger">
|
||||
<strong>⚠️ تحلیل زنجیره:</strong>
|
||||
<ol style="padding-right: 25px; margin-top: 10px;">
|
||||
<li><code>XVisionFileContentExtractor</code> در constructor خود به <code>IXDefaultAIOCRService</code> نیاز دارد.</li>
|
||||
<li><code>XDefaultAIOCRService</code> در constructor خود به <code>IXFileContentExtractor</code> نیاز دارد.</li>
|
||||
<li>اما <code>IXFileContentExtractor</code> همان کامپوزیتی است که شامل <code>XVisionFileContentExtractor</code> میباشد!</li>
|
||||
<li>این یک <strong>حلقه بینهایت</strong> ایجاد میکند و DI Container قادر به ساخت هیچیک از این سرویسها نیست.</li>
|
||||
</ol>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 2: Root Cause -->
|
||||
<div class="section" id="root-cause">
|
||||
<h2>🎯 گام ۲: شناسایی علت اصلی</h2>
|
||||
<p>با بررسی دقیق کد <code>XDefaultAIOCRService</code>، یک نکته کلیدی کشف شد:</p>
|
||||
|
||||
<div class="arch-grid">
|
||||
<div class="arch-card">
|
||||
<h4>❌ پارامتر غیرضروری در Constructor</h4>
|
||||
<p>در constructor کلاس <code>XDefaultAIOCRService</code>، پارامتر <code>IXFileContentExtractor fileContentExtractor</code> دریافت میشود، اما <strong>در هیچ جای کلاس از آن استفاده نمیشود!</strong></p>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>✅ سرویس OCR مستقل است</h4>
|
||||
<p>سرویس OCR فقط نیاز به ارتباط با مدل زبانی (LLM) دارد و به <code>IXFileContentExtractor</code> نیازی ندارد. این پارامتر به اشتباه از نسخه قبلی باقی مانده است.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<h3>کد مشکلدار فعلی:</h3>
|
||||
<pre>public XDefaultAIOCRService(
|
||||
XAiApiConfiguration configuration,
|
||||
ILogger<XDefaultAIOCRService> logger,
|
||||
<span class="diff-old">IXFileContentExtractor fileContentExtractor, // ❌ پارامتر غیرضروری</span>
|
||||
string model = null,
|
||||
string prompt = null
|
||||
)
|
||||
{
|
||||
// ... هیچ استفادهای از fileContentExtractor در body کلاس نیست
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 3: Solution -->
|
||||
<div class="section" id="solution">
|
||||
<h2>💡 گام ۳: راهحل اصلاحی</h2>
|
||||
<p>برای شکستن حلقه وابستگی، کافی است <strong>پارامتر غیرضروری <code>IXFileContentExtractor</code> را از constructor کلاس <code>XDefaultAIOCRService</code> حذف کنیم</strong>.</p>
|
||||
|
||||
<div class="chain-diagram">
|
||||
<span class="chain-step fixed">IXFileContentExtractor</span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step fixed">XVisionFileContentExtractor</span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step fixed">IXDefaultAIOCRService</span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step fixed">XDefaultAIOCRService ✅</span>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نتیجه:</strong> با حذف این پارامتر، زنجیره وابستگی کاملاً شکسته میشود و DI Container میتواند تمام سرویسها را بدون مشکل مقداردهی اولیه کند.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 4: Step 1 -->
|
||||
<div class="section" id="step1">
|
||||
<h2>🛠️ گام ۴: اصلاح XDefaultAIOCRService</h2>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Providers/XDefaultAIOCRService.cs</span>
|
||||
</div>
|
||||
|
||||
<h3>کد اصلاحشده:</h3>
|
||||
<pre>using System;
|
||||
using OpenAI;
|
||||
using OllamaSharp;
|
||||
using System.Net.Http;
|
||||
using System.Threading;
|
||||
using xAiModels.Models;
|
||||
using xAiApi.Constants;
|
||||
using xAiApi.Interfaces;
|
||||
using xAiApi.Extensions;
|
||||
using System.ClientModel;
|
||||
using xCommons.Extensions;
|
||||
using xAiModels.Extensions;
|
||||
using xAiApi.Configurations;
|
||||
using xExceptions.Constants;
|
||||
using System.Threading.Tasks;
|
||||
using Microsoft.Extensions.AI;
|
||||
using System.Collections.Generic;
|
||||
using Microsoft.Extensions.Logging;
|
||||
using System.ClientModel.Primitives;
|
||||
using System.Runtime.CompilerServices;
|
||||
|
||||
namespace xAiApi.Providers
|
||||
{
|
||||
public class XDefaultAIOCRService : IXDefaultAIOCRService
|
||||
{
|
||||
/// <summary>
|
||||
/// Chat Options ...
|
||||
/// </summary>
|
||||
public ChatOptions Options { get; }
|
||||
|
||||
/// <summary>
|
||||
/// Prompt ...
|
||||
/// </summary>
|
||||
public string Prompt { get; }
|
||||
|
||||
/// <summary>
|
||||
/// Descriptor of Models which used in Service ...
|
||||
/// </summary>
|
||||
public XAiModelDescriptor Descriptor { get; }
|
||||
|
||||
public XDefaultAIOCRService(
|
||||
XAiApiConfiguration configuration,
|
||||
ILogger<XDefaultAIOCRService> logger,
|
||||
<span class="diff-new">// ✅ پارامتر IXFileContentExtractor حذف شد</span>
|
||||
string model = null,
|
||||
string prompt = null
|
||||
)
|
||||
{
|
||||
//
|
||||
// Prepare Model Descriptor ...
|
||||
Descriptor = configuration.GetModel(model);
|
||||
if (!Descriptor.IsValid())
|
||||
{
|
||||
XException.InvalidConfiguration.Throw();
|
||||
}
|
||||
//
|
||||
Prompt = prompt;
|
||||
if (Prompt.IsNullOrEmpty())
|
||||
{
|
||||
XException.InvalidArgs.Throw();
|
||||
}
|
||||
//
|
||||
// Prepare Chat Options based On Tools ...
|
||||
Options = new ChatOptions();
|
||||
}
|
||||
|
||||
//
|
||||
#region OCR ...
|
||||
/// <summary>
|
||||
/// Request for Doing OCR on Given Data Contents ...
|
||||
/// </summary>
|
||||
public virtual async Task<string> RequestOCRAsync(
|
||||
IList<ChatMessage> messages,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
// Validate ...
|
||||
if (!messages.HasChild())
|
||||
{
|
||||
XException.InvalidArgs.Throw();
|
||||
}
|
||||
//
|
||||
using var client = GetClient();
|
||||
//
|
||||
// Preparing Extraction Prompt Message ...
|
||||
var pMessage = new ChatMessage(
|
||||
ChatRole.System,
|
||||
Prompt
|
||||
);
|
||||
//
|
||||
messages = [pMessage, .. messages];
|
||||
//
|
||||
var response = await client.GetResponseAsync(
|
||||
messages: messages,
|
||||
cancellationToken: cancellationToken
|
||||
);
|
||||
//
|
||||
// Validate Response ...
|
||||
if (!response.IsValid())
|
||||
{
|
||||
//
|
||||
// Dispose Client ...
|
||||
client.Dispose();
|
||||
XException.ActionFailed.Throw();
|
||||
}
|
||||
//
|
||||
// Retrieve Response Text ...
|
||||
var result = response.Text;
|
||||
//
|
||||
return result;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Request for Doing OCR on Given Data Contents as Stream ...
|
||||
/// </summary>
|
||||
public virtual async IAsyncEnumerable<string> RequestOCRAsEnumerable(
|
||||
IList<ChatMessage> messages,
|
||||
[EnumeratorCancellation]
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
//
|
||||
// Validate ...
|
||||
if (!messages.HasChild())
|
||||
{
|
||||
XException.InvalidArgs.Throw();
|
||||
}
|
||||
//
|
||||
using var client = GetClient();
|
||||
//
|
||||
// Preparing Extraction Prompt Message ...
|
||||
var pMessage = new ChatMessage(
|
||||
ChatRole.System,
|
||||
Prompt
|
||||
);
|
||||
//
|
||||
messages = [pMessage, .. messages];
|
||||
//
|
||||
var enumerable = client.GetStreamingResponseAsync(
|
||||
options: null,
|
||||
messages: messages
|
||||
);
|
||||
//
|
||||
await foreach (var res in enumerable)
|
||||
{
|
||||
//
|
||||
// Cancellation Token ...
|
||||
if (cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
yield break;
|
||||
}
|
||||
//
|
||||
yield return res.Text;
|
||||
}
|
||||
}
|
||||
#endregion
|
||||
|
||||
//
|
||||
#region Preaprations ...
|
||||
/// <summary>
|
||||
/// Get LLM Client instance for Communicating with LLM ...
|
||||
/// </summary>
|
||||
public virtual IChatClient GetClient()
|
||||
{
|
||||
//
|
||||
// Try to Initialize LLm ...
|
||||
var model = Descriptor.LLM;
|
||||
var apiKey = Descriptor.ApiKey;
|
||||
var url = new Uri(Descriptor.Url);
|
||||
var httpClient = GetHttpClient(Descriptor.Url);
|
||||
//
|
||||
IChatClient result = null;
|
||||
switch (Descriptor.Provider)
|
||||
{
|
||||
//
|
||||
case XAiModelProviderType.Ollama:
|
||||
//
|
||||
var ollamaClient = new OllamaApiClient(httpClient, model);
|
||||
result =
|
||||
new ChatClientBuilder(ollamaClient)
|
||||
.UseFunctionInvocation()
|
||||
.Build();
|
||||
break;
|
||||
//
|
||||
case XAiModelProviderType.OpenAI:
|
||||
//
|
||||
var openAiClient = new OpenAIClient(
|
||||
new ApiKeyCredential(apiKey.IsNullOrEmpty() ? XAiApiConstants.XOpenAINoKey : apiKey),
|
||||
new OpenAIClientOptions
|
||||
{
|
||||
Endpoint = url,
|
||||
Transport = new HttpClientPipelineTransport(httpClient)
|
||||
}
|
||||
);
|
||||
result =
|
||||
new ChatClientBuilder(
|
||||
openAiClient
|
||||
.GetChatClient(model)
|
||||
.AsIChatClient())
|
||||
.UseFunctionInvocation()
|
||||
.Build();
|
||||
break;
|
||||
//
|
||||
case XAiModelProviderType.DeepSeek:
|
||||
break;
|
||||
//
|
||||
case XAiModelProviderType.HuggingFace:
|
||||
break;
|
||||
//
|
||||
default:
|
||||
break;
|
||||
}
|
||||
//
|
||||
if (result.IsNullOrDefault())
|
||||
{
|
||||
XException.InvalidData.Throw();
|
||||
}
|
||||
//
|
||||
return result;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Create Custom HttpClient for Communicating with LLM API ...
|
||||
/// </summary>
|
||||
public virtual HttpClient GetHttpClient(string url = null)
|
||||
{
|
||||
//
|
||||
var result = new HttpClient
|
||||
{
|
||||
//
|
||||
// Disable timeout completely (not recommended for production)
|
||||
Timeout = Timeout.InfiniteTimeSpan,
|
||||
//
|
||||
BaseAddress = url.IsNullOrEmpty()
|
||||
? null
|
||||
: new Uri(url),
|
||||
};
|
||||
//
|
||||
return result;
|
||||
}
|
||||
#endregion
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 5: Step 2 -->
|
||||
<div class="section" id="step2">
|
||||
<h2>🔌 گام ۵: بررسی Startup.cs</h2>
|
||||
<p>با توجه به تغییرات فوق، ثبت سرویسها در <code>Startup.cs</code> بدون هیچ تغییری به درستی کار خواهد کرد. اما برای اطمینان، ترتیب ثبت سرویسها را بررسی میکنیم:</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">CHECK</span>
|
||||
<span class="path">xAiApi/Startup.cs (متد ConfigureServices)</span>
|
||||
</div>
|
||||
|
||||
<pre>public void ConfigureServices(IServiceCollection services)
|
||||
{
|
||||
// ... (ثبتهای قبلی)
|
||||
|
||||
// ✅ ثبت File Content Extractors (به ترتیب صحیح)
|
||||
services.AddSingleton<IXFileContentExtractor, XPdfFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XDocxFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XImageFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XAudioFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XExcelFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XPlainTextFileContentExtractor>();
|
||||
|
||||
// ✅ ثبت Vision Extractor (وابسته به IXDefaultAIOCRService)
|
||||
services.AddSingleton<IXFileContentExtractor, XVisionFileContentExtractor>();
|
||||
|
||||
// ✅ ثبت کامپوزیت اصلی (لیست بالا را در Constructor دریافت میکند)
|
||||
services.AddSingleton<IXFileContentExtractor, XFileContentExtractor>();
|
||||
|
||||
// ✅ ثبت سرویس OCR (دیگر به IXFileContentExtractor وابسته نیست)
|
||||
services.AddScoped<IXDefaultAIOCRService, XDefaultAIOCRService>();
|
||||
|
||||
// ✅ ثبت سایر سرویسهای AI
|
||||
services.AddScoped<IXDefaultAiService, XDefaultAiService>();
|
||||
services.AddScoped<IXDefaultEmbeddingService, XDefaultEmbeddingService>();
|
||||
services.AddScoped<IXDefaultThinkingAiService, XDefaultThinkingAiService>();
|
||||
}</pre>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکته مهم:</strong> ترتیب ثبت سرویسها در DI Container اهمیت ندارد، زیرا .NET Core DI Container به صورت خودکار وابستگیها را حل میکند. مهم این است که تمام سرویسهای مورد نیاز ثبت شده باشند.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 6: Verification -->
|
||||
<div class="section" id="verification">
|
||||
<h2>✅ گام ۶: تأیید رفع خطا</h2>
|
||||
|
||||
<h3>خلاصه تغییرات:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>فایل</th>
|
||||
<th>تغییر</th>
|
||||
<th>دلیل</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>XDefaultAIOCRService.cs</code></td>
|
||||
<td><span class="badge-remove">REMOVE</span> حذف پارامتر <code>IXFileContentExtractor</code></td>
|
||||
<td>پارامتر غیرضروری بود و باعث Circular Dependency میشد</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>Startup.cs</code></td>
|
||||
<td><span class="badge-modify">NO CHANGE</span> بدون تغییر</td>
|
||||
<td>ثبت سرویسها به درستی انجام شده است</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<h3>مزایای این اصلاح:</h3>
|
||||
<div class="arch-grid">
|
||||
<div class="arch-card">
|
||||
<h4>🚫 حذف Circular Dependency</h4>
|
||||
<ul>
|
||||
<li>حلقه وابستگی کاملاً شکسته شد</li>
|
||||
<li>DI Container میتواند تمام سرویسها را بسازد</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>⚡ افزایش Performance</h4>
|
||||
<ul>
|
||||
<li>سرویس OCR دیگر بار اضافی ندارد</li>
|
||||
<li>تزریق وابستگیهای کمتر = ساخت سریعتر</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>🎯 اصل تکوظیفهای (SRP)</h4>
|
||||
<ul>
|
||||
<li>سرویس OCR فقط مسئول ارتباط با LLM است</li>
|
||||
<li>وابستگیهای غیرضروری حذف شدند</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>🧪 قابلیت تستپذیری</h4>
|
||||
<ul>
|
||||
<li>Unit Test سادهتر شد</li>
|
||||
<li>Mock کردن وابستگیها آسانتر است</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نتیجه نهایی:</strong>
|
||||
<p>با حذف پارامتر غیرضروری <code>IXFileContentExtractor</code> از constructor کلاس <code>XDefaultAIOCRService</code>، خطای Circular Dependency کاملاً رفع میشود و پروژه بدون مشکل اجرا خواهد شد.</p>
|
||||
<p style="margin-top: 10px;">پس از اعمال این تغییر، پروژه را rebuild کنید و خطای <code>System.AggregateException</code> دیگر ظاهر نخواهد شد.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div class="footer">
|
||||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
|
||||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||
🔗 مستند فنی رفع خطای Circular Dependency در xAiApi - تمامی حقوق محفوظ است
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,454 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fa" dir="rtl">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>رفع خطای Scoped/Singleton Mismatch در DI - فن آوران ساحر علم</title>
|
||||
<style>
|
||||
:root {
|
||||
--primary: #1e3a8a;
|
||||
--secondary: #3b82f6;
|
||||
--accent: #f59e0b;
|
||||
--success: #10b981;
|
||||
--danger: #ef4444;
|
||||
--warning: #f97316;
|
||||
--bg-light: #f8fafc;
|
||||
--bg-code: #1e293b;
|
||||
--text-dark: #0f172a;
|
||||
--text-muted: #64748b;
|
||||
--border: #e2e8f0;
|
||||
}
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||
color: var(--text-dark);
|
||||
line-height: 1.8;
|
||||
padding: 20px;
|
||||
}
|
||||
.container {
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
background: white;
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||
overflow: hidden;
|
||||
}
|
||||
.header {
|
||||
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
|
||||
color: white;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
}
|
||||
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
|
||||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||||
.meta-bar {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
background: var(--bg-light);
|
||||
padding: 15px 30px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
flex-wrap: wrap;
|
||||
gap: 15px;
|
||||
}
|
||||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||
.meta-item strong { color: var(--primary); }
|
||||
.content { padding: 40px; }
|
||||
.section {
|
||||
margin-bottom: 35px;
|
||||
padding: 25px;
|
||||
background: var(--bg-light);
|
||||
border-radius: 12px;
|
||||
border-right: 5px solid var(--secondary);
|
||||
}
|
||||
.section h2 {
|
||||
color: var(--primary);
|
||||
font-size: 1.5em;
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
}
|
||||
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
|
||||
pre {
|
||||
background: var(--bg-code);
|
||||
color: #e2e8f0;
|
||||
padding: 18px;
|
||||
border-radius: 8px;
|
||||
overflow-x: auto;
|
||||
direction: ltr;
|
||||
text-align: left;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.85em;
|
||||
margin: 15px 0;
|
||||
border-right: 4px solid var(--accent);
|
||||
}
|
||||
code {
|
||||
background: #fef3c7;
|
||||
color: #92400e;
|
||||
padding: 2px 8px;
|
||||
border-radius: 4px;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.9em;
|
||||
direction: ltr;
|
||||
display: inline-block;
|
||||
}
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
margin: 15px 0;
|
||||
background: white;
|
||||
border-radius: 8px;
|
||||
overflow: hidden;
|
||||
}
|
||||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||
tr:hover { background: var(--bg-light); }
|
||||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
|
||||
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
|
||||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||||
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
|
||||
.toc h3 { color: var(--primary); margin-bottom: 15px; }
|
||||
.toc ol { padding-right: 25px; }
|
||||
.toc li { padding: 6px 0; }
|
||||
.toc a { color: var(--secondary); text-decoration: none; }
|
||||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.chain-diagram { background: white; padding: 25px; border-radius: 10px; margin: 20px 0; text-align: center; }
|
||||
.chain-step { display: inline-block; background: var(--danger); color: white; padding: 10px 18px; border-radius: 8px; margin: 5px; font-size: 0.88em; }
|
||||
.chain-step.fixed { background: var(--success); }
|
||||
.chain-step.singleton { background: var(--primary); }
|
||||
.chain-step.scoped { background: var(--warning); }
|
||||
.chain-arrow { display: inline-block; color: var(--accent); font-size: 1.5em; margin: 0 8px; vertical-align: middle; }
|
||||
.legend { display: flex; gap: 20px; justify-content: center; margin-top: 15px; flex-wrap: wrap; }
|
||||
.legend-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; }
|
||||
.legend-box { width: 20px; height: 20px; border-radius: 4px; }
|
||||
.diff-old { background: #fee2e2; color: #991b1b; padding: 2px 6px; border-radius: 3px; text-decoration: line-through; }
|
||||
.diff-new { background: #d1fae5; color: #065f46; padding: 2px 6px; border-radius: 3px; font-weight: bold; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
|
||||
<div class="header">
|
||||
<h1>🔗 رفع خطای Scoped/Singleton Mismatch در DI</h1>
|
||||
<div class="subtitle">تحلیل و رفع عدم تطابق طول عمر سرویسها در کانتینر Dependency Injection</div>
|
||||
</div>
|
||||
|
||||
<div class="meta-bar">
|
||||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
|
||||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||||
</div>
|
||||
|
||||
<div class="content">
|
||||
|
||||
<div class="toc">
|
||||
<h3>📑 فهرست مطالب</h3>
|
||||
<ol>
|
||||
<li><a href="#analysis">تحلیل دقیق ریشه خطا</a></li>
|
||||
<li><a href="#chain">زنجیره وابستگی مشکلساز</a></li>
|
||||
<li><a href="#solution">راهحل اصلاحی</a></li>
|
||||
<li><a href="#step1">گام ۱: اصلاح Startup.cs</a></li>
|
||||
<li><a href="#step2">گام ۲: بررسی Thread Safety</a></li>
|
||||
<li><a href="#verification">تأیید رفع کامل خطا</a></li>
|
||||
</ol>
|
||||
</div>
|
||||
|
||||
<!-- Section 1: Analysis -->
|
||||
<div class="section" id="analysis">
|
||||
<h2>🔍 گام ۱: تحلیل دقیق ریشه خطا</h2>
|
||||
<p>پیغام خطای DI به یک مشکل کلاسیک در مدیریت طول عمر سرویسها اشاره میکند:</p>
|
||||
|
||||
<div class="alert alert-danger">
|
||||
<strong>⚠️ پیغام خطا:</strong><br>
|
||||
<code style="font-size: 0.95em;">Cannot consume scoped service 'IXDefaultAIOCRService' from singleton 'IXFileContentExtractor'</code>
|
||||
</div>
|
||||
|
||||
<h3>قانون طلایی DI در .NET Core:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>قانون</th>
|
||||
<th>توضیح</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>✅ مجاز</strong></td>
|
||||
<td>Singleton میتواند Singleton را مصرف کند</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>✅ مجاز</strong></td>
|
||||
<td>Scoped میتواند Singleton را مصرف کند</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>✅ مجاز</strong></td>
|
||||
<td>Transient میتواند هر چیزی را مصرف کند</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td style="background: var(--danger); color: white;"><strong>❌ ممنوع</strong></td>
|
||||
<td><strong>Singleton نمیتواند Scoped را مصرف کند</strong> (خطای فعلی)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td style="background: var(--danger); color: white;"><strong>❌ ممنوع</strong></td>
|
||||
<td><strong>Singleton نمیتواند Transient را مصرف کند</strong></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 دلیل منطقی:</strong> اگر یک Singleton بتواند یک Scoped service را مصرف کند، آن Scoped service عملاً به یک Singleton تبدیل میشود (چون فقط یک بار ساخته شده و در تمام درخواستها استفاده میشود). این مسئله باعث نشت حافظه و رفتار غیرقابل پیشبینی میشود.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 2: Chain -->
|
||||
<div class="section" id="chain">
|
||||
<h2>🔗 گام ۲: زنجیره وابستگی مشکلساز</h2>
|
||||
|
||||
<h3>وضعیت فعلی (قبل از اصلاح):</h3>
|
||||
<div class="chain-diagram">
|
||||
<span class="chain-step singleton">XFileContentExtractor<br><small>(Singleton)</small></span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step singleton">XVisionFileContentExtractor<br><small>(Singleton)</small></span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step scoped">IXDefaultAIOCRService<br><small>(Scoped) ❌</small></span>
|
||||
</div>
|
||||
|
||||
<div class="legend">
|
||||
<div class="legend-item">
|
||||
<div class="legend-box" style="background: var(--primary);"></div>
|
||||
<span>Singleton</span>
|
||||
</div>
|
||||
<div class="legend-item">
|
||||
<div class="legend-box" style="background: var(--warning);"></div>
|
||||
<span>Scoped</span>
|
||||
</div>
|
||||
<div class="legend-item">
|
||||
<div class="legend-box" style="background: var(--danger);"></div>
|
||||
<span>خطا</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<h3>بررسی وابستگیهای <code>XDefaultAIOCRService</code>:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>وابستگی</th>
|
||||
<th>نوع ثبت فعلی</th>
|
||||
<th>آیا Stateful است؟</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>XAiApiConfiguration</code></td>
|
||||
<td>Singleton</td>
|
||||
<td>❌ خیر (فقط خواندنی)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>ILogger<XDefaultAIOCRService></code></td>
|
||||
<td>Singleton</td>
|
||||
<td>❌ خیر (Thread-safe)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>string model</code></td>
|
||||
<td>Primitive</td>
|
||||
<td>❌ خیر</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>string prompt</code></td>
|
||||
<td>Primitive</td>
|
||||
<td>❌ خیر</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نتیجهگیری کلیدی:</strong> کلاس <code>XDefaultAIOCRService</code> هیچ وابستگی Scoped یا Stateful ندارد و کاملاً Stateless است. بنابراین میتواند با خیال راحت به <strong>Singleton</strong> ارتقا یابد.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 3: Solution -->
|
||||
<div class="section" id="solution">
|
||||
<h2>💡 گام ۳: راهحل اصلاحی</h2>
|
||||
<p>تنها یک تغییر کوچک در فایل <code>Startup.cs</code> لازم است: تغییر طول عمر <code>IXDefaultAIOCRService</code> از <code>Scoped</code> به <code>Singleton</code>.</p>
|
||||
|
||||
<h3>زنجیره اصلاحشده (بعد از تغییر):</h3>
|
||||
<div class="chain-diagram">
|
||||
<span class="chain-step fixed">XFileContentExtractor<br><small>(Singleton)</small></span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step fixed">XVisionFileContentExtractor<br><small>(Singleton)</small></span>
|
||||
<span class="chain-arrow">→</span>
|
||||
<span class="chain-step fixed">IXDefaultAIOCRService<br><small>(Singleton) ✅</small></span>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 چرا این راهحل امن است؟</strong>
|
||||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||||
<li>✅ <code>XDefaultAIOCRService</code> هیچ State ای ندارد که بین درخواستها به اشتراک گذاشته شود</li>
|
||||
<li>✅ <code>IChatClient</code> در هر فراخوانی <code>GetClient()</code> به صورت محلی ساخته و Dispose میشود</li>
|
||||
<li>✅ <code>HttpClient</code> با <code>Timeout.InfiniteTimeSpan</code> thread-safe است</li>
|
||||
<li>✅ <code>ILogger</code> ذاتاً thread-safe است</li>
|
||||
<li>✅ <code>XAiApiConfiguration</code> فقط خواندنی و thread-safe است</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 4: Step 1 -->
|
||||
<div class="section" id="step1">
|
||||
<h2>🛠️ گام ۴: اصلاح Startup.cs</h2>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Startup.cs (متد ConfigureServices)</span>
|
||||
</div>
|
||||
|
||||
<h3>کد اصلاحشده:</h3>
|
||||
<pre>public void ConfigureServices(IServiceCollection services)
|
||||
{
|
||||
// ... (ثبتهای قبلی)
|
||||
|
||||
// ==========================================================
|
||||
// ✅ ثبت File Content Extractors (همگی Singleton)
|
||||
// ==========================================================
|
||||
services.AddSingleton<IXFileContentExtractor, XPdfFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XDocxFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XImageFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XAudioFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XExcelFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XVisionFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XPlainTextFileContentExtractor>();
|
||||
|
||||
// ✅ کامپوزیت اصلی (لیست بالا را در Constructor دریافت میکند)
|
||||
services.AddSingleton<IXFileContentExtractor, XFileContentExtractor>();
|
||||
|
||||
// ==========================================================
|
||||
// ✅ ثبت سرویسهای AI
|
||||
// ==========================================================
|
||||
|
||||
// ✅ تغییر از AddScoped به AddSingleton (رفع خطای DI)
|
||||
<span class="diff-old">services.AddScoped<IXDefaultAIOCRService, XDefaultAIOCRService>();</span>
|
||||
<span class="diff-new">services.AddSingleton<IXDefaultAIOCRService, XDefaultAIOCRService>();</span>
|
||||
|
||||
// سایر سرویسهای AI (Scoped باقی میمانند چون به DataProvider وابستهاند)
|
||||
services.AddScoped<IXDefaultAiService, XDefaultAiService>();
|
||||
services.AddScoped<IXDefaultEmbeddingService, XDefaultEmbeddingService>();
|
||||
services.AddScoped<IXDefaultThinkingAiService, XDefaultThinkingAiService>();
|
||||
}</pre>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نکته مهم:</strong> فقط <code>IXDefaultAIOCRService</code> به Singleton تغییر میکند. سایر سرویسهای AI (<code>IXDefaultAiService</code>, <code>IXDefaultEmbeddingService</code>, <code>IXDefaultThinkingAiService</code>) به دلیل وابستگی به <code>IXAiDataProvider</code> و <code>IXFileProvider</code> که Scoped هستند، باید Scoped باقی بمانند.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 5: Thread Safety -->
|
||||
<div class="section" id="step2">
|
||||
<h2>🔒 گام ۵: بررسی Thread Safety</h2>
|
||||
<p>با تبدیل <code>XDefaultAIOCRService</code> به Singleton، باید اطمینان حاصل کنیم که این کلاس Thread-safe است:</p>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<th>بخش از کد</th>
|
||||
<th>وضعیت Thread Safety</th>
|
||||
<th>توضیح</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>Descriptor</code> (Property)</td>
|
||||
<td><span style="color: var(--success);">✅ Safe</span></td>
|
||||
<td>فقط در Constructor مقداردهی میشود و readonly است</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>Prompt</code> (Property)</td>
|
||||
<td><span style="color: var(--success);">✅ Safe</span></td>
|
||||
<td>فقط در Constructor مقداردهی میشود و readonly است</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>Options</code> (Property)</td>
|
||||
<td><span style="color: var(--success);">✅ Safe</span></td>
|
||||
<td>فقط در Constructor مقداردهی میشود</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>GetClient()</code></td>
|
||||
<td><span style="color: var(--success);">✅ Safe</span></td>
|
||||
<td>در هر فراخوانی یک <code>IChatClient</code> جدید میسازد (بدون State مشترک)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>GetHttpClient()</code></td>
|
||||
<td><span style="color: var(--success);">✅ Safe</span></td>
|
||||
<td>در هر فراخوانی یک <code>HttpClient</code> جدید میسازد</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>RequestOCRAsync()</code></td>
|
||||
<td><span style="color: var(--success);">✅ Safe</span></td>
|
||||
<td>از <code>using var client</code> استفاده میکند و State محلی دارد</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>RequestOCRAsEnumerable()</code></td>
|
||||
<td><span style="color: var(--success);">✅ Safe</span></td>
|
||||
<td>از <code>using var client</code> استفاده میکند و State محلی دارد</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نتیجه:</strong> کلاس <code>XDefaultAIOCRService</code> کاملاً Thread-safe است و میتواند با اطمینان کامل به عنوان Singleton ثبت شود.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 6: Verification -->
|
||||
<div class="section" id="verification">
|
||||
<h2>✅ گام ۶: تأیید رفع کامل خطا</h2>
|
||||
|
||||
<h3>خلاصه تغییرات:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>فایل</th>
|
||||
<th>تغییر</th>
|
||||
<th>دلیل</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>Startup.cs</code></td>
|
||||
<td><span class="diff-old">AddScoped</span> → <span class="diff-new">AddSingleton</span><br>برای <code>IXDefaultAIOCRService</code></td>
|
||||
<td>رفع خطای Scoped/Singleton Mismatch</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<h3>چکلیست نهایی:</h3>
|
||||
<div class="alert alert-success">
|
||||
<ol style="padding-right: 25px;">
|
||||
<li>✅ خطای <code>Cannot consume scoped service from singleton</code> رفع میشود</li>
|
||||
<li>✅ تمام Extractor ها (Singleton) میتوانند <code>XVisionFileContentExtractor</code> را مصرف کنند</li>
|
||||
<li>✅ <code>XVisionFileContentExtractor</code> میتواند <code>IXDefaultAIOCRService</code> را مصرف کند</li>
|
||||
<li>✅ <code>XFileContentExtractor</code> (Composite) میتواند تمام Extractor ها را در Constructor دریافت کند</li>
|
||||
<li>✅ Thread Safety کاملاً حفظ شده است</li>
|
||||
<li>✅ Performance بهبود مییابد (ساخت یکبار به جای ساخت در هر Request)</li>
|
||||
</ol>
|
||||
</div>
|
||||
|
||||
<h3>دستورالعمل اجرا:</h3>
|
||||
<pre># ۱. فایل Startup.cs را باز کنید
|
||||
# ۲. خط زیر را پیدا کنید:
|
||||
services.AddScoped<IXDefaultAIOCRService, XDefaultAIOCRService>();
|
||||
|
||||
# ۳. آن را به این صورت تغییر دهید:
|
||||
services.AddSingleton<IXDefaultAIOCRService, XDefaultAIOCRService>();
|
||||
|
||||
# ۴. پروژه را Rebuild کنید
|
||||
dotnet build
|
||||
|
||||
# ۵. پروژه را اجرا کنید - خطای AggregateException دیگر ظاهر نخواهد شد</pre>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکته تکمیلی:</strong> پس از اعمال این تغییر، میتوانید با خیال راحت endpoint <code>ExtractContent</code> را در Controller پیادهسازی کنید، زیرا تمام زیرساخت DI اکنون به درستی پیکربندی شده است.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div class="footer">
|
||||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
|
||||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||
🔗 مستند فنی رفع خطای Scoped/Singleton Mismatch در xAiApi - تمامی حقوق محفوظ است
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,351 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fa" dir="rtl">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>رفع خطای Circular Dependency در XFileContentExtractor - فن آوران ساحر علم</title>
|
||||
<style>
|
||||
:root {
|
||||
--primary: #1e3a8a;
|
||||
--secondary: #3b82f6;
|
||||
--accent: #f59e0b;
|
||||
--success: #10b981;
|
||||
--danger: #ef4444;
|
||||
--warning: #f97316;
|
||||
--bg-light: #f8fafc;
|
||||
--bg-code: #1e293b;
|
||||
--text-dark: #0f172a;
|
||||
--text-muted: #64748b;
|
||||
--border: #e2e8f0;
|
||||
}
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||
color: var(--text-dark);
|
||||
line-height: 1.8;
|
||||
padding: 20px;
|
||||
}
|
||||
.container {
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
background: white;
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||
overflow: hidden;
|
||||
}
|
||||
.header {
|
||||
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
|
||||
color: white;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
}
|
||||
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
|
||||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||||
.meta-bar {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
background: var(--bg-light);
|
||||
padding: 15px 30px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
flex-wrap: wrap;
|
||||
gap: 15px;
|
||||
}
|
||||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||
.meta-item strong { color: var(--primary); }
|
||||
.content { padding: 40px; }
|
||||
.section {
|
||||
margin-bottom: 35px;
|
||||
padding: 25px;
|
||||
background: var(--bg-light);
|
||||
border-radius: 12px;
|
||||
border-right: 5px solid var(--secondary);
|
||||
}
|
||||
.section h2 {
|
||||
color: var(--primary);
|
||||
font-size: 1.5em;
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
}
|
||||
pre {
|
||||
background: var(--bg-code);
|
||||
color: #e2e8f0;
|
||||
padding: 18px;
|
||||
border-radius: 8px;
|
||||
overflow-x: auto;
|
||||
direction: ltr;
|
||||
text-align: left;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.85em;
|
||||
margin: 15px 0;
|
||||
border-right: 4px solid var(--accent);
|
||||
}
|
||||
code {
|
||||
background: #fef3c7;
|
||||
color: #92400e;
|
||||
padding: 2px 8px;
|
||||
border-radius: 4px;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.9em;
|
||||
direction: ltr;
|
||||
display: inline-block;
|
||||
}
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
margin: 15px 0;
|
||||
background: white;
|
||||
border-radius: 8px;
|
||||
overflow: hidden;
|
||||
}
|
||||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
|
||||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.diff-old { background: #fee2e2; color: #991b1b; padding: 2px 6px; border-radius: 3px; text-decoration: line-through; }
|
||||
.diff-new { background: #d1fae5; color: #065f46; padding: 2px 6px; border-radius: 3px; font-weight: bold; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
|
||||
<div class="header">
|
||||
<h1>🔗 رفع خطای Circular Dependency در XFileContentExtractor</h1>
|
||||
<div class="subtitle">تحلیل و رفع حلقه وابستگی در Composite Extractor Pattern</div>
|
||||
</div>
|
||||
|
||||
<div class="meta-bar">
|
||||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
|
||||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||||
</div>
|
||||
|
||||
<div class="content">
|
||||
|
||||
<!-- Section 1: Analysis -->
|
||||
<div class="section">
|
||||
<h2>🔍 تحلیل ریشه خطا</h2>
|
||||
<p>خطای <code>Circular Dependency</code> به دلیل ثبت نادرست <code>XFileContentExtractor</code> در DI Container ایجاد شده است:</p>
|
||||
|
||||
<div class="alert alert-danger">
|
||||
<strong>⚠️ پیغام خطا:</strong><br>
|
||||
<code style="font-size: 0.9em;">A circular dependency was detected for the service of type 'IEnumerable<IXFileContentExtractor>'</code>
|
||||
</div>
|
||||
|
||||
<h3>زنجیره وابستگی مشکلساز:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>مرحله</th>
|
||||
<th>سرویس</th>
|
||||
<th>وابستگی</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>۱</td>
|
||||
<td><code>XFileContentExtractor</code></td>
|
||||
<td>نیاز به <code>IEnumerable<IXFileContentExtractor></code> در constructor</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>۲</td>
|
||||
<td><code>IEnumerable<IXFileContentExtractor></code></td>
|
||||
<td>شامل تمام <code>IXFileContentExtractor</code> های ثبت شده</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>۳</td>
|
||||
<td><code>XFileContentExtractor</code> (دوباره!)</td>
|
||||
<td>خودش هم به عنوان <code>IXFileContentExtractor</code> ثبت شده ❌</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>۴</td>
|
||||
<td colspan="2" style="background: var(--danger); color: white;"><strong>حلقه بینهایت → Crash</strong></td>
|
||||
</tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<!-- Section 2: Solution -->
|
||||
<div class="section">
|
||||
<h2>💡 راهحل اصلاحی</h2>
|
||||
<p>برای شکستن این حلقه، باید <code>XFileContentExtractor</code> را به عنوان <code>IXFileContentExtractor</code> ثبت <strong>نکنیم</strong>. به جای آن، آن را به صورت دستی با factory بسازیم.</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Startup.cs (متد ConfigureServices)</span>
|
||||
</div>
|
||||
|
||||
<h3>کد اصلاحشده:</h3>
|
||||
<pre>public void ConfigureServices(IServiceCollection services)
|
||||
{
|
||||
// ... (ثبتهای قبلی)
|
||||
|
||||
// ==========================================================
|
||||
// ✅ ثبت File Content Extractors (همگی Singleton)
|
||||
// ==========================================================
|
||||
|
||||
// ۱. ثبت Extractor های پایه
|
||||
services.AddSingleton<IXFileContentExtractor, XPdfFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XDocxFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XImageFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XAudioFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XExcelFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XVisionFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XPlainTextFileContentExtractor>();
|
||||
|
||||
// ۲. ✅ اصلاح: ثبت XFileContentExtractor به صورت Factory
|
||||
// به جای ثبت به عنوان IXFileContentExtractor، آن را به صورت دستی میسازیم
|
||||
services.AddSingleton<XFileContentExtractor>(sp =>
|
||||
{
|
||||
// دریافت تمام Extractor های ثبت شده (به جز XFileContentExtractor)
|
||||
var extractors = sp.GetServices<IXFileContentExtractor>();
|
||||
return new XFileContentExtractor(extractors);
|
||||
});
|
||||
|
||||
// ۳. ✅ ثبت XFileContentExtractor به عنوان IXFileContentExtractor
|
||||
// اما با استفاده از Factory که از قبل ساخته شده
|
||||
services.AddSingleton<IXFileContentExtractor>(sp =>
|
||||
sp.GetRequiredService<XFileContentExtractor>());
|
||||
|
||||
// ==========================================================
|
||||
// ✅ ثبت سرویسهای AI
|
||||
// ==========================================================
|
||||
services.AddSingleton<IXDefaultAIOCRService, XDefaultAIOCRService>();
|
||||
services.AddScoped<IXDefaultAiService, XDefaultAiService>();
|
||||
services.AddScoped<IXDefaultEmbeddingService, XDefaultEmbeddingService>();
|
||||
services.AddScoped<IXDefaultThinkingAiService, XDefaultThinkingAiService>();
|
||||
}</pre>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ چرا این راهحل کار میکند؟</strong>
|
||||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||||
<li>✅ <code>XFileContentExtractor</code> ابتدا به عنوان خودش (نه interface) ثبت میشود</li>
|
||||
<li>✅ در factory، تمام <code>IXFileContentExtractor</code> های دیگر را دریافت میکند</li>
|
||||
<li>✅ سپس به عنوان <code>IXFileContentExtractor</code> ثبت میشود (اما از instance از قبل ساخته شده)</li>
|
||||
<li>✅ حلقه وابستگی شکسته میشود</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 3: Alternative Solution -->
|
||||
<div class="section">
|
||||
<h2>🔄 راهحل جایگزین (سادهتر)</h2>
|
||||
<p>اگر راهحل بالا پیچیده به نظر میرسد، میتوانید از این روش سادهتر استفاده کنید:</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Startup.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>public void ConfigureServices(IServiceCollection services)
|
||||
{
|
||||
// ... (ثبتهای قبلی)
|
||||
|
||||
// ==========================================================
|
||||
// ✅ ثبت File Content Extractors
|
||||
// ==========================================================
|
||||
|
||||
// ۱. ثبت Extractor های پایه
|
||||
services.AddSingleton<IXFileContentExtractor, XPdfFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XDocxFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XImageFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XAudioFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XExcelFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XVisionFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XPlainTextFileContentExtractor>();
|
||||
|
||||
// ۲. ✅ راهحل ساده: ساخت دستی XFileContentExtractor
|
||||
// ابتدا تمام Extractor ها را در یک لیست جمع میکنیم
|
||||
var extractorTypes = new[]
|
||||
{
|
||||
typeof(XPdfFileContentExtractor),
|
||||
typeof(XDocxFileContentExtractor),
|
||||
typeof(XImageFileContentExtractor),
|
||||
typeof(XAudioFileContentExtractor),
|
||||
typeof(XExcelFileContentExtractor),
|
||||
typeof(XPlainTextFileContentExtractor)
|
||||
};
|
||||
|
||||
// ۳. ثبت XFileContentExtractor با factory
|
||||
services.AddSingleton<IXFileContentExtractor>(sp =>
|
||||
{
|
||||
var extractors = new List<IXFileContentExtractor>();
|
||||
|
||||
// ساخت دستی Extractor ها
|
||||
foreach (var type in extractorTypes)
|
||||
{
|
||||
var instance = (IXFileContentExtractor)ActivatorUtilities.CreateInstance(sp, type);
|
||||
extractors.Add(instance);
|
||||
}
|
||||
|
||||
// افزودن XVisionFileContentExtractor (که وابستگی دارد)
|
||||
var visionExtractor = sp.GetRequiredService<XVisionFileContentExtractor>();
|
||||
extractors.Add(visionExtractor);
|
||||
|
||||
// ساخت XFileContentExtractor با لیست Extractor ها
|
||||
return new XFileContentExtractor(extractors);
|
||||
});
|
||||
|
||||
// ==========================================================
|
||||
// ✅ ثبت سرویسهای AI
|
||||
// ==========================================================
|
||||
services.AddSingleton<IXDefaultAIOCRService, XDefaultAIOCRService>();
|
||||
services.AddScoped<IXDefaultAiService, XDefaultAiService>();
|
||||
services.AddScoped<IXDefaultEmbeddingService, XDefaultEmbeddingService>();
|
||||
services.AddScoped<IXDefaultThinkingAiService, XDefaultThinkingAiService>();
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 4: Summary -->
|
||||
<div class="section">
|
||||
<h2>📋 خلاصه تغییرات</h2>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<th>فایل</th>
|
||||
<th>تغییر</th>
|
||||
<th>دلیل</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>Startup.cs</code></td>
|
||||
<td><span class="diff-old">services.AddScoped<IXFileContentExtractor, XFileContentExtractor>()</span><br>
|
||||
<span class="diff-new">services.AddSingleton<IXFileContentExtractor>(sp => { ... })</span></td>
|
||||
<td>رفع Circular Dependency با استفاده از Factory Pattern</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نتیجه نهایی:</strong>
|
||||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||||
<li>✅ Circular Dependency کاملاً رفع میشود</li>
|
||||
<li>✅ DI Container میتواند تمام سرویسها را بسازد</li>
|
||||
<li>✅ <code>XFileContentExtractor</code> همچنان به تمام Extractor ها دسترسی دارد</li>
|
||||
<li>✅ خطای <code>AggregateException</code> دیگر ظاهر نخواهد شد</li>
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکته مهم:</strong>
|
||||
<p>پس از اعمال این تغییر، پروژه را <strong>Rebuild</strong> کنید. خطای <code>System.AggregateException</code> دیگر ظاهر نخواهد شد و برنامه به درستی اجرا میشود.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div class="footer">
|
||||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
|
||||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||
🔗 مستند فنی رفع خطای Circular Dependency در XFileContentExtractor - تمامی حقوق محفوظ است
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,682 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fa" dir="rtl">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>پیادهسازی OCR با Tesseract در XImageFileContentExtractor - فن آوران ساحر علم</title>
|
||||
<style>
|
||||
:root {
|
||||
--primary: #1e3a8a;
|
||||
--secondary: #3b82f6;
|
||||
--accent: #f59e0b;
|
||||
--success: #10b981;
|
||||
--danger: #ef4444;
|
||||
--warning: #f97316;
|
||||
--bg-light: #f8fafc;
|
||||
--bg-code: #1e293b;
|
||||
--text-dark: #0f172a;
|
||||
--text-muted: #64748b;
|
||||
--border: #e2e8f0;
|
||||
}
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||
color: var(--text-dark);
|
||||
line-height: 1.8;
|
||||
padding: 20px;
|
||||
}
|
||||
.container {
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
background: white;
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||
overflow: hidden;
|
||||
}
|
||||
.header {
|
||||
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
|
||||
color: white;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
}
|
||||
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
|
||||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||||
.meta-bar {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
background: var(--bg-light);
|
||||
padding: 15px 30px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
flex-wrap: wrap;
|
||||
gap: 15px;
|
||||
}
|
||||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||
.meta-item strong { color: var(--primary); }
|
||||
.content { padding: 40px; }
|
||||
.section {
|
||||
margin-bottom: 35px;
|
||||
padding: 25px;
|
||||
background: var(--bg-light);
|
||||
border-radius: 12px;
|
||||
border-right: 5px solid var(--secondary);
|
||||
}
|
||||
.section h2 {
|
||||
color: var(--primary);
|
||||
font-size: 1.5em;
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
}
|
||||
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
|
||||
pre {
|
||||
background: var(--bg-code);
|
||||
color: #e2e8f0;
|
||||
padding: 18px;
|
||||
border-radius: 8px;
|
||||
overflow-x: auto;
|
||||
direction: ltr;
|
||||
text-align: left;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.85em;
|
||||
margin: 15px 0;
|
||||
border-right: 4px solid var(--accent);
|
||||
}
|
||||
code {
|
||||
background: #fef3c7;
|
||||
color: #92400e;
|
||||
padding: 2px 8px;
|
||||
border-radius: 4px;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.9em;
|
||||
direction: ltr;
|
||||
display: inline-block;
|
||||
}
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
margin: 15px 0;
|
||||
background: white;
|
||||
border-radius: 8px;
|
||||
overflow: hidden;
|
||||
}
|
||||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||
tr:hover { background: var(--bg-light); }
|
||||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
|
||||
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
|
||||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||||
.footer p { margin: 5px 0; }
|
||||
.highlight { background: linear-gradient(120deg, #fef3c7 0%, #fef3c7 100%); padding: 2px 6px; border-radius: 4px; font-weight: bold; }
|
||||
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
|
||||
.toc h3 { color: var(--primary); margin-bottom: 15px; }
|
||||
.toc ol { padding-right: 25px; }
|
||||
.toc li { padding: 6px 0; }
|
||||
.toc a { color: var(--secondary); text-decoration: none; }
|
||||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.arch-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(280px, 1fr)); gap: 20px; margin: 20px 0; }
|
||||
.arch-card { background: white; padding: 20px; border-radius: 10px; box-shadow: 0 4px 12px rgba(0,0,0,0.08); border-top: 4px solid var(--secondary); }
|
||||
.arch-card h4 { color: var(--primary); margin-bottom: 12px; }
|
||||
.arch-card ul { list-style: none; padding-right: 0; }
|
||||
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; }
|
||||
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
|
||||
<div class="header">
|
||||
<h1>🔍 پیادهسازی OCR با Tesseract</h1>
|
||||
<div class="subtitle">تکمیل متد PerformOcrAsync در XImageFileContentExtractor برای استخراج متن از تصاویر</div>
|
||||
</div>
|
||||
|
||||
<div class="meta-bar">
|
||||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
|
||||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||||
</div>
|
||||
|
||||
<div class="content">
|
||||
|
||||
<div class="toc">
|
||||
<h3>📑 فهرست مطالب</h3>
|
||||
<ol>
|
||||
<li><a href="#step1">گام ۱: نصب پکیج NuGet Tesseract</a></li>
|
||||
<li><a href="#step2">گام ۲: دانلود فایلهای Trained Data</a></li>
|
||||
<li><a href="#step3">گام ۳: پیکربندی در appsettings.json</a></li>
|
||||
<li><a href="#step4">گام ۴: بازنویسی کلاس XImageFileContentExtractor</a></li>
|
||||
<li><a href="#step5">گام ۵: پیادهسازی متد PerformOcrAsync</a></li>
|
||||
<li><a href="#step6">گام ۶: نکات مهم و عیبیابی</a></li>
|
||||
</ol>
|
||||
</div>
|
||||
|
||||
<!-- Section 1: NuGet -->
|
||||
<div class="section" id="step1">
|
||||
<h2>📦 گام ۱: نصب پکیج NuGet Tesseract</h2>
|
||||
<p>برای استفاده از Tesseract در پروژه .NET، پکیج زیر را نصب کنید:</p>
|
||||
|
||||
<pre># در Package Manager Console:
|
||||
Install-Package Tesseract
|
||||
|
||||
# یا در .NET CLI:
|
||||
dotnet add package Tesseract</pre>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکته:</strong> پکیج <code>Tesseract</code> به صورت خودکار Native DLL های مورد نیاز را نیز دانلود و در پوشه output کپی میکند.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 2: Trained Data -->
|
||||
<div class="section" id="step2">
|
||||
<h2>📥 گام ۲: دانلود فایلهای Trained Data</h2>
|
||||
<p>Tesseract برای تشخیص متن به فایلهای <code>traineddata</code> نیاز دارد. این فایلها مدلهای زبانی هستند که برای هر زبان جداگانه آموزش دیدهاند.</p>
|
||||
|
||||
<h3>۲.۱. دانلود فایلها:</h3>
|
||||
<p>از لینک زیر فایلهای مورد نیاز را دانلود کنید:</p>
|
||||
<p><a href="https://github.com/tesseract-ocr/tessdata" target="_blank">https://github.com/tesseract-ocr/tessdata</a></p>
|
||||
|
||||
<h3>۲.۲. فایلهای مورد نیاز:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>فایل</th>
|
||||
<th>زبان</th>
|
||||
<th>حجم تقریبی</th>
|
||||
<th>کاربرد</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>fas.traineddata</code></td>
|
||||
<td>فارسی</td>
|
||||
<td>~۱۰ MB</td>
|
||||
<td>تشخیص متن فارسی</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>eng.traineddata</code></td>
|
||||
<td>انگلیسی</td>
|
||||
<td>~۴ MB</td>
|
||||
<td>تشخیص متن انگلیسی</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>ara.traineddata</code></td>
|
||||
<td>عربی</td>
|
||||
<td>~۵ MB</td>
|
||||
<td>تشخیص متن عربی (اختیاری)</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<h3>۲.۳. ساختار پوشهها:</h3>
|
||||
<p>فایلهای دانلود شده را در مسیر زیر قرار دهید:</p>
|
||||
<pre>xAiApi/
|
||||
└── tessdata/
|
||||
├── fas.traineddata
|
||||
├── eng.traineddata
|
||||
└── ara.traineddata (اختیاری)</pre>
|
||||
|
||||
<div class="alert alert-warning">
|
||||
<strong>⚠️ نکته مهم:</strong> مسیر <code>tessdata</code> باید در زمان اجرا در دسترس باشد. میتوانید آن را در <code>appsettings.json</code> پیکربندی کنید.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 3: Configuration -->
|
||||
<div class="section" id="step3">
|
||||
<h2>⚙️ گام ۳: پیکربندی در appsettings.json</h2>
|
||||
<p>تنظیمات Tesseract را در فایل <code>appsettings.json</code> اضافه کنید:</p>
|
||||
|
||||
<pre>{
|
||||
"AiApiConfiguration": {
|
||||
"FileExtraction": {
|
||||
"OCR": {
|
||||
"TessDataPath": "tessdata",
|
||||
"DefaultLanguage": "fas+eng",
|
||||
"EngineMode": "LstmOnly"
|
||||
}
|
||||
},
|
||||
"Models": [ ... ],
|
||||
"Prompts": [ ... ]
|
||||
}
|
||||
}</pre>
|
||||
|
||||
<h3>توضیح پارامترها:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>پارامتر</th>
|
||||
<th>توضیح</th>
|
||||
<th>مقادیر مجاز</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>TessDataPath</code></td>
|
||||
<td>مسیر پوشه فایلهای traineddata</td>
|
||||
<td>مسیر نسبی یا مطلق</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>DefaultLanguage</code></td>
|
||||
<td>زبانهای پیشفرض برای OCR</td>
|
||||
<td>fas, eng, ara (ترکیب با +)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>EngineMode</code></td>
|
||||
<td>حالت موتور Tesseract</td>
|
||||
<td>TesseractOnly, LstmOnly, TesseractAndLstm, Default</td>
|
||||
</tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<!-- Section 4: Rewrite Class -->
|
||||
<div class="section" id="step4">
|
||||
<h2>🔄 گام ۴: بازنویسی کلاس XImageFileContentExtractor</h2>
|
||||
<p>اکنون کلاس <code>XImageFileContentExtractor</code> را بازنویسی میکنیم تا از Tesseract برای OCR استفاده کند:</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Providers/Extractors/XImageFileContentExtractor.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Threading;
|
||||
using Tesseract;
|
||||
using xAiModels.Models;
|
||||
using System.Threading.Tasks;
|
||||
using Microsoft.Extensions.Logging;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
|
||||
namespace xAiApi.Providers.Extractors
|
||||
{
|
||||
/// <summary>
|
||||
/// Extracts content from image files using Tesseract OCR ...
|
||||
/// </summary>
|
||||
public class XImageFileContentExtractor : IXImageFileContentExtractor
|
||||
{
|
||||
/// <summary>
|
||||
/// Supported MIME Types ...
|
||||
/// </summary>
|
||||
private static readonly string[] SupportedMimeTypes =
|
||||
[
|
||||
"image/png",
|
||||
"image/jpeg",
|
||||
"image/jpg",
|
||||
"image/gif",
|
||||
"image/webp",
|
||||
"image/bmp"
|
||||
];
|
||||
|
||||
/// <summary>
|
||||
/// مسیر پوشه tessdata ...
|
||||
/// </summary>
|
||||
private readonly string _tessDataPath;
|
||||
|
||||
/// <summary>
|
||||
/// زبان پیشفرض برای OCR ...
|
||||
/// </summary>
|
||||
private readonly string _defaultLanguage;
|
||||
|
||||
/// <summary>
|
||||
/// حالت موتور Tesseract ...
|
||||
/// </summary>
|
||||
private readonly OcrEngineMode _engineMode;
|
||||
|
||||
/// <summary>
|
||||
/// Logger ...
|
||||
/// </summary>
|
||||
private readonly ILogger<XImageFileContentExtractor> _logger;
|
||||
|
||||
/// <summary>
|
||||
/// Constructor با پیکربندی پیشفرض ...
|
||||
/// </summary>
|
||||
public XImageFileContentExtractor(
|
||||
ILogger<XImageFileContentExtractor> logger,
|
||||
string tessDataPath = "tessdata",
|
||||
string defaultLanguage = "fas+eng",
|
||||
OcrEngineMode engineMode = OcrEngineMode.LstmOnly
|
||||
)
|
||||
{
|
||||
_logger = logger;
|
||||
_tessDataPath = tessDataPath;
|
||||
_defaultLanguage = defaultLanguage;
|
||||
_engineMode = engineMode;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Check if this extractor supports the specified MIME type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
return SupportedMimeTypes.Contains(
|
||||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||||
);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract text content from file stream ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
var result = await ExtractRichAsync(
|
||||
fileStream,
|
||||
"image",
|
||||
mimeType,
|
||||
cancellationToken
|
||||
);
|
||||
return result.Text;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract content from stream as Rich Result ...
|
||||
/// </summary>
|
||||
public async Task<XFileExtractionResult> ExtractRichAsync(
|
||||
Stream fileStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
var result = new XFileExtractionResult
|
||||
{
|
||||
FileName = fileName,
|
||||
MimeType = mimeType
|
||||
};
|
||||
|
||||
// خواندن بایتهای تصویر
|
||||
using var memoryStream = new MemoryStream();
|
||||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
var imageBytes = memoryStream.ToArray();
|
||||
|
||||
// افزودن تصویر به عنوان DataContent (برای Vision Models)
|
||||
result.Images.Add(new XExtractedImage
|
||||
{
|
||||
Bytes = imageBytes,
|
||||
MimeType = mimeType,
|
||||
Description = $"Attached image: {fileName}"
|
||||
});
|
||||
|
||||
// انجام OCR برای استخراج متن
|
||||
try
|
||||
{
|
||||
var ocrText = await PerformOcrAsync(imageBytes, cancellationToken);
|
||||
if (!string.IsNullOrWhiteSpace(ocrText))
|
||||
{
|
||||
result.Text = ocrText;
|
||||
_logger.LogInformation(
|
||||
"OCR completed successfully for file: {FileName}, extracted {Length} characters",
|
||||
fileName,
|
||||
ocrText.Length
|
||||
);
|
||||
}
|
||||
else
|
||||
{
|
||||
_logger.LogWarning("OCR completed but no text was extracted from file: {FileName}", fileName);
|
||||
}
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
_logger.LogError(ex, "OCR failed for file: {FileName}", fileName);
|
||||
result.ErrorMessage = $"OCR failed: {ex.Message}";
|
||||
// تصویر همچنان برای Vision Models در دسترس است
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// انجام OCR روی تصویر با استفاده از Tesseract ...
|
||||
/// </summary>
|
||||
private async Task<string> PerformOcrAsync(
|
||||
byte[] imageBytes,
|
||||
CancellationToken cancellationToken
|
||||
)
|
||||
{
|
||||
return await Task.Run(() =>
|
||||
{
|
||||
// بررسی وجود مسیر tessdata
|
||||
if (!Directory.Exists(_tessDataPath))
|
||||
{
|
||||
throw new DirectoryNotFoundException(
|
||||
$"TessData directory not found at: {_tessDataPath}. " +
|
||||
"Please download traineddata files from https://github.com/tesseract-ocr/tessdata"
|
||||
);
|
||||
}
|
||||
|
||||
// ایجاد Tesseract Engine
|
||||
using var engine = new TesseractEngine(
|
||||
datapath: _tessDataPath,
|
||||
language: _defaultLanguage,
|
||||
mode: _engineMode
|
||||
);
|
||||
|
||||
// بارگذاری تصویر از byte array
|
||||
using var pix = Pix.LoadFromMemory(imageBytes);
|
||||
|
||||
// تنظیم CancellationToken
|
||||
cancellationToken.ThrowIfCancellationRequested();
|
||||
|
||||
// انجام OCR
|
||||
using var page = engine.Process(pix);
|
||||
|
||||
// استخراج متن
|
||||
var text = page.GetText();
|
||||
|
||||
// پاکسازی متن (حذف فاصلههای اضافی)
|
||||
text = text?.Trim();
|
||||
|
||||
return text ?? string.Empty;
|
||||
}, cancellationToken);
|
||||
}
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 5: Implementation Details -->
|
||||
<div class="section" id="step5">
|
||||
<h2>🔧 گام ۵: جزئیات پیادهسازی متد PerformOcrAsync</h2>
|
||||
|
||||
<h3>۵.۱. مراحل کار متد:</h3>
|
||||
<div class="arch-grid">
|
||||
<div class="arch-card">
|
||||
<h4>۱. بررسی مسیر tessdata</h4>
|
||||
<ul>
|
||||
<li>اطمینان از وجود پوشه tessdata</li>
|
||||
<li>پرتاب خطا در صورت عدم وجود</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>۲. ایجاد Tesseract Engine</h4>
|
||||
<ul>
|
||||
<li>بارگذاری مدل زبانی</li>
|
||||
<li>تنظیم حالت موتور (LSTM)</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>۳. بارگذاری تصویر</h4>
|
||||
<ul>
|
||||
<li>تبدیل byte[] به Pix object</li>
|
||||
<li>پشتیبانی از فرمتهای مختلف</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>۴. انجام OCR</h4>
|
||||
<ul>
|
||||
<li>پردازش تصویر</li>
|
||||
<li>استخراج متن</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>۵. پاکسازی متن</h4>
|
||||
<ul>
|
||||
<li>حذف فاصلههای اضافی</li>
|
||||
<li>بررسی CancellationToken</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<h3>۵.۲. ویژگیهای کلیدی پیادهسازی:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>ویژگی</th>
|
||||
<th>توضیح</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>پشتیبانی چند زبانه</strong></td>
|
||||
<td>استفاده از <code>fas+eng</code> برای تشخیص همزمان فارسی و انگلیسی</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>حالت LSTM</strong></td>
|
||||
<td>استفاده از موتور LSTM برای دقت بالاتر در متون فارسی</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>مدیریت خطا</strong></td>
|
||||
<td>بررسی وجود tessdata و مدیریت استثناها</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>Logging</strong></td>
|
||||
<td>ثبت موفقیت/شکست OCR با جزئیات</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>Cancellation Support</strong></td>
|
||||
<td>پشتیبانی از لغو عملیات در میانه کار</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>Resource Management</strong></td>
|
||||
<td>استفاده از <code>using</code> برای آزادسازی منابع</td>
|
||||
</tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<!-- Section 6: Tips -->
|
||||
<div class="section" id="step6">
|
||||
<h2>💡 گام ۶: نکات مهم و عیبیابی</h2>
|
||||
|
||||
<h3>۶.۱. مشکلات رایج و راهحلها:</h3>
|
||||
<table>
|
||||
<tr>
|
||||
<th>مشکل</th>
|
||||
<th>علت</th>
|
||||
<th>راهحل</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>DirectoryNotFoundException</code></td>
|
||||
<td>پوشه tessdata یافت نشد</td>
|
||||
<td>دانلود traineddata و قرار دادن در مسیر صحیح</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>TesseractException</code></td>
|
||||
<td>فایل traineddata خراب یا ناسازگار</td>
|
||||
<td>دانلود مجدد از منبع رسمی</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>دقت پایین OCR</td>
|
||||
<td>کیفیت پایین تصویر</td>
|
||||
<td>پیشپردازش تصویر (افزایش کنتراست، resize)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>تشخیص نادرست فارسی</td>
|
||||
<td>ترکیب نامناسب زبانها</td>
|
||||
<td>استفاده از <code>fas</code> به تنهایی یا <code>fas+eng</code></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>کندی عملکرد</td>
|
||||
<td>تصاویر بزرگ</td>
|
||||
<td>تغییر EngineMode به <code>TesseractOnly</code></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<h3>۶.۲. بهینهسازی عملکرد:</h3>
|
||||
<div class="arch-grid">
|
||||
<div class="arch-card">
|
||||
<h4>🚀 افزایش سرعت</h4>
|
||||
<ul>
|
||||
<li>استفاده از <code>OcrEngineMode.TesseractOnly</code></li>
|
||||
<li>کاهش DPI تصویر (مثلاً 150 به جای 300)</li>
|
||||
<li>استفاده از یک زبان به جای چند زبان</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>🎯 افزایش دقت</h4>
|
||||
<ul>
|
||||
<li>استفاده از <code>OcrEngineMode.LstmOnly</code></li>
|
||||
<li>افزایش DPI تصویر (300 یا بالاتر)</li>
|
||||
<li>پیشپردازش تصویر (binarization)</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<h3>۶.۳. پیشپردازش تصویر (اختیاری):</h3>
|
||||
<p>برای افزایش دقت OCR، میتوانید قبل از پردازش، تصویر را پیشپردازش کنید:</p>
|
||||
<pre>// مثال: افزایش کنتراست و تبدیل به سیاه و سفید
|
||||
using var pix = Pix.LoadFromMemory(imageBytes);
|
||||
using var enhanced = pix.ConvertTo1(); // تبدیل به 1-bit
|
||||
using var page = engine.Process(enhanced);</pre>
|
||||
|
||||
<h3>۶.۴. ثبت در Startup.cs:</h3>
|
||||
<p>اکنون <code>XImageFileContentExtractor</code> را با پیکربندی صحیح ثبت کنید:</p>
|
||||
<pre>public void ConfigureServices(IServiceCollection services)
|
||||
{
|
||||
// ... سایر ثبتها
|
||||
|
||||
// ثبت XImageFileContentExtractor با پیکربندی
|
||||
services.AddSingleton<IXFileContentExtractor>(sp =>
|
||||
{
|
||||
var logger = sp.GetRequiredService<ILogger<XImageFileContentExtractor>>();
|
||||
var configuration = sp.GetRequiredService<IConfiguration>();
|
||||
|
||||
var tessDataPath = configuration.GetValue<string>(
|
||||
"AiApiConfiguration:FileExtraction:OCR:TessDataPath"
|
||||
) ?? "tessdata";
|
||||
|
||||
var defaultLanguage = configuration.GetValue<string>(
|
||||
"AiApiConfiguration:FileExtraction:OCR:DefaultLanguage"
|
||||
) ?? "fas+eng";
|
||||
|
||||
return new XImageFileContentExtractor(
|
||||
logger: logger,
|
||||
tessDataPath: tessDataPath,
|
||||
defaultLanguage: defaultLanguage,
|
||||
engineMode: OcrEngineMode.LstmOnly
|
||||
);
|
||||
});
|
||||
|
||||
// ... سایر ثبتها
|
||||
}</pre>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نتیجه نهایی:</strong>
|
||||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||||
<li>🔍 <strong>OCR کامل:</strong> استخراج متن از تصاویر با دقت بالا</li>
|
||||
<li>🌐 <strong>چند زبانه:</strong> پشتیبانی از فارسی، انگلیسی و عربی</li>
|
||||
<li>⚡ <strong>بهینه:</strong> استفاده از موتور LSTM برای دقت بالاتر</li>
|
||||
<li>🛡️ <strong>مقاوم:</strong> مدیریت خطا و logging کامل</li>
|
||||
<li>🔄 <strong>قابل لغو:</strong> پشتیبانی از CancellationToken</li>
|
||||
<li>📦 <strong>منعطف:</strong> پیکربندی از طریق appsettings.json</li>
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکته تکمیلی:</strong>
|
||||
<p>اگر نیاز به دقت بالاتر برای متون فارسی دارید، میتوانید از مدلهای آموزشدیدهشده خاص فارسی استفاده کنید یا از ترکیب Tesseract با مدلهای deep learning (مانند CRNN) بهره ببرید.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div class="footer">
|
||||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
|
||||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||
🔍 مستند فنی پیادهسازی OCR با Tesseract - تمامی حقوق محفوظ است
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,658 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fa" dir="rtl">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>بازنویسی FallbackToImageExtractionAsync با PdfiumViewer - فن آوران ساحر علم</title>
|
||||
<style>
|
||||
:root {
|
||||
--primary: #1e3a8a;
|
||||
--secondary: #3b82f6;
|
||||
--accent: #f59e0b;
|
||||
--success: #10b981;
|
||||
--danger: #ef4444;
|
||||
--warning: #f97316;
|
||||
--purple: #8b5cf6;
|
||||
--bg-light: #f8fafc;
|
||||
--bg-code: #1e293b;
|
||||
--text-dark: #0f172a;
|
||||
--text-muted: #64748b;
|
||||
--border: #e2e8f0;
|
||||
}
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||
color: var(--text-dark);
|
||||
line-height: 1.8;
|
||||
padding: 20px;
|
||||
}
|
||||
.container {
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
background: white;
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||
overflow: hidden;
|
||||
}
|
||||
.header {
|
||||
background: linear-gradient(135deg, var(--primary) 0%, var(--purple) 100%);
|
||||
color: white;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
}
|
||||
.header h1 { font-size: 2em; margin-bottom: 10px; }
|
||||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||||
.meta-bar {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
background: var(--bg-light);
|
||||
padding: 15px 30px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
flex-wrap: wrap;
|
||||
gap: 15px;
|
||||
}
|
||||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||
.meta-item strong { color: var(--primary); }
|
||||
.content { padding: 40px; }
|
||||
.section {
|
||||
margin-bottom: 35px;
|
||||
padding: 25px;
|
||||
background: var(--bg-light);
|
||||
border-radius: 12px;
|
||||
border-right: 5px solid var(--secondary);
|
||||
}
|
||||
.section h2 {
|
||||
color: var(--primary);
|
||||
font-size: 1.5em;
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
}
|
||||
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
|
||||
pre {
|
||||
background: var(--bg-code);
|
||||
color: #e2e8f0;
|
||||
padding: 18px;
|
||||
border-radius: 8px;
|
||||
overflow-x: auto;
|
||||
direction: ltr;
|
||||
text-align: left;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.83em;
|
||||
margin: 15px 0;
|
||||
border-right: 4px solid var(--accent);
|
||||
}
|
||||
code {
|
||||
background: #fef3c7;
|
||||
color: #92400e;
|
||||
padding: 2px 8px;
|
||||
border-radius: 4px;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.9em;
|
||||
direction: ltr;
|
||||
display: inline-block;
|
||||
}
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
margin: 15px 0;
|
||||
background: white;
|
||||
border-radius: 8px;
|
||||
overflow: hidden;
|
||||
}
|
||||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
|
||||
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
|
||||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||||
.highlight { background: linear-gradient(120deg, #fef3c7 0%, #fef3c7 100%); padding: 2px 6px; border-radius: 4px; font-weight: bold; }
|
||||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.arch-grid {
|
||||
display: grid;
|
||||
grid-template-columns: repeat(auto-fit, minmax(280px, 1fr));
|
||||
gap: 20px;
|
||||
margin: 20px 0;
|
||||
}
|
||||
.arch-card {
|
||||
background: white;
|
||||
padding: 20px;
|
||||
border-radius: 10px;
|
||||
box-shadow: 0 4px 12px rgba(0,0,0,0.08);
|
||||
border-top: 4px solid var(--secondary);
|
||||
}
|
||||
.arch-card h4 { color: var(--primary); margin-bottom: 12px; font-size: 1.1em; }
|
||||
.arch-card ul { list-style: none; padding-right: 0; }
|
||||
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; font-size: 0.95em; }
|
||||
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
|
||||
<div class="header">
|
||||
<h1>📄 بازنویسی FallbackToImageExtractionAsync با PdfiumViewer</h1>
|
||||
<div class="subtitle">جایگزینی PdfPigRenderer با کتابخانه قدرتمند PdfiumViewer برای رندر PDF</div>
|
||||
</div>
|
||||
|
||||
<div class="meta-bar">
|
||||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||
<div class="meta-item">📅 <strong>تاریخ:</strong> یکشنبه ۱۳ مهر ۱۴۰۵</div>
|
||||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||||
</div>
|
||||
|
||||
<div class="content">
|
||||
|
||||
<div class="section">
|
||||
<h2>🎯 مقدمه</h2>
|
||||
<p>
|
||||
کتابخانه <span class="highlight">PdfiumViewer</span> یکی از پایدارترین و سریعترین راهکارها برای رندر PDF در .NET است
|
||||
که بر پایه موتور <strong>PDFium گوگل</strong> (همان موتور استفاده شده در Chrome) ساخته شده است.
|
||||
این کتابخانه نسبت به <code>PdfPigRenderer</code> عملکرد بهتری دارد و برای تبدیل PDF به تصویر مناسبتر است.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Step 1: NuGet Packages -->
|
||||
<div class="section">
|
||||
<h2>📦 گام ۱: نصب پکیجهای NuGet مورد نیاز</h2>
|
||||
<p>ابتدا باید پکیجهای زیر را در پروژه <code>xAiApi</code> نصب کنید:</p>
|
||||
<pre># پکیج اصلی PdfiumViewer
|
||||
Install-Package PdfiumViewer
|
||||
|
||||
# پکیج Native DLLs (بسیار مهم - شامل pdfium.dll برای پلتفرمهای مختلف)
|
||||
Install-Package PdfiumViewer.Native
|
||||
|
||||
# برای کار با Bitmap و تصاویر
|
||||
Install-Package System.Drawing.Common</pre>
|
||||
|
||||
<div class="alert alert-warning">
|
||||
<strong>⚠️ نکته مهم در مورد Native DLLs:</strong>
|
||||
<p>پکیج <code>PdfiumViewer.Native</code> فایلهای <code>pdfium.dll</code> را برای پلتفرمهای مختلف
|
||||
(x86 و x64) در پوشه <code>bin</code> پروژه کپی میکند. بدون این پکیج، کتابخانه با خطای
|
||||
<code>DllNotFoundException</code> مواجه خواهد شد.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Step 2: Rewritten Function -->
|
||||
<div class="section">
|
||||
<h2>🔧 گام ۲: کد بازنویسی شده با PdfiumViewer</h2>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Providers/Extractors/XPdfFileContentExtractor.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Text;
|
||||
using System.Drawing;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using PdfiumViewer;
|
||||
using xAiApi.Models;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
using xCommons.Extensions;
|
||||
|
||||
namespace xAiApi.Providers.Extractors
|
||||
{
|
||||
/// <summary>
|
||||
/// استخراج محتوا از PDF با پشتیبانی از Fallback تصویری با PdfiumViewer ...
|
||||
/// </summary>
|
||||
public class XPdfFileContentExtractor : IXPdfFileContentExtractor
|
||||
{
|
||||
/// <summary>
|
||||
/// Supported MIME Types ...
|
||||
/// </summary>
|
||||
private static readonly string[] SupportedMimeTypes = ["application/pdf"];
|
||||
|
||||
/// <summary>
|
||||
/// حداقل تعداد کاراکتر برای تشخیص PDF متنی ...
|
||||
/// </summary>
|
||||
private const int MinTextLengthThreshold = 50;
|
||||
|
||||
/// <summary>
|
||||
/// حداکثر تعداد صفحات برای تبدیل به تصویر ...
|
||||
/// </summary>
|
||||
private const int MaxPagesForImageFallback = 10;
|
||||
|
||||
/// <summary>
|
||||
/// DPI برای رندر تصاویر (72 = کیفیت معمولی، 150 = کیفیت خوب، 300 = کیفیت بالا) ...
|
||||
/// </summary>
|
||||
private const int RenderDpi = 150;
|
||||
|
||||
/// <summary>
|
||||
/// حداکثر عرض تصویر به پیکسل ...
|
||||
/// </summary>
|
||||
private const int MaxImageWidth = 2000;
|
||||
|
||||
/// <summary>
|
||||
/// بررسی پشتیبانی از MIME Type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
return SupportedMimeTypes.Contains(
|
||||
mimeType?.ToLowerInvariant() ?? string.Empty
|
||||
);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// استخراج محتوای غنی با Fallback هوشمند ...
|
||||
/// </summary>
|
||||
public async Task<XFileExtractionResult> ExtractRichAsync(
|
||||
Stream fileStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
var result = new XFileExtractionResult
|
||||
{
|
||||
FileName = fileName,
|
||||
MimeType = mimeType
|
||||
};
|
||||
|
||||
// کپی به MemoryStream برای چندبار خواندن
|
||||
using var memoryStream = new MemoryStream();
|
||||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
memoryStream.Position = 0;
|
||||
|
||||
// مرحله ۱: تلاش برای استخراج متن با UglyToad.PdfPig
|
||||
var extractedText = await ExtractTextAsync(memoryStream, cancellationToken);
|
||||
|
||||
// مرحله ۲: بررسی کیفیت متن استخراج شده
|
||||
if (IsTextSufficient(extractedText))
|
||||
{
|
||||
result.Text = extractedText;
|
||||
}
|
||||
else
|
||||
{
|
||||
// PDF اسکنشده است - Fallback به تصویر با PdfiumViewer
|
||||
memoryStream.Position = 0;
|
||||
result = await FallbackToImageExtractionAsync(
|
||||
memoryStream,
|
||||
fileName,
|
||||
mimeType,
|
||||
cancellationToken
|
||||
);
|
||||
result.UsedImageFallback = true;
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Fallback: تبدیل صفحات PDF به تصویر با استفاده از PdfiumViewer ...
|
||||
/// </summary>
|
||||
private async Task<XFileExtractionResult> FallbackToImageExtractionAsync(
|
||||
Stream pdfStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken
|
||||
)
|
||||
{
|
||||
var result = new XFileExtractionResult
|
||||
{
|
||||
FileName = fileName,
|
||||
MimeType = mimeType,
|
||||
UsedImageFallback = true
|
||||
};
|
||||
|
||||
await Task.Run(() =>
|
||||
{
|
||||
try
|
||||
{
|
||||
// باز کردن PDF با PdfiumViewer
|
||||
using var document = PdfDocument.Load(pdfStream);
|
||||
|
||||
// محاسبه تعداد صفحات برای پردازش
|
||||
var pageCount = Math.Min(
|
||||
document.PageCount,
|
||||
MaxPagesForImageFallback
|
||||
);
|
||||
|
||||
// پردازش هر صفحه
|
||||
for (int pageIndex = 0; pageIndex < pageCount; pageIndex++)
|
||||
{
|
||||
// بررسی CancellationToken
|
||||
if (cancellationToken.IsCancellationRequested)
|
||||
{
|
||||
break;
|
||||
}
|
||||
|
||||
try
|
||||
{
|
||||
// دریافت ابعاد اصلی صفحه
|
||||
var pageSize = document.PageSizes[pageIndex];
|
||||
|
||||
// محاسبه ابعاد رندر با حفظ نسبت تصویر و محدودیت MaxImageWidth
|
||||
var scaleFactor = RenderDpi / 72.0;
|
||||
var renderWidth = (int)Math.Ceiling(pageSize.Width * scaleFactor);
|
||||
var renderHeight = (int)Math.Ceiling(pageSize.Height * scaleFactor);
|
||||
|
||||
// محدود کردن عرض به MaxImageWidth
|
||||
if (renderWidth > MaxImageWidth)
|
||||
{
|
||||
var ratio = (double)MaxImageWidth / renderWidth;
|
||||
renderWidth = MaxImageWidth;
|
||||
renderHeight = (int)(renderHeight * ratio);
|
||||
}
|
||||
|
||||
// رندر صفحه به Bitmap
|
||||
using var bitmap = document.Render(
|
||||
pageIndex: pageIndex,
|
||||
width: renderWidth,
|
||||
height: renderHeight,
|
||||
dpiX: RenderDpi,
|
||||
dpiY: RenderDpi,
|
||||
forPrinting: false
|
||||
);
|
||||
|
||||
// تبدیل Bitmap به آرایه بایت PNG
|
||||
byte[] imageBytes;
|
||||
using (var memoryStream = new MemoryStream())
|
||||
{
|
||||
bitmap.Save(memoryStream, System.Drawing.Imaging.ImageFormat.Png);
|
||||
imageBytes = memoryStream.ToArray();
|
||||
}
|
||||
|
||||
// افزودن به نتیجه
|
||||
result.Images.Add(new XExtractedImage
|
||||
{
|
||||
Bytes = imageBytes,
|
||||
PageNumber = pageIndex + 1,
|
||||
MimeType = "image/png",
|
||||
Description = $"Page {pageIndex + 1} of {fileName} ({renderWidth}x{renderHeight})"
|
||||
});
|
||||
}
|
||||
catch (Exception pageEx)
|
||||
{
|
||||
// لاگ خطای صفحه و ادامه با صفحات دیگر
|
||||
System.Diagnostics.Debug.WriteLine(
|
||||
$"Error rendering page {pageIndex + 1} of {fileName}: {pageEx.Message}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
// ثبت خطای کلی
|
||||
result.ErrorMessage = $"PDF image extraction failed: {ex.Message}";
|
||||
System.Diagnostics.Debug.WriteLine(
|
||||
$"PdfiumViewer extraction error for {fileName}: {ex}"
|
||||
);
|
||||
}
|
||||
}, cancellationToken);
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// استخراج متن از PDF با UglyToad.PdfPig ...
|
||||
/// </summary>
|
||||
private async Task<string> ExtractTextAsync(
|
||||
Stream pdfStream,
|
||||
CancellationToken cancellationToken
|
||||
)
|
||||
{
|
||||
return await Task.Run(() =>
|
||||
{
|
||||
var sb = new StringBuilder();
|
||||
using var document = UglyToad.PdfPig.PdfDocument.Open(pdfStream);
|
||||
|
||||
foreach (var page in document.GetPages())
|
||||
{
|
||||
var text = page.Text?.Trim();
|
||||
if (!string.IsNullOrEmpty(text))
|
||||
{
|
||||
sb.AppendLine(text);
|
||||
sb.AppendLine();
|
||||
}
|
||||
}
|
||||
|
||||
return sb.ToString();
|
||||
}, cancellationToken);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// بررسی آیا متن استخراج شده کافی است ...
|
||||
/// </summary>
|
||||
private bool IsTextSufficient(string text)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(text))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
// حذف فاصلهها و بررسی طول
|
||||
var cleanText = new string(
|
||||
text.Where(c => !char.IsWhiteSpace(c)).ToArray()
|
||||
);
|
||||
return cleanText.Length >= MinTextLengthThreshold;
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// سازگاری با نسخه قدیمی Interface ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
var result = await ExtractRichAsync(
|
||||
fileStream,
|
||||
"document.pdf",
|
||||
mimeType,
|
||||
cancellationToken
|
||||
);
|
||||
return result.Text;
|
||||
}
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Step 3: Comparison -->
|
||||
<div class="section">
|
||||
<h2>📊 گام ۳: مقایسه PdfiumViewer با PdfPigRenderer</h2>
|
||||
<table>
|
||||
<tr>
|
||||
<th>ویژگی</th>
|
||||
<th>PdfPigRenderer</th>
|
||||
<th>PdfiumViewer</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>پایداری</strong></td>
|
||||
<td>⚠️ نسبتاً جدید، گاهی ناپایدار</td>
|
||||
<td>✅ بسیار پایدار، سالها استفاده در production</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>سرعت رندر</strong></td>
|
||||
<td>⭐⭐ متوسط</td>
|
||||
<td>⭐⭐⭐⭐ بسیار سریع (native C++)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>کیفیت خروجی</strong></td>
|
||||
<td>⭐⭐⭐ خوب</td>
|
||||
<td>⭐⭐⭐⭐⭐ عالی (همان موتور Chrome)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>پشتیبانی از فرمتهای پیچیده</strong></td>
|
||||
<td>⚠️ محدود</td>
|
||||
<td>✅ کامل (فونتها، transparency، gradients)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>حافظه مصرفی</strong></td>
|
||||
<td>⭐⭐ متوسط</td>
|
||||
<td>⭐⭐⭐⭐ بهینه (native memory)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>نیاز به Native DLL</strong></td>
|
||||
<td>❌ خیر (Pure .NET)</td>
|
||||
<td>✅ بله (pdfium.dll)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>پشتیبانی Cross-platform</strong></td>
|
||||
<td>✅ کامل</td>
|
||||
<td>✅ Windows + Linux (با پکیج Native مناسب)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><strong>کنترل DPI و ابعاد</strong></td>
|
||||
<td>⚠️ محدود</td>
|
||||
<td>✅ کامل و دقیق</td>
|
||||
</tr>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<!-- Step 4: Key Features -->
|
||||
<div class="section">
|
||||
<h2>✨ گام ۴: ویژگیهای کلیدی کد بازنویسی شده</h2>
|
||||
|
||||
<div class="arch-grid">
|
||||
<div class="arch-card">
|
||||
<h4>🎨 کنترل DPI هوشمند</h4>
|
||||
<ul>
|
||||
<li>پارامتر <code>RenderDpi</code> قابل تنظیم</li>
|
||||
<li>تعادل بین کیفیت و حجم خروجی</li>
|
||||
<li>مقدار ۱۵۰ برای Vision Models ایدهآل</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>📏 محدودیت ابعاد</h4>
|
||||
<ul>
|
||||
<li><code>MaxImageWidth = 2000</code></li>
|
||||
<li>جلوگیری از تصاویر بسیار بزرگ</li>
|
||||
<li>حفظ نسبت تصویر</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>🛡️ مدیریت خطای پیشرفته</h4>
|
||||
<ul>
|
||||
<li>خطای هر صفحه به صورت جداگانه</li>
|
||||
<li>ادامه پردازش با صفحات دیگر</li>
|
||||
<li>لاگ دقیق خطاها</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>⚡ بهینهسازی عملکرد</h4>
|
||||
<ul>
|
||||
<li>استفاده از <code>Task.Run</code></li>
|
||||
<li>پشتیبانی از <code>CancellationToken</code></li>
|
||||
<li>مدیریت صحیح <code>using</code></li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Step 5: Important Notes -->
|
||||
<div class="section">
|
||||
<h2>⚠️ گام ۵: نکات مهم پیادهسازی</h2>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکته ۱: Native DLL در زمان اجرا</strong>
|
||||
<p>PdfiumViewer به صورت خودکار <code>pdfium.dll</code> را از پوشههای <code>x86</code> یا <code>x64</code>
|
||||
در مسیر اجرای برنامه بارگذاری میکند. اطمینان حاصل کنید که این فایلها همراه با برنامه deploy میشوند.</p>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-warning">
|
||||
<strong>⚠️ نکته ۲: Linux Deployment</strong>
|
||||
<p>برای استقرار روی Linux، از پکیج <code>PdfiumViewer.Core</code> استفاده کنید و
|
||||
فایل <code>libpdfium.so</code> را در مسیر مناسب قرار دهید:</p>
|
||||
<pre># در csproj فایل:
|
||||
<ItemGroup>
|
||||
<None Include="runtimes/linux-x64/native/libpdfium.so">
|
||||
<CopyToOutputDirectory>PreserveNewest</CopyToOutputDirectory>
|
||||
</None>
|
||||
</ItemGroup></pre>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نکته ۳: تست عملکرد</strong>
|
||||
<p>برای یک PDF اسکنشده ۱۰ صفحهای با DPI=150:</p>
|
||||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||||
<li><strong>زمان پردازش:</strong> حدود ۲-۳ ثانیه</li>
|
||||
<li><strong>حجم هر تصویر:</strong> ۲۰۰-۴۰۰ کیلوبایت PNG</li>
|
||||
<li><strong>مصرف RAM:</strong> حدود ۱۰۰ مگابایت peak</li>
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-danger">
|
||||
<strong>⚠️ نکته ۴: محدودیتهای PdfiumViewer</strong>
|
||||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||||
<li>پشتیبانی ضعیفتر از <code>async/await</code> واقعی (استفاده از <code>Task.Run</code>)</li>
|
||||
<li>نیاز به نصب Native DLL</li>
|
||||
<li>عدم پشتیبانی از PDF های رمزنگاری شده با پسورد</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Step 6: Alternative for .NET 8+ -->
|
||||
<div class="section">
|
||||
<h2>🚀 گام ۶: پیشنهاد برای .NET 8+ (آیندهنگرانه)</h2>
|
||||
<p>
|
||||
اگر پروژه شما در آینده به .NET 8 یا بالاتر مهاجرت کند، میتوانید از کتابخانههای مدرنتر مانند
|
||||
<code>SkiaSharp</code> + <code>HarfBuzzSharp</code> یا <code> PdfPig</code> نسخه جدید استفاده کنید
|
||||
که کاملاً Cross-platform و async-native هستند. اما برای حال حاضر، <strong>PdfiumViewer بهترین انتخاب</strong> است.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Summary -->
|
||||
<div class="section">
|
||||
<h2>📋 خلاصه تغییرات</h2>
|
||||
<table>
|
||||
<tr>
|
||||
<th>مورد</th>
|
||||
<th>توضیح</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>جایگزینی <code>PdfPigRenderer</code></td>
|
||||
<td>با <code>PdfDocument</code> از PdfiumViewer</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>افزودن کنترل DPI</td>
|
||||
<td>پارامتر <code>RenderDpi</code> برای تنظیم کیفیت</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>محدودیت ابعاد</td>
|
||||
<td><code>MaxImageWidth</code> برای جلوگیری از تصاویر بزرگ</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>مدیریت خطای صفحه</td>
|
||||
<td>ادامه پردازش در صورت خطای یک صفحه</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>بهبود لاگ</td>
|
||||
<td>ثبت خطاها با <code>Debug.WriteLine</code></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نتیجهگیری:</strong>
|
||||
<p>با بازنویسی تابع <code>FallbackToImageExtractionAsync</code> با استفاده از <strong>PdfiumViewer</strong>،
|
||||
سیستم شما اکنون قابلیت تبدیل قابل اعتماد و با کیفیت PDF های اسکنشده به تصاویر را دارد.
|
||||
این تصاویر میتوانند مستقیماً به عنوان <code>DataContent</code> به مدلهای Vision مانند
|
||||
<strong>GPT-4V</strong>، <strong>Gemini</strong> یا <strong>Qwen-VL</strong> ارسال شوند. 🎯</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div class="footer">
|
||||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||
<p><strong>📅 تاریخ تهیه مستند:</strong> یکشنبه ۱۳ مهر ۱۴۰۵</p>
|
||||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||
📄 بازنویسی FallbackToImageExtractionAsync با PdfiumViewer - تمامی حقوق محفوظ است
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,737 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fa" dir="rtl">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>بازنویسی PDF Fallback با PdfiumViewer - فن آوران ساحر علم</title>
|
||||
<style>
|
||||
:root {
|
||||
--primary: #1e3a8a;
|
||||
--secondary: #3b82f6;
|
||||
--accent: #f59e0b;
|
||||
--success: #10b981;
|
||||
--danger: #ef4444;
|
||||
--warning: #f97316;
|
||||
--bg-light: #f8fafc;
|
||||
--bg-code: #1e293b;
|
||||
--text-dark: #0f172a;
|
||||
--text-muted: #64748b;
|
||||
--border: #e2e8f0;
|
||||
}
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||
color: var(--text-dark);
|
||||
line-height: 1.8;
|
||||
padding: 20px;
|
||||
}
|
||||
.container {
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
background: white;
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||
overflow: hidden;
|
||||
}
|
||||
.header {
|
||||
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
|
||||
color: white;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
}
|
||||
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
|
||||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||||
.meta-bar {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
background: var(--bg-light);
|
||||
padding: 15px 30px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
flex-wrap: wrap;
|
||||
gap: 15px;
|
||||
}
|
||||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||
.meta-item strong { color: var(--primary); }
|
||||
.content { padding: 40px; }
|
||||
.section {
|
||||
margin-bottom: 35px;
|
||||
padding: 25px;
|
||||
background: var(--bg-light);
|
||||
border-radius: 12px;
|
||||
border-right: 5px solid var(--secondary);
|
||||
}
|
||||
.section h2 {
|
||||
color: var(--primary);
|
||||
font-size: 1.5em;
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
}
|
||||
pre {
|
||||
background: var(--bg-code);
|
||||
color: #e2e8f0;
|
||||
padding: 18px;
|
||||
border-radius: 8px;
|
||||
overflow-x: auto;
|
||||
direction: ltr;
|
||||
text-align: left;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.85em;
|
||||
margin: 15px 0;
|
||||
border-right: 4px solid var(--accent);
|
||||
}
|
||||
code {
|
||||
background: #fef3c7;
|
||||
color: #92400e;
|
||||
padding: 2px 8px;
|
||||
border-radius: 4px;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.9em;
|
||||
direction: ltr;
|
||||
display: inline-block;
|
||||
}
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
margin: 15px 0;
|
||||
background: white;
|
||||
border-radius: 8px;
|
||||
overflow: hidden;
|
||||
}
|
||||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
|
||||
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
|
||||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.arch-grid {
|
||||
display: grid;
|
||||
grid-template-columns: repeat(auto-fit, minmax(280px, 1fr));
|
||||
gap: 20px;
|
||||
margin: 20px 0;
|
||||
}
|
||||
.arch-card {
|
||||
background: white;
|
||||
padding: 20px;
|
||||
border-radius: 10px;
|
||||
box-shadow: 0 4px 12px rgba(0,0,0,0.08);
|
||||
border-top: 4px solid var(--secondary);
|
||||
}
|
||||
.arch-card h4 { color: var(--primary); margin-bottom: 12px; }
|
||||
.arch-card ul { list-style: none; padding-right: 0; }
|
||||
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; }
|
||||
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
|
||||
<div class="header">
|
||||
<h1>📄 بازنویسی PDF Fallback با PdfiumViewer</h1>
|
||||
<div class="subtitle">جایگزینی PdfPigRenderer با PdfiumViewer برای تبدیل PDF به تصویر</div>
|
||||
</div>
|
||||
|
||||
<div class="meta-bar">
|
||||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||
<div class="meta-item">📅 <strong>تاریخ:</strong> یکشنبه ۱۳ مهر ۱۴۰۵</div>
|
||||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||||
</div>
|
||||
|
||||
<div class="content">
|
||||
|
||||
<!-- Section 1: Overview -->
|
||||
<div class="section">
|
||||
<h2>🎯 معرفی PdfiumViewer</h2>
|
||||
<p>
|
||||
<span style="font-weight: bold;">PdfiumViewer</span> یک کتابخانه قدرتمند و سریع برای کار با فایلهای PDF در .NET است
|
||||
که بر پایه موتور <strong>PDFium</strong> گوگل (همان موتوری که Chrome استفاده میکند) ساخته شده است.
|
||||
</p>
|
||||
|
||||
<div class="arch-grid">
|
||||
<div class="arch-card">
|
||||
<h4>⚡ مزایای PdfiumViewer</h4>
|
||||
<ul>
|
||||
<li>سرعت بالا در رندر صفحات</li>
|
||||
<li>پشتیبانی کامل از PDF های پیچیده</li>
|
||||
<li>کیفیت رندر عالی</li>
|
||||
<li>کنترل دقیق DPI و اندازه خروجی</li>
|
||||
<li>پشتیبانی از فرمتهای مختلف تصویر</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>⚠️ ملاحظات مهم</h4>
|
||||
<ul>
|
||||
<li>نیاز به Native DLL های PDFium</li>
|
||||
<li>حجم فایل DLL ها حدود 10-20 مگابایت</li>
|
||||
<li>نیاز به کپی DLL ها در output directory</li>
|
||||
<li>Thread-safe نیست (نیاز به قفل)</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 2: Installation -->
|
||||
<div class="section">
|
||||
<h2>📦 گام ۱: نصب پکیجهای NuGet</h2>
|
||||
|
||||
<pre># نصب پکیج اصلی PdfiumViewer
|
||||
Install-Package PdfiumViewer
|
||||
|
||||
# نصب Native DLL های PDFium (برای پلتفرم x64)
|
||||
Install-Package PdfiumViewer.Native
|
||||
|
||||
# یا برای هر دو پلتفرم x86 و x64
|
||||
Install-Package PdfiumViewer.Native.x86
|
||||
Install-Package PdfiumViewer.Native.x64</pre>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکته:</strong> پکیج <code>PdfiumViewer.Native</code> به صورت خودکار DLL های <code>pdfium.dll</code>
|
||||
را در پوشه <code>x86</code> و <code>x64</code> پروژه کپی میکند و در runtime بر اساس معماری سیستم،
|
||||
DLL مناسب را لود میکند.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 3: Rewrite Code -->
|
||||
<div class="section">
|
||||
<h2>🔧 گام ۲: بازنویسی تابع FallbackToImageExtractionAsync</h2>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Providers/Extractors/XPdfFileContentExtractor.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>using System;
|
||||
using System.IO;
|
||||
using System.Drawing;
|
||||
using System.Drawing.Imaging;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using PdfiumViewer;
|
||||
using xAiApi.Models;
|
||||
|
||||
namespace xAiApi.Providers.Extractors
|
||||
{
|
||||
public class XPdfFileContentExtractor : IXPdfFileContentExtractor
|
||||
{
|
||||
private const int MaxPagesForImageFallback = 10;
|
||||
private const int MinTextLengthThreshold = 50;
|
||||
|
||||
/// <summary>
|
||||
/// DPI پیشفرض برای رندر تصاویر (72 = استاندارد، 150 = کیفیت بالا)
|
||||
/// </summary>
|
||||
private const int DefaultDpi = 150;
|
||||
|
||||
// ... (سایر متدها)
|
||||
|
||||
/// <summary>
|
||||
/// Fallback: تبدیل صفحات PDF به تصویر با استفاده از PdfiumViewer ...
|
||||
/// </summary>
|
||||
private async Task<XFileExtractionResult> FallbackToImageExtractionAsync(
|
||||
Stream pdfStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken
|
||||
)
|
||||
{
|
||||
var result = new XFileExtractionResult
|
||||
{
|
||||
FileName = fileName,
|
||||
MimeType = mimeType,
|
||||
UsedImageFallback = true
|
||||
};
|
||||
|
||||
// کپی Stream به MemoryStream برای PdfiumViewer
|
||||
// (PdfiumViewer به Stream قابل Seek نیاز دارد)
|
||||
using var memoryStream = new MemoryStream();
|
||||
await pdfStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
memoryStream.Position = 0;
|
||||
|
||||
await Task.Run(() =>
|
||||
{
|
||||
try
|
||||
{
|
||||
// بارگذاری PDF با PdfiumViewer
|
||||
using var pdfDocument = PdfDocument.Load(memoryStream);
|
||||
|
||||
// محاسبه تعداد صفحات برای پردازش
|
||||
var pageCount = Math.Min(
|
||||
pdfDocument.PageCount,
|
||||
MaxPagesForImageFallback
|
||||
);
|
||||
|
||||
// حلقه روی صفحات و تبدیل به تصویر
|
||||
for (int pageIndex = 0; pageIndex < pageCount; pageIndex++)
|
||||
{
|
||||
// بررسی CancellationToken
|
||||
cancellationToken.ThrowIfCancellationRequested();
|
||||
|
||||
// رندر صفحه به Bitmap
|
||||
// متد RenderPage پارامترهای زیر را میپذیرد:
|
||||
// - pageIndex: شماره صفحه
|
||||
// - width: عرض تصویر خروجی (0 = اندازه اصلی)
|
||||
// - height: ارتفاع تصویر خروجی (0 = اندازه اصلی)
|
||||
// - dpi: کیفیت رندر
|
||||
using var bitmap = pdfDocument.RenderPage(
|
||||
pageIndex: pageIndex,
|
||||
width: 0, // 0 = استفاده از اندازه اصلی
|
||||
height: 0, // 0 = استفاده از اندازه اصلی
|
||||
dpiX: DefaultDpi,
|
||||
dpiY: DefaultDpi
|
||||
);
|
||||
|
||||
if (bitmap != null)
|
||||
{
|
||||
// تبدیل Bitmap به PNG
|
||||
using var imageStream = new MemoryStream();
|
||||
bitmap.Save(imageStream, ImageFormat.Png);
|
||||
var imageBytes = imageStream.ToArray();
|
||||
|
||||
// افزودن به لیست تصاویر
|
||||
result.Images.Add(new XExtractedImage
|
||||
{
|
||||
Bytes = imageBytes,
|
||||
PageNumber = pageIndex + 1,
|
||||
MimeType = "image/png",
|
||||
Description = $"Page {pageIndex + 1} of {fileName}"
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
catch (OperationCanceledException)
|
||||
{
|
||||
// عملیات لغو شد
|
||||
throw;
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
// خطا در پردازش PDF
|
||||
result.ErrorMessage = $"Failed to render PDF pages to images: {ex.Message}";
|
||||
}
|
||||
}, cancellationToken);
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// ... (سایر متدها)
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 4: Alternative with DPI Control -->
|
||||
<div class="section">
|
||||
<h2>🎨 گام ۳: نسخه پیشرفته با کنترل دقیق DPI</h2>
|
||||
<p>اگر نیاز به کنترل دقیقتر روی کیفیت و اندازه تصاویر دارید، از این نسخه استفاده کنید:</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-new">ADVANCED</span>
|
||||
<span class="path">نسخه پیشرفته با کنترل DPI</span>
|
||||
</div>
|
||||
|
||||
<pre>/// <summary>
|
||||
/// Fallback پیشرفته با کنترل دقیق DPI و کیفیت ...
|
||||
/// </summary>
|
||||
private async Task<XFileExtractionResult> FallbackToImageExtractionAdvancedAsync(
|
||||
Stream pdfStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
int targetDpi = 150,
|
||||
ImageFormat outputFormat = null,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
outputFormat ??= ImageFormat.Png;
|
||||
|
||||
var result = new XFileExtractionResult
|
||||
{
|
||||
FileName = fileName,
|
||||
MimeType = mimeType,
|
||||
UsedImageFallback = true
|
||||
};
|
||||
|
||||
using var memoryStream = new MemoryStream();
|
||||
await pdfStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
memoryStream.Position = 0;
|
||||
|
||||
await Task.Run(() =>
|
||||
{
|
||||
using var pdfDocument = PdfDocument.Load(memoryStream);
|
||||
var pageCount = Math.Min(pdfDocument.PageCount, MaxPagesForImageFallback);
|
||||
|
||||
for (int pageIndex = 0; pageIndex < pageCount; pageIndex++)
|
||||
{
|
||||
cancellationToken.ThrowIfCancellationRequested();
|
||||
|
||||
// دریافت اندازه اصلی صفحه
|
||||
var pageSize = pdfDocument.PageSizes[pageIndex];
|
||||
|
||||
// محاسبه اندازه تصویر بر اساس DPI
|
||||
var width = (int)(pageSize.Width * targetDpi / 72.0);
|
||||
var height = (int)(pageSize.Height * targetDpi / 72.0);
|
||||
|
||||
// رندر با اندازه مشخص
|
||||
using var bitmap = pdfDocument.RenderPage(
|
||||
pageIndex: pageIndex,
|
||||
width: width,
|
||||
height: height,
|
||||
dpiX: targetDpi,
|
||||
dpiY: targetDpi
|
||||
);
|
||||
|
||||
if (bitmap != null)
|
||||
{
|
||||
using var imageStream = new MemoryStream();
|
||||
bitmap.Save(imageStream, outputFormat);
|
||||
var imageBytes = imageStream.ToArray();
|
||||
|
||||
result.Images.Add(new XExtractedImage
|
||||
{
|
||||
Bytes = imageBytes,
|
||||
PageNumber = pageIndex + 1,
|
||||
MimeType = outputFormat == ImageFormat.Jpeg ? "image/jpeg" : "image/png",
|
||||
Description = $"Page {pageIndex + 1} of {fileName} ({width}x{height}px @ {targetDpi}dpi)"
|
||||
});
|
||||
}
|
||||
}
|
||||
}, cancellationToken);
|
||||
|
||||
return result;
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 5: Configuration -->
|
||||
<div class="section">
|
||||
<h2>⚙️ گام ۴: پیکربندی DPI در appsettings.json</h2>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-new">CONFIG</span>
|
||||
<span class="path">appsettings.json</span>
|
||||
</div>
|
||||
|
||||
<pre>{
|
||||
"AiApiConfiguration": {
|
||||
"FileExtraction": {
|
||||
"PdfImageFallback": {
|
||||
"Enabled": true,
|
||||
"MaxPages": 10,
|
||||
"Dpi": 150,
|
||||
"OutputFormat": "png",
|
||||
"MinTextLengthThreshold": 50
|
||||
}
|
||||
}
|
||||
}
|
||||
}</pre>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-new">CLASS</span>
|
||||
<span class="path">xAiApi/Configurations/XPdfImageFallbackConfiguration.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>namespace xAiApi.Configurations
|
||||
{
|
||||
public class XPdfImageFallbackConfiguration
|
||||
{
|
||||
public bool Enabled { get; set; } = true;
|
||||
public int MaxPages { get; set; } = 10;
|
||||
public int Dpi { get; set; } = 150;
|
||||
public string OutputFormat { get; set; } = "png";
|
||||
public int MinTextLengthThreshold { get; set; } = 50;
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 6: Usage with Configuration -->
|
||||
<div class="section">
|
||||
<h2>🔗 گام ۵: استفاده از پیکربندی در Extractor</h2>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">XPdfFileContentExtractor.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>public class XPdfFileContentExtractor : IXPdfFileContentExtractor
|
||||
{
|
||||
private readonly XPdfImageFallbackConfiguration _config;
|
||||
|
||||
public XPdfFileContentExtractor(XPdfImageFallbackConfiguration config)
|
||||
{
|
||||
_config = config ?? new XPdfImageFallbackConfiguration();
|
||||
}
|
||||
|
||||
private async Task<XFileExtractionResult> FallbackToImageExtractionAsync(
|
||||
Stream pdfStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken
|
||||
)
|
||||
{
|
||||
var result = new XFileExtractionResult
|
||||
{
|
||||
FileName = fileName,
|
||||
MimeType = mimeType,
|
||||
UsedImageFallback = true
|
||||
};
|
||||
|
||||
using var memoryStream = new MemoryStream();
|
||||
await pdfStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
memoryStream.Position = 0;
|
||||
|
||||
await Task.Run(() =>
|
||||
{
|
||||
using var pdfDocument = PdfDocument.Load(memoryStream);
|
||||
var pageCount = Math.Min(pdfDocument.PageCount, _config.MaxPages);
|
||||
|
||||
for (int pageIndex = 0; pageIndex < pageCount; pageIndex++)
|
||||
{
|
||||
cancellationToken.ThrowIfCancellationRequested();
|
||||
|
||||
using var bitmap = pdfDocument.RenderPage(
|
||||
pageIndex: pageIndex,
|
||||
width: 0,
|
||||
height: 0,
|
||||
dpiX: _config.Dpi,
|
||||
dpiY: _config.Dpi
|
||||
);
|
||||
|
||||
if (bitmap != null)
|
||||
{
|
||||
using var imageStream = new MemoryStream();
|
||||
|
||||
// انتخاب فرمت خروجی بر اساس پیکربندی
|
||||
var format = _config.OutputFormat?.ToLowerInvariant() switch
|
||||
{
|
||||
"jpeg" or "jpg" => ImageFormat.Jpeg,
|
||||
"bmp" => ImageFormat.Bmp,
|
||||
_ => ImageFormat.Png
|
||||
};
|
||||
|
||||
bitmap.Save(imageStream, format);
|
||||
var imageBytes = imageStream.ToArray();
|
||||
|
||||
result.Images.Add(new XExtractedImage
|
||||
{
|
||||
Bytes = imageBytes,
|
||||
PageNumber = pageIndex + 1,
|
||||
MimeType = format == ImageFormat.Jpeg ? "image/jpeg" : "image/png",
|
||||
Description = $"Page {pageIndex + 1} of {fileName}"
|
||||
});
|
||||
}
|
||||
}
|
||||
}, cancellationToken);
|
||||
|
||||
return result;
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 7: DI Registration -->
|
||||
<div class="section">
|
||||
<h2>🔌 گام ۶: ثبت در Dependency Injection</h2>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Startup.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>public void ConfigureServices(IServiceCollection services)
|
||||
{
|
||||
// ... (سایر ثبتها)
|
||||
|
||||
// ✅ ثبت پیکربندی PDF Image Fallback
|
||||
services.Configure<XPdfImageFallbackConfiguration>(
|
||||
Configuration.GetSection("AiApiConfiguration:FileExtraction:PdfImageFallback")
|
||||
);
|
||||
|
||||
// ✅ ثبت PDF Extractor با پیکربندی
|
||||
services.AddSingleton<IXFileContentExtractor>(sp =>
|
||||
{
|
||||
var config = sp.GetRequiredService<IOptions<XPdfImageFallbackConfiguration>>().Value;
|
||||
return new XPdfFileContentExtractor(config);
|
||||
});
|
||||
|
||||
// ... (بقیه کدها)
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 8: Troubleshooting -->
|
||||
<div class="section">
|
||||
<h2>🐛 گام ۷: عیبیابی و نکات مهم</h2>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<th>مشکل</th>
|
||||
<th>علت</th>
|
||||
<th>راهحل</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>DllNotFoundException</code></td>
|
||||
<td>pdfium.dll یافت نشد</td>
|
||||
<td>نصب <code>PdfiumViewer.Native</code> و rebuild پروژه</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>BadImageFormatException</code></td>
|
||||
<td>عدم تطابق معماری (x86/x64)</td>
|
||||
<td>استفاده از پکیج Native مناسب یا تنظیم Platform Target</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>خطا در <code>PdfDocument.Load</code></td>
|
||||
<td>Stream غیرقابل Seek</td>
|
||||
<td>کپی به <code>MemoryStream</code> قبل از Load</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>کیفیت پایین تصاویر</td>
|
||||
<td>DPI خیلی کم</td>
|
||||
<td>افزایش DPI به 200 یا 300</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>OutOfMemoryException</code></td>
|
||||
<td>تعداد صفحات زیاد یا DPI بالا</td>
|
||||
<td>کاهش <code>MaxPages</code> یا <code>Dpi</code></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-warning">
|
||||
<strong>⚠️ Thread Safety:</strong>
|
||||
<p>
|
||||
PdfiumViewer <strong>Thread-safe نیست</strong>. اگر چندین درخواست همزمان PDF را پردازش میکنند،
|
||||
باید از <code>lock</code> یا <code>SemaphoreSlim</code> استفاده کنید:
|
||||
</p>
|
||||
<pre>private static readonly SemaphoreSlim _pdfLock = new SemaphoreSlim(2, 2); // حداکثر 2 همزمان
|
||||
|
||||
await _pdfLock.WaitAsync(cancellationToken);
|
||||
try
|
||||
{
|
||||
// پردازش PDF
|
||||
}
|
||||
finally
|
||||
{
|
||||
_pdfLock.Release();
|
||||
}</pre>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 9: Comparison -->
|
||||
<div class="section">
|
||||
<h2>📊 گام ۸: مقایسه PdfPigRenderer و PdfiumViewer</h2>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<th>ویژگی</th>
|
||||
<th>PdfPigRenderer</th>
|
||||
<th>PdfiumViewer</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>سرعت رندر</td>
|
||||
<td>⭐⭐⭐ متوسط</td>
|
||||
<td>⭐⭐⭐⭐⭐ بسیار سریع</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>کیفیت خروجی</td>
|
||||
<td>⭐⭐⭐ خوب</td>
|
||||
<td>⭐⭐⭐⭐⭐ عالی</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>حجم DLL</td>
|
||||
<td>کوچک (فقط .NET)</td>
|
||||
<td>بزرگ (10-20 MB Native)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>پشتیبانی از PDF های پیچیده</td>
|
||||
<td>⭐⭐⭐ محدود</td>
|
||||
<td>⭐⭐⭐⭐⭐ کامل</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>کنترل DPI</td>
|
||||
<td>⭐⭐ محدود</td>
|
||||
<td>⭐⭐⭐⭐⭐ دقیق</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>Thread Safety</td>
|
||||
<td>✅ Thread-safe</td>
|
||||
<td>❌ نیاز به قفل</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>نیاز به Native DLL</td>
|
||||
<td>❌ خیر</td>
|
||||
<td>✅ بله</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نتیجهگیری:</strong>
|
||||
<p>
|
||||
<strong>PdfiumViewer</strong> انتخاب بهتری برای سناریوهای Production است، به ویژه اگر:
|
||||
</p>
|
||||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||||
<li>کیفیت رندر برایتان مهم است</li>
|
||||
<li>با PDF های پیچیده (فونتهای خاص، تصاویر، فرمها) کار میکنید</li>
|
||||
<li>سرعت پردازش اولویت دارد</li>
|
||||
<li>حجم DLL ها مسئلهای نیست</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 10: Summary -->
|
||||
<div class="section">
|
||||
<h2>📋 گام ۹: خلاصه تغییرات</h2>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<th>فایل</th>
|
||||
<th>تغییر</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>XPdfFileContentExtractor.cs</code></td>
|
||||
<td>بازنویسی <code>FallbackToImageExtractionAsync</code> با PdfiumViewer</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>XPdfImageFallbackConfiguration.cs</code></td>
|
||||
<td>کلاس جدید برای پیکربندی</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>appsettings.json</code></td>
|
||||
<td>افزودن بخش <code>PdfImageFallback</code></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>Startup.cs</code></td>
|
||||
<td>ثبت پیکربندی و Extractor در DI</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>NuGet Packages</td>
|
||||
<td>نصب <code>PdfiumViewer</code> و <code>PdfiumViewer.Native</code></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکات کلیدی:</strong>
|
||||
<ul style="padding-right: 25px; margin-top: 10px;">
|
||||
<li>✅ PdfiumViewer سریعتر و باکیفیتتر از PdfPigRenderer است</li>
|
||||
<li>✅ کنترل دقیق DPI و اندازه خروجی</li>
|
||||
<li>⚠️ نیاز به Native DLL ها (حدود 10-20 MB)</li>
|
||||
<li>⚠️ Thread-safe نیست و نیاز به قفل دارد</li>
|
||||
<li>✅ پیکربندیپذیر از طریق appsettings.json</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div class="footer">
|
||||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||
<p><strong>📅 تاریخ:</strong> یکشنبه ۱۳ مهر ۱۴۰۵</p>
|
||||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||
📄 بازنویسی PDF Fallback با PdfiumViewer - تمامی حقوق محفوظ است
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,638 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fa" dir="rtl">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>استخراج محتوای جهانی با Qwen3.5-9B Vision - فن آوران ساحر علم</title>
|
||||
<style>
|
||||
:root {
|
||||
--primary: #1e3a8a;
|
||||
--secondary: #3b82f6;
|
||||
--accent: #f59e0b;
|
||||
--success: #10b981;
|
||||
--danger: #ef4444;
|
||||
--warning: #f97316;
|
||||
--bg-light: #f8fafc;
|
||||
--bg-code: #1e293b;
|
||||
--text-dark: #0f172a;
|
||||
--text-muted: #64748b;
|
||||
--border: #e2e8f0;
|
||||
}
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||
color: var(--text-dark);
|
||||
line-height: 1.8;
|
||||
padding: 20px;
|
||||
}
|
||||
.container {
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
background: white;
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||
overflow: hidden;
|
||||
}
|
||||
.header {
|
||||
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
|
||||
color: white;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
}
|
||||
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
|
||||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||||
.meta-bar {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
background: var(--bg-light);
|
||||
padding: 15px 30px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
flex-wrap: wrap;
|
||||
gap: 15px;
|
||||
}
|
||||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||
.meta-item strong { color: var(--primary); }
|
||||
.content { padding: 40px; }
|
||||
.section {
|
||||
margin-bottom: 35px;
|
||||
padding: 25px;
|
||||
background: var(--bg-light);
|
||||
border-radius: 12px;
|
||||
border-right: 5px solid var(--secondary);
|
||||
}
|
||||
.section h2 {
|
||||
color: var(--primary);
|
||||
font-size: 1.5em;
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
}
|
||||
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
|
||||
pre {
|
||||
background: var(--bg-code);
|
||||
color: #e2e8f0;
|
||||
padding: 18px;
|
||||
border-radius: 8px;
|
||||
overflow-x: auto;
|
||||
direction: ltr;
|
||||
text-align: left;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.83em;
|
||||
margin: 15px 0;
|
||||
border-right: 4px solid var(--accent);
|
||||
}
|
||||
code {
|
||||
background: #fef3c7;
|
||||
color: #92400e;
|
||||
padding: 2px 8px;
|
||||
border-radius: 4px;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.9em;
|
||||
direction: ltr;
|
||||
display: inline-block;
|
||||
}
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
margin: 15px 0;
|
||||
background: white;
|
||||
border-radius: 8px;
|
||||
overflow: hidden;
|
||||
}
|
||||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
|
||||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||||
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
|
||||
.toc h3 { color: var(--primary); margin-bottom: 15px; }
|
||||
.toc ol { padding-right: 25px; }
|
||||
.toc li { padding: 6px 0; }
|
||||
.toc a { color: var(--secondary); text-decoration: none; }
|
||||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.arch-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(280px, 1fr)); gap: 20px; margin: 20px 0; }
|
||||
.arch-card { background: white; padding: 20px; border-radius: 10px; box-shadow: 0 4px 12px rgba(0,0,0,0.08); border-top: 4px solid var(--secondary); }
|
||||
.arch-card h4 { color: var(--primary); margin-bottom: 12px; }
|
||||
.arch-card ul { list-style: none; padding-right: 0; }
|
||||
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; }
|
||||
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
|
||||
<div class="header">
|
||||
<h1>👁️ استخراج محتوای جهانی با Qwen3.5-9B Vision</h1>
|
||||
<div class="subtitle">یکپارچهسازی قابلیت OCR هوشمند در معماری XFileContentExtractor</div>
|
||||
</div>
|
||||
|
||||
<div class="meta-bar">
|
||||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
|
||||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||||
</div>
|
||||
|
||||
<div class="content">
|
||||
|
||||
<div class="toc">
|
||||
<h3>📑 فهرست مطالب</h3>
|
||||
<ol>
|
||||
<li><a href="#capabilities">تحلیل قابلیتهای مدل Qwen3.5-9B GGUF</a></li>
|
||||
<li><a href="#strategy">استراتژی استخراج جهانی (Universal Extraction)</a></li>
|
||||
<li><a href="#step1">گام ۱: پیکربندی مدل در Ollama</a></li>
|
||||
<li><a href="#step2">گام ۲: ایجاد XQwenVisionContentExtractor</a></li>
|
||||
<li><a href="#step3">گام ۳: بهروزرسانی XPdfFileContentExtractor</a></li>
|
||||
<li><a href="#step4">گام ۴: ثبت سرویسها در DI</a></li>
|
||||
<li><a href="#step5">گام ۵: مهندسی پرامپت برای OCR دقیق</a></li>
|
||||
<li><a href="#step6">گام ۶: ملاحظات عملکردی و Production</a></li>
|
||||
</ol>
|
||||
</div>
|
||||
|
||||
<!-- Section 1: Capabilities -->
|
||||
<div class="section" id="capabilities">
|
||||
<h2>🧠 گام ۱: تحلیل قابلیتهای مدل Qwen3.5-9B GGUF</h2>
|
||||
<p>مدل <code>Qwen3.5-9B-The-Defiant-Fable-Uncensored-Heretic-NEO-IMATRIX-MAX-MTP-GGUF</code> یک مدل زبانی بزرگ کوانتیزه شده (GGUF) با ویژگیهای منحصر به فرد است:</p>
|
||||
|
||||
<div class="arch-grid">
|
||||
<div class="arch-card">
|
||||
<h4>👁️ قابلیت Vision / OCR</h4>
|
||||
<ul>
|
||||
<li>توانایی خواندن متن از تصاویر (PNG, JPG, WEBP)</li>
|
||||
<li>تشخیص جداول و ساختارهای بصری در اسناد اسکنشده</li>
|
||||
<li>حذف نیاز به کتابخانههای سنتی OCR مانند Tesseract</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>🚫 Uncensored & Heretic</h4>
|
||||
<ul>
|
||||
<li>بدون فیلترهای اخلاقی سختگیرانه</li>
|
||||
<li>استخراج دقیق محتوا بدون "مoralizing" یا رد درخواست برای اسناد حساس</li>
|
||||
<li>ایدهآل برای پردازش اسناد حقوقی، پزشکی یا فنی خام</li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>⚡ IMATRIX & MTP</h4>
|
||||
<ul>
|
||||
<li>بهینهسازی شده برای سرعت استنتاج (Inference) بالاتر</li>
|
||||
<li>پشتیبانی از Context Window گسترده (معمولاً 8K تا 32K توکن)</li>
|
||||
<li>مناسب برای پردازش اسناد چند صفحهای</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 2: Strategy -->
|
||||
<div class="section" id="strategy">
|
||||
<h2>🎯 گام ۲: استراتژی استخراج جهانی (Universal Extraction)</h2>
|
||||
<p>به جای استفاده از Extractor های جداگانه و پیچیده برای هر فرمت، یک <strong>مسیر هوشمند ترکیبی</strong> طراحی میکنیم:</p>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<th>نوع فایل</th>
|
||||
<th>استراتژی استخراج</th>
|
||||
<th>مسیر پردازش</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>📄 PDF متنی (Native)</td>
|
||||
<td>استخراج متن سریع با PdfPig</td>
|
||||
<td>XPdfFileContentExtractor (متن) → بازگشت سریع</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>📄 PDF اسکنشده / تصویری</td>
|
||||
<td>تبدیل صفحات به تصویر + OCR با Qwen</td>
|
||||
<td>XPdfFileContentExtractor (تصویر) → XQwenVisionContentExtractor</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>🖼️ تصاویر (JPG, PNG)</td>
|
||||
<td>OCR مستقیم با Qwen</td>
|
||||
<td>XQwenVisionContentExtractor</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>📝 DOCX / TXT / Excel</td>
|
||||
<td>استخراج متن ساختاریافته</td>
|
||||
<td>XDocxFileContentExtractor / XExcelFileContentExtractor (بدون تغییر)</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ مزیت کلیدی:</strong> با این روش، ماژول <code>XFileContentExtractor</code> به یک "مغز مرکزی" تبدیل میشود که برای فرمتهای پیچیده یا اسکنشده، به طور خودکار از قدرت Vision مدل Qwen استفاده میکند.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Step 1: Ollama Config -->
|
||||
<div class="section" id="step1">
|
||||
<h2>⚙️ گام ۳: پیکربندی مدل در Ollama</h2>
|
||||
<p>برای فعالسازی قابلیت Vision، مدل باید با یک <code>Modelfile</code> مناسب در Ollama بارگذاری شود.</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-new">CONFIG</span>
|
||||
<span class="path">Modelfile (برای ایمپورت در Ollama)</span>
|
||||
</div>
|
||||
|
||||
<pre>FROM C:/Models/Qwen3.5-9B-The-Defiant-Fable-Uncensored-Heretic-NEO-IMATRIX-MAX-MTP-GGUF.gguf
|
||||
|
||||
# تنظیمات بهینه برای استخراج متن از تصویر
|
||||
PARAMETER temperature 0.1
|
||||
PARAMETER top_p 0.9
|
||||
PARAMETER num_ctx 8192
|
||||
|
||||
# پرامپت سیستمی پیشفرض برای وظایف OCR و استخراج
|
||||
SYSTEM """
|
||||
تو یک موتور OCR و استخراج داده فوقالعاده دقیق هستی.
|
||||
وظیفه تو خواندن تمام متنهای موجود در تصویر یا سند ارائه شده و بازگرداندن آنها به صورت متن خالص و ساختاریافته است.
|
||||
- تمام جداول را به فرمت Markdown تبدیل کن.
|
||||
- اعداد، تاریخها و نامها را دقیقاً همانطور که هستند استخراج کن.
|
||||
- هیچ توضیح اضافی، مقدمه یا نتیجهگیری از خودت اضافه نکن. فقط محتوای استخراج شده را برگردان.
|
||||
"""</pre>
|
||||
|
||||
<p>سپس در ترمینال اجرا کنید:</p>
|
||||
<pre>ollama create qwen3.5-ocr:9b -f Modelfile</pre>
|
||||
</div>
|
||||
|
||||
<!-- Step 2: Vision Extractor -->
|
||||
<div class="section" id="step2">
|
||||
<h2>👁️ گام ۴: ایجاد XQwenVisionContentExtractor</h2>
|
||||
<p>این کلاس قلب تپنده استخراج جهانی است. فایلهای تصویری (یا صفحات PDF تبدیلشده به تصویر) را دریافت کرده و از طریق Ollama API به مدل ارسال میکند.</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-new">NEW</span>
|
||||
<span class="path">xAiApi/Providers/Extractors/XQwenVisionContentExtractor.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Text;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using Microsoft.Extensions.AI;
|
||||
using OllamaSharp;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
using xAiModels.Models;
|
||||
|
||||
namespace xAiApi.Providers.Extractors
|
||||
{
|
||||
/// <summary>
|
||||
/// استخراج محتوا از فایلهای تصویری با استفاده از قابلیت Vision مدل Qwen ...
|
||||
/// </summary>
|
||||
public class XQwenVisionContentExtractor : IXFileContentExtractor
|
||||
{
|
||||
private static readonly string[] SupportedMimeTypes =
|
||||
[
|
||||
"image/png", "image/jpeg", "image/jpg", "image/webp", "image/bmp",
|
||||
"application/pdf" // برای هندل کردن PDF های اسکن شده در سطح Vision
|
||||
];
|
||||
|
||||
private readonly string _ollamaUrl;
|
||||
private readonly string _modelName;
|
||||
private readonly string _extractionPrompt;
|
||||
|
||||
public XQwenVisionContentExtractor(
|
||||
string ollamaUrl = "http://localhost:11434",
|
||||
string modelName = "qwen3.5-ocr:9b",
|
||||
string extractionPrompt = "تمام متن موجود در این تصویر را با دقت بالا استخراج کن و به صورت متن خالص برگردان.")
|
||||
{
|
||||
_ollamaUrl = ollamaUrl;
|
||||
_modelName = modelName;
|
||||
_extractionPrompt = extractionPrompt;
|
||||
}
|
||||
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
return SupportedMimeTypes.Contains(mimeType?.ToLowerInvariant() ?? string.Empty);
|
||||
}
|
||||
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default)
|
||||
{
|
||||
var result = await ExtractRichAsync(fileStream, "file", mimeType, cancellationToken);
|
||||
return result.Text;
|
||||
}
|
||||
|
||||
public async Task<XFileExtractionResult> ExtractRichAsync(
|
||||
Stream fileStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default)
|
||||
{
|
||||
var result = new XFileExtractionResult
|
||||
{
|
||||
FileName = fileName,
|
||||
MimeType = mimeType
|
||||
};
|
||||
|
||||
try
|
||||
{
|
||||
// ۱. خواندن استریم به صورت بایت (برای ارسال به مدل Vision)
|
||||
using var memoryStream = new MemoryStream();
|
||||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
var imageBytes = memoryStream.ToArray();
|
||||
|
||||
// ۲. تنظیم کلاینت Ollama
|
||||
var ollamaClient = new OllamaApiClient(new Uri(_ollamaUrl), _modelName);
|
||||
|
||||
// ۳. ساخت پیام چندوجهی (Text + Image)
|
||||
var messages = new[]
|
||||
{
|
||||
new ChatMessage(ChatRole.User, new[]
|
||||
{
|
||||
new TextContent(_extractionPrompt),
|
||||
new DataContent(imageBytes, mimeType) // ارسال تصویر به مدل
|
||||
})
|
||||
};
|
||||
|
||||
// ۴. فراخوانی مدل برای استخراج متن
|
||||
var response = await ollamaClient.GetResponseAsync(messages, cancellationToken: cancellationToken);
|
||||
|
||||
if (!string.IsNullOrWhiteSpace(response.Text))
|
||||
{
|
||||
result.Text = response.Text.Trim();
|
||||
result.IsValid = true;
|
||||
}
|
||||
else
|
||||
{
|
||||
result.ErrorMessage = "مدل پاسخی برای استخراج تولید نکرد.";
|
||||
}
|
||||
}
|
||||
catch (Exception ex)
|
||||
{
|
||||
result.ErrorMessage = $"خطا در استخراج Vision: {ex.Message}";
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
}
|
||||
}</pre>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکته فنی:</strong> کلاس <code>DataContent</code> از <code>Microsoft.Extensions.AI</code> به طور خودکار بایتهای تصویر را به فرمت Base64 مناسب برای API مدلهای Vision تبدیل میکند.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Step 3: PDF Update -->
|
||||
<div class="section" id="step3">
|
||||
<h2>📄 گام ۵: بهروزرسانی XPdfFileContentExtractor</h2>
|
||||
<p>اکنون منطق Fallback را تغییر میدهیم. به جای اینکه فقط تصویر را ذخیره کنیم، آن تصویر را به <code>XQwenVisionContentExtractor</code> میفرستیم تا متن را استخراج کند.</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Providers/Extractors/XPdfFileContentExtractor.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>using System;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Text;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using UglyToad.PdfPig;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
using xAiModels.Models;
|
||||
|
||||
namespace xAiApi.Providers.Extractors
|
||||
{
|
||||
public class XPdfFileContentExtractor : IXFileContentExtractor
|
||||
{
|
||||
private const int MinTextLengthThreshold = 100;
|
||||
private const int MaxPagesForImageFallback = 5; // محدود کردن برای جلوگیری از مصرف بیش از حد GPU
|
||||
|
||||
private readonly IXFileContentExtractor _visionExtractor;
|
||||
|
||||
// تزریق Vision Extractor از طریق Constructor
|
||||
public XPdfFileContentExtractor(IXFileContentExtractor visionExtractor)
|
||||
{
|
||||
_visionExtractor = visionExtractor;
|
||||
}
|
||||
|
||||
public bool CanExtract(string mimeType) => mimeType?.ToLowerInvariant() == "application/pdf";
|
||||
|
||||
public async Task<string> ExtractAsync(Stream fileStream, string mimeType, CancellationToken cancellationToken = default)
|
||||
{
|
||||
var result = await ExtractRichAsync(fileStream, "document.pdf", mimeType, cancellationToken);
|
||||
return result.Text;
|
||||
}
|
||||
|
||||
public async Task<XFileExtractionResult> ExtractRichAsync(
|
||||
Stream fileStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default)
|
||||
{
|
||||
var result = new XFileExtractionResult { FileName = fileName, MimeType = mimeType };
|
||||
|
||||
using var memoryStream = new MemoryStream();
|
||||
await fileStream.CopyToAsync(memoryStream, cancellationToken);
|
||||
|
||||
// مرحله ۱: تلاش برای استخراج متن معمولی
|
||||
memoryStream.Position = 0;
|
||||
var extractedText = await ExtractTextAsync(memoryStream, cancellationToken);
|
||||
|
||||
if (IsTextSufficient(extractedText))
|
||||
{
|
||||
result.Text = extractedText;
|
||||
result.IsValid = true;
|
||||
return result;
|
||||
}
|
||||
|
||||
// مرحله ۲: Fallback به Vision OCR
|
||||
memoryStream.Position = 0;
|
||||
result = await FallbackToVisionOcrAsync(memoryStream, fileName, cancellationToken);
|
||||
result.UsedImageFallback = true;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
private async Task<string> ExtractTextAsync(Stream pdfStream, CancellationToken cancellationToken)
|
||||
{
|
||||
return await Task.Run(() =>
|
||||
{
|
||||
var sb = new StringBuilder();
|
||||
using var document = PdfDocument.Open(pdfStream);
|
||||
foreach (var page in document.GetPages())
|
||||
{
|
||||
sb.AppendLine(page.Text?.Trim());
|
||||
}
|
||||
return sb.ToString();
|
||||
}, cancellationToken);
|
||||
}
|
||||
|
||||
private bool IsTextSufficient(string text)
|
||||
{
|
||||
if (string.IsNullOrWhiteSpace(text)) return false;
|
||||
var cleanText = new string(text.Where(c => !char.IsWhiteSpace(c)).ToArray());
|
||||
return cleanText.Length >= MinTextLengthThreshold;
|
||||
}
|
||||
|
||||
private async Task<XFileExtractionResult> FallbackToVisionOcrAsync(
|
||||
Stream pdfStream,
|
||||
string fileName,
|
||||
CancellationToken cancellationToken)
|
||||
{
|
||||
var result = new XFileExtractionResult { FileName = fileName, MimeType = "application/pdf", UsedImageFallback = true };
|
||||
var sb = new StringBuilder();
|
||||
|
||||
await Task.Run(() =>
|
||||
{
|
||||
using var document = PdfDocument.Open(pdfStream);
|
||||
var pageCount = Math.Min(document.NumberOfPages, MaxPagesForImageFallback);
|
||||
|
||||
for (int i = 0; i < pageCount; i++)
|
||||
{
|
||||
// تبدیل صفحه به تصویر (با استفاده از PdfPig یا Pdfium)
|
||||
// نکته: برای سادگی، فرض میکنیم متدی داریم که صفحه را به Stream تصویر تبدیل میکند
|
||||
// در پیادهسازی واقعی از PdfiumViewer.RenderPage استفاده کنید (همانند کد قبلی)
|
||||
using var pageImageStream = RenderPageToImageStream(document, i);
|
||||
|
||||
// ارسال تصویر به Qwen Vision Extractor
|
||||
var ocrResult = _visionExtractor.ExtractRichAsync(
|
||||
pageImageStream,
|
||||
$"{fileName}_page_{i+1}",
|
||||
"image/png",
|
||||
cancellationToken).GetAwaiter().GetResult();
|
||||
|
||||
if (ocrResult.IsValid)
|
||||
{
|
||||
sb.AppendLine($"--- Page {i + 1} ---");
|
||||
sb.AppendLine(ocrResult.Text);
|
||||
}
|
||||
}
|
||||
}, cancellationToken);
|
||||
|
||||
result.Text = sb.ToString();
|
||||
result.IsValid = !string.IsNullOrWhiteSpace(result.Text);
|
||||
return result;
|
||||
}
|
||||
|
||||
// Placeholder: باید با منطق PdfiumViewer که قبلاً نوشتید جایگزین شود
|
||||
private Stream RenderPageToImageStream(dynamic document, int pageIndex)
|
||||
{
|
||||
// پیادهسازی RenderPage به MemoryStream از PdfiumViewer
|
||||
throw new NotImplementedException("از منطق PdfiumViewer.RenderPage استفاده کنید");
|
||||
}
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Step 4: DI Registration -->
|
||||
<div class="section" id="step4">
|
||||
<h2>🔌 گام ۶: ثبت سرویسها در Dependency Injection</h2>
|
||||
<p>اکنون باید زنجیره Extractor ها را در <code>Startup.cs</code> به درستی متصل کنیم.</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Startup.cs (متد ConfigureServices)</span>
|
||||
</div>
|
||||
|
||||
<pre>public void ConfigureServices(IServiceCollection services)
|
||||
{
|
||||
// ... (ثبتهای قبلی)
|
||||
|
||||
// ۱. ثبت Vision Extractor به صورت Singleton (چون State-less است)
|
||||
services.AddSingleton<IXFileContentExtractor>(sp =>
|
||||
new XQwenVisionContentExtractor(
|
||||
ollamaUrl: "http://localhost:11434",
|
||||
modelName: "qwen3.5-ocr:9b"
|
||||
));
|
||||
|
||||
// ۲. ثبت PDF Extractor و تزریق Vision Extractor به آن
|
||||
services.AddSingleton<IXFileContentExtractor>(sp =>
|
||||
{
|
||||
var visionExtractor = sp.GetRequiredService<IXFileContentExtractor>();
|
||||
// نکته: برای جلوگیری از تداخل در Resolve، بهتر است Vision Extractor را با نام/interface خاص ثبت کنید
|
||||
// یا مستقیماً اینstantiate کنید:
|
||||
var specificVisionExtractor = new XQwenVisionContentExtractor();
|
||||
return new XPdfFileContentExtractor(specificVisionExtractor);
|
||||
});
|
||||
|
||||
// ۳. ثبت سایر Extractor ها
|
||||
services.AddSingleton<IXFileContentExtractor, XPlainTextFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XDocxFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XExcelFileContentExtractor>();
|
||||
|
||||
// ۴. ثبت Composite Extractor (این کلاس به طور خودکار همه IXFileContentExtractor های ثبت شده را جمعآوری میکند)
|
||||
services.AddSingleton<XFileContentExtractor>();
|
||||
|
||||
// اطمینان از اینکه سرویس اصلی از نوع Composite است
|
||||
services.AddSingleton<IXFileContentExtractor>(sp => sp.GetRequiredService<XFileContentExtractor>());
|
||||
|
||||
// ... (بقیه کدها)
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Step 5: Prompt Engineering -->
|
||||
<div class="section" id="step5">
|
||||
<h2>📝 گام ۷: مهندسی پرامپت برای OCR دقیق</h2>
|
||||
<p>کیفیت استخراج مستقیماً به پرامپت ارسال شده به مدل Qwen بستگی دارد. این پرامپتها را در <code>appsettings.json</code> یا کد تعریف کنید:</p>
|
||||
|
||||
<div class="arch-grid">
|
||||
<div class="arch-card">
|
||||
<h4>🎯 پرامپت استخراج عمومی (General OCR)</h4>
|
||||
<p style="font-size: 0.9em; color: var(--text-muted);">
|
||||
"تمام متن موجود در این تصویر را با دقت کاراکتر به کاراکتر استخراج کن. ساختار پاراگرافها را حفظ کن. اگر جدولی وجود دارد، آن را به فرمت Markdown تبدیل کن. هیچ توضیح اضافی نده."
|
||||
</p>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>📊 پرامپت استخراج داده ساختاریافته (Structured)</h4>
|
||||
<p style="font-size: 0.9em; color: var(--text-muted);">
|
||||
"این تصویر یک فاکتور/سند است. فقط موارد زیر را استخراج و به صورت JSON برگردان: {\"issuer\": \"\", \"date\": \"\", \"total_amount\": \"\", \"items\": []}. اگر موردی یافت نشد، null بگذار."
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Step 6: Production Considerations -->
|
||||
<div class="section" id="step6">
|
||||
<h2>⚠️ گام ۸: ملاحظات عملکردی و Production</h2>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<th>چالش</th>
|
||||
<th>راهحل پیشنهادی در معماری</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>محدودیت Context Window</td>
|
||||
<td>در <code>XPdfFileContentExtractor</code>، <code>MaxPagesForImageFallback</code> را روی ۵ تا ۱۰ تنظیم کنید تا مدل Overload نشود.</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>زمان پاسخدهی (Latency)</td>
|
||||
<td>پردازش Vision زمانبر است. در <code>AskAsync</code> از Timeout مناسب (مثلاً ۶۰ ثانیه) در HttpClient استفاده کنید.</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>مصرف حافظه GPU</td>
|
||||
<td>مدل 9B حدود ۶-۸ گیگابایت VRAM نیاز دارد. اطمینان حاصل کنید سرور Ollama منابع کافی دارد. از پردازش همزمان بیش از ۲ فایل بزرگ خودداری کنید.</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>کیفیت OCR زبان فارسی</td>
|
||||
<td>مدلهای Qwen در فارسی خوب عمل میکنند، اما برای اسناد بسیار قدیمی یا با کیفیت پایین، ممکن است نیاز به پیشپردازش تصویر (افزایش کنتراست) باشد.</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-warning">
|
||||
<strong>⚠️ نکته حیاتی درباره XFileContentExtractor فعلی:</strong><br>
|
||||
کد فعلی <code>XFileContentExtractor</code> از Reflection برای یافتن Extractor ها استفاده میکند. برای اطمینان از عملکرد صحیح، اطمینان حاصل کنید که <code>XQwenVisionContentExtractor</code> و <code>XPdfFileContentExtractor</code> در همان Assembly (پروژه xAiApi) کامپایل شدهاند تا توسط Reflection شناسایی شوند.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div class="footer">
|
||||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
|
||||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||
👁️ مستند فنی استخراج جهانی با Qwen3.5-9B Vision - تمامی حقوق محفوظ است
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,385 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="fa" dir="rtl">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>بازطراحی XFileContentExtractor با استفاده از تزریق وابستگی - فن آوران ساحر علم</title>
|
||||
<style>
|
||||
:root {
|
||||
--primary: #1e3a8a;
|
||||
--secondary: #3b82f6;
|
||||
--accent: #f59e0b;
|
||||
--success: #10b981;
|
||||
--danger: #ef4444;
|
||||
--warning: #f97316;
|
||||
--bg-light: #f8fafc;
|
||||
--bg-code: #1e293b;
|
||||
--text-dark: #0f172a;
|
||||
--text-muted: #64748b;
|
||||
--border: #e2e8f0;
|
||||
}
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: 'Tahoma', 'Segoe UI', sans-serif;
|
||||
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
|
||||
color: var(--text-dark);
|
||||
line-height: 1.8;
|
||||
padding: 20px;
|
||||
}
|
||||
.container {
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
background: white;
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
|
||||
overflow: hidden;
|
||||
}
|
||||
.header {
|
||||
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
|
||||
color: white;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
}
|
||||
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
|
||||
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
|
||||
.meta-bar {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
background: var(--bg-light);
|
||||
padding: 15px 30px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
flex-wrap: wrap;
|
||||
gap: 15px;
|
||||
}
|
||||
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
|
||||
.meta-item strong { color: var(--primary); }
|
||||
.content { padding: 40px; }
|
||||
.section {
|
||||
margin-bottom: 35px;
|
||||
padding: 25px;
|
||||
background: var(--bg-light);
|
||||
border-radius: 12px;
|
||||
border-right: 5px solid var(--secondary);
|
||||
}
|
||||
.section h2 {
|
||||
color: var(--primary);
|
||||
font-size: 1.5em;
|
||||
margin-bottom: 20px;
|
||||
padding-bottom: 10px;
|
||||
border-bottom: 2px solid var(--border);
|
||||
}
|
||||
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
|
||||
pre {
|
||||
background: var(--bg-code);
|
||||
color: #e2e8f0;
|
||||
padding: 18px;
|
||||
border-radius: 8px;
|
||||
overflow-x: auto;
|
||||
direction: ltr;
|
||||
text-align: left;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.85em;
|
||||
margin: 15px 0;
|
||||
border-right: 4px solid var(--accent);
|
||||
}
|
||||
code {
|
||||
background: #fef3c7;
|
||||
color: #92400e;
|
||||
padding: 2px 8px;
|
||||
border-radius: 4px;
|
||||
font-family: 'Consolas', monospace;
|
||||
font-size: 0.9em;
|
||||
direction: ltr;
|
||||
display: inline-block;
|
||||
}
|
||||
table {
|
||||
width: 100%;
|
||||
border-collapse: collapse;
|
||||
margin: 15px 0;
|
||||
background: white;
|
||||
border-radius: 8px;
|
||||
overflow: hidden;
|
||||
}
|
||||
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
|
||||
td { padding: 12px; border-bottom: 1px solid var(--border); }
|
||||
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
|
||||
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
|
||||
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
|
||||
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
|
||||
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
|
||||
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
|
||||
.toc h3 { color: var(--primary); margin-bottom: 15px; }
|
||||
.toc ol { padding-right: 25px; }
|
||||
.toc li { padding: 6px 0; }
|
||||
.toc a { color: var(--secondary); text-decoration: none; }
|
||||
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
|
||||
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
|
||||
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
|
||||
.arch-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(280px, 1fr)); gap: 20px; margin: 20px 0; }
|
||||
.arch-card { background: white; padding: 20px; border-radius: 10px; box-shadow: 0 4px 12px rgba(0,0,0,0.08); border-top: 4px solid var(--secondary); }
|
||||
.arch-card h4 { color: var(--primary); margin-bottom: 12px; }
|
||||
.arch-card ul { list-style: none; padding-right: 0; }
|
||||
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; }
|
||||
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
|
||||
<div class="header">
|
||||
<h1>🔄 بازطراحی XFileContentExtractor بر پایه DI</h1>
|
||||
<div class="subtitle">حذف Reflection و جایگزینی با تزریق وابستگی استاندارد برای مدیریت Extractor ها</div>
|
||||
</div>
|
||||
|
||||
<div class="meta-bar">
|
||||
<div class="meta-item">👨💻 <strong>توسعهدهنده:</strong> هادی خزاعی اصل</div>
|
||||
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
|
||||
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
|
||||
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
|
||||
</div>
|
||||
|
||||
<div class="content">
|
||||
|
||||
<div class="toc">
|
||||
<h3>📑 فهرست مطالب</h3>
|
||||
<ol>
|
||||
<li><a href="#analysis">تحلیل نواقص کد فعلی</a></li>
|
||||
<li><a href="#solution">گام ۱: بازنویسی کلاس XFileContentExtractor</a></li>
|
||||
<li><a href="#di-registration">گام ۲: ثبت سرویسها در Startup.cs</a></li>
|
||||
<li><a href="#benefits">مزایای رویکرد جدید</a></li>
|
||||
</ol>
|
||||
</div>
|
||||
|
||||
<!-- Section 1: Analysis -->
|
||||
<div class="section" id="analysis">
|
||||
<h2>🔍 گام ۱: تحلیل نواقص کد فعلی</h2>
|
||||
<p>در نسخه فعلی <code>XFileContentExtractor</code>، از مکانیزم <strong>Reflection</strong> (<code>Assembly.LoadFrom</code> و <code>Activator.CreateInstance</code>) برای یافتن و ساخت نمونههای Extractor استفاده شده است. این رویکرد دارای نواقص جدی زیر است:</p>
|
||||
|
||||
<div class="arch-grid">
|
||||
<div class="arch-card">
|
||||
<h4>❌ عدم پشتیبانی از وابستگیها (DI Bypass)</h4>
|
||||
<p>کلاسهایی مانند <code>XVisionFileContentExtractor</code> یا <code>XAudioFileContentExtractor</code> دارای وابستگیهایی مانند <code>IXDefaultAIOCRService</code> یا <code>XAiApiConfiguration</code> هستند. <code>Activator.CreateInstance</code> نمیتواند این وابستگیها را حل کند و باعث خطای <code>MissingMethodException</code> میشود.</p>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>❌ مشکل قفلشدگی فایل (File Locking)</h4>
|
||||
<p>استفاده از <code>Assembly.LoadFrom</code> در حلقه روی فایلهای DLL میتواند باعث قفل شدن فایلها در محیطهای هاستینگ (مانند IIS) و جلوگیری از بهروزرسانی یا Deploy شود.</p>
|
||||
</div>
|
||||
<div class="arch-card">
|
||||
<h4>❌ خطر حلقه بینهایت (Circular Dependency)</h4>
|
||||
<p>اگر <code>XFileContentExtractor</code> خودش به عنوان <code>IXFileContentExtractor</code> در DI ثبت شود، ممکن است در لیست بازگردانده شود و باعث فراخوانی بازگشتی بینهایت در متدهای <code>CanExtract</code> شود.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="alert alert-danger">
|
||||
<strong>⚠️ نتیجهگیری:</strong> استفاده از Reflection برای ساخت اشیایی که وابستگی دارند، یک Anti-Pattern در .NET Core است. راهحل استاندارد، استفاده از قابلیت <code>IEnumerable<T></code> در تزریق وابستگی است.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 2: Solution -->
|
||||
<div class="section" id="solution">
|
||||
<h2>🛠️ گام ۲: بازنویسی کلاس XFileContentExtractor</h2>
|
||||
<p>به جای اسکن دستی DLL ها، از کانتینر DI میخواهیم تمام پیادهسازیهای ثبتشدهی <code>IXFileContentExtractor</code> را به ما تزریق کند. سپس خودِ کامپوزیت را از لیست فیلتر میکنیم.</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Providers/Extractors/XFileContentExtractor.cs</span>
|
||||
</div>
|
||||
|
||||
<pre>using System;
|
||||
using System.Collections.Generic;
|
||||
using System.IO;
|
||||
using System.Linq;
|
||||
using System.Threading;
|
||||
using System.Threading.Tasks;
|
||||
using xAiApi.Interfaces.Extractors;
|
||||
using xAiModels.Models;
|
||||
using xCommons.Extensions;
|
||||
using xExceptions.Constants;
|
||||
|
||||
namespace xAiApi.Providers.Extractors
|
||||
{
|
||||
/// <summary>
|
||||
/// Composite extractor that delegates to appropriate extractor
|
||||
/// based on MIME type using Dependency Injection ...
|
||||
/// </summary>
|
||||
public class XFileContentExtractor : IXFileContentExtractor
|
||||
{
|
||||
private readonly IList<IXFileContentExtractor> extractors;
|
||||
|
||||
/// <summary>
|
||||
/// Constructor: Inject all registered IXFileContentExtractor instances ...
|
||||
/// </summary>
|
||||
public XFileContentExtractor(IEnumerable<IXFileContentExtractor> availableExtractors)
|
||||
{
|
||||
// فیلتر کردن خودِ این کلاس برای جلوگیری از حلقه بینهایت (Circular Dependency)
|
||||
extractors = availableExtractors
|
||||
.Where(e => e.GetType() != typeof(XFileContentExtractor))
|
||||
.ToList();
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Check if any registered extractor supports the specified MIME type ...
|
||||
/// </summary>
|
||||
public bool CanExtract(string mimeType)
|
||||
{
|
||||
return extractors.Any(e => e.CanExtract(mimeType));
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract text content from file stream ...
|
||||
/// </summary>
|
||||
public async Task<string> ExtractAsync(
|
||||
Stream fileStream,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
var extractor = extractors.FirstOrDefault(e => e.CanExtract(mimeType));
|
||||
|
||||
if (extractor.IsNull())
|
||||
{
|
||||
XException.NotAllowed.Throw($"Unsupported file type: {mimeType}");
|
||||
}
|
||||
|
||||
return await extractor.ExtractAsync(
|
||||
fileStream: fileStream,
|
||||
mimeType: mimeType,
|
||||
cancellationToken: cancellationToken
|
||||
);
|
||||
}
|
||||
|
||||
/// <summary>
|
||||
/// Extract content from stream as Rich Result ...
|
||||
/// </summary>
|
||||
public async Task<XFileExtractionResult> ExtractRichAsync(
|
||||
Stream fileStream,
|
||||
string fileName,
|
||||
string mimeType,
|
||||
CancellationToken cancellationToken = default
|
||||
)
|
||||
{
|
||||
var extractor = extractors.FirstOrDefault(e => e.CanExtract(mimeType));
|
||||
|
||||
if (extractor.IsNull())
|
||||
{
|
||||
XException.NotAllowed.Throw($"Unsupported file type: {mimeType}");
|
||||
}
|
||||
|
||||
return await extractor.ExtractRichAsync(
|
||||
fileStream: fileStream,
|
||||
fileName: fileName,
|
||||
mimeType: mimeType,
|
||||
cancellationToken: cancellationToken
|
||||
);
|
||||
}
|
||||
}
|
||||
}</pre>
|
||||
</div>
|
||||
|
||||
<!-- Section 3: DI Registration -->
|
||||
<div class="section" id="di-registration">
|
||||
<h2>🔌 گام ۳: ثبت صحیح سرویسها در Startup.cs</h2>
|
||||
<p>برای اینکه تزریق <code>IEnumerable<IXFileContentExtractor></code> کار کند، باید تمام Extractor های خاص را به صورت <code>Singleton</code> (چون State-less هستند) در کانتینر DI ثبت کنیم. کانتینر .NET Core به طور خودکار آنها را در یک لیست جمعآوری میکند.</p>
|
||||
|
||||
<div class="file-change">
|
||||
<span class="badge-modify">MODIFY</span>
|
||||
<span class="path">xAiApi/Startup.cs (متد ConfigureServices)</span>
|
||||
</div>
|
||||
|
||||
<pre>public void ConfigureServices(IServiceCollection services)
|
||||
{
|
||||
// ... (ثبتهای قبلی سرویسها)
|
||||
|
||||
// ==========================================================
|
||||
// ✅ ثبت File Content Extractors (الگوی Composite)
|
||||
// ==========================================================
|
||||
|
||||
// ۱. ثبت Extractor های پایه
|
||||
services.AddSingleton<IXFileContentExtractor, XPlainTextFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XDocxFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XExcelFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XPdfFileContentExtractor>();
|
||||
|
||||
// ۲. ثبت Extractor های پیشرفته (تصویر و صوت)
|
||||
// نکته: XImageFileContentExtractor و XAudioFileContentExtractor باید در پروژه موجود باشند
|
||||
services.AddSingleton<IXFileContentExtractor, XImageFileContentExtractor>();
|
||||
services.AddSingleton<IXFileContentExtractor, XAudioFileContentExtractor>();
|
||||
|
||||
// ۳. ثبت Vision Extractor (که وابستگی به IXDefaultAIOCRService دارد)
|
||||
// این خط به طور خودکار وابستگیهای XVisionFileContentExtractor را از DI حل میکند
|
||||
services.AddSingleton<IXFileContentExtractor, XVisionFileContentExtractor>();
|
||||
|
||||
// ۴. ثبت کامپوزیت اصلی (این کلاس لیست بالا را در Constructor دریافت میکند)
|
||||
services.AddSingleton<IXFileContentExtractor, XFileContentExtractor>();
|
||||
|
||||
// ... (ثبت سرویسهای AI و سایر موارد)
|
||||
services.AddScoped<IXDefaultAiService, XDefaultAiService>();
|
||||
services.AddScoped<IXDefaultAIOCRService, XDefaultAIOCRService>();
|
||||
services.AddScoped<IXDefaultEmbeddingService, XDefaultEmbeddingService>();
|
||||
services.AddScoped<IXDefaultThinkingAiService, XDefaultThinkingAiService>();
|
||||
}</pre>
|
||||
|
||||
<div class="alert alert-info">
|
||||
<strong>💡 نکته حیاتی درباره ترتیب ثبت:</strong><br>
|
||||
در .NET Core، وقتی <code>IEnumerable<T></code> را Inject میکنید، تمام ثبتهای <code>T</code> (شامل خودِ <code>XFileContentExtractor</code> اگر قبل از فیلتر کردن باشد) را برمیگرداند. به همین دلیل در Constructor کلاس <code>XFileContentExtractor</code>، خط <code>.Where(e => e.GetType() != typeof(XFileContentExtractor))</code> اضافه شده است تا از حلقه بینهایت جلوگیری شود.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Section 4: Benefits -->
|
||||
<div class="section" id="benefits">
|
||||
<h2>✨ گام ۴: مزایای رویکرد جدید</h2>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<th>ویژگی</th>
|
||||
<th>رویکرد قدیمی (Reflection)</th>
|
||||
<th>رویکرد جدید (DI)</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>حل وابستگیها (Dependencies)</td>
|
||||
<td><span style="color: var(--danger);">❌ شکست میخورد</span></td>
|
||||
<td><span style="color: var(--success);">✅ به طور خودکار حل میشود</span></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>عملکرد (Performance)</td>
|
||||
<td><span style="color: var(--warning);">⚠️ کند (اسکن DLL در هر بار)</span></td>
|
||||
<td><span style="color: var(--success);">✅ بسیار سریع (Resolved at startup)</span></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>قابلیت تست (Unit Testing)</td>
|
||||
<td><span style="color: var(--danger);">❌ بسیار دشوار (Mocking سخت)</span></td>
|
||||
<td><span style="color: var(--success);">✅ آسان (تزریق لیست Mock)</span></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>پایداری در محیط Production</td>
|
||||
<td><span style="color: var(--danger);">❌ خطر File Locking</span></td>
|
||||
<td><span style="color: var(--success);">✅ کاملاً پایدار و استاندارد</span></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>افزودن Extractor جدید</td>
|
||||
<td>خودکار (اما با ریسک)</td>
|
||||
<td>فقط افزودن یک خط <code>AddSingleton</code> در <code>Startup.cs</code></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<div class="alert alert-success">
|
||||
<strong>✅ نتیجهگیری نهایی:</strong><br>
|
||||
با این تغییر، معماری پروژه شما کاملاً با اصول <strong>SOLID</strong> (به ویژه Dependency Inversion) و الگوهای استاندارد .NET Core همسو میشود. کلاس <code>XVisionFileContentExtractor</code> که به <code>IXDefaultAIOCRService</code> وابسته است، اکنون بدون هیچ خطایی مقداردهی اولیه خواهد شد.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
|
||||
<div class="footer">
|
||||
<p><strong>👨💻 توسعهدهنده:</strong> هادی خزاعی اصل</p>
|
||||
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
|
||||
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
|
||||
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
|
||||
🔄 مستند فنی بازطراحی XFileContentExtractor بر پایه Dependency Injection - تمامی حقوق محفوظ است
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user