This commit is contained in:
2026-10-03 18:04:36 +03:30
parent 92988ea155
commit 81b900b887
20 changed files with 48612 additions and 4 deletions
+682
View File
@@ -0,0 +1,682 @@
<!DOCTYPE html>
<html lang="fa" dir="rtl">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>پیاده‌سازی OCR با Tesseract در XImageFileContentExtractor - فن آوران ساحر علم</title>
<style>
:root {
--primary: #1e3a8a;
--secondary: #3b82f6;
--accent: #f59e0b;
--success: #10b981;
--danger: #ef4444;
--warning: #f97316;
--bg-light: #f8fafc;
--bg-code: #1e293b;
--text-dark: #0f172a;
--text-muted: #64748b;
--border: #e2e8f0;
}
* { box-sizing: border-box; margin: 0; padding: 0; }
body {
font-family: 'Tahoma', 'Segoe UI', sans-serif;
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
color: var(--text-dark);
line-height: 1.8;
padding: 20px;
}
.container {
max-width: 1200px;
margin: 0 auto;
background: white;
border-radius: 16px;
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
overflow: hidden;
}
.header {
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
color: white;
padding: 40px;
text-align: center;
}
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
.meta-bar {
display: flex;
justify-content: space-between;
background: var(--bg-light);
padding: 15px 30px;
border-bottom: 2px solid var(--border);
flex-wrap: wrap;
gap: 15px;
}
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
.meta-item strong { color: var(--primary); }
.content { padding: 40px; }
.section {
margin-bottom: 35px;
padding: 25px;
background: var(--bg-light);
border-radius: 12px;
border-right: 5px solid var(--secondary);
}
.section h2 {
color: var(--primary);
font-size: 1.5em;
margin-bottom: 20px;
padding-bottom: 10px;
border-bottom: 2px solid var(--border);
}
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
pre {
background: var(--bg-code);
color: #e2e8f0;
padding: 18px;
border-radius: 8px;
overflow-x: auto;
direction: ltr;
text-align: left;
font-family: 'Consolas', monospace;
font-size: 0.85em;
margin: 15px 0;
border-right: 4px solid var(--accent);
}
code {
background: #fef3c7;
color: #92400e;
padding: 2px 8px;
border-radius: 4px;
font-family: 'Consolas', monospace;
font-size: 0.9em;
direction: ltr;
display: inline-block;
}
table {
width: 100%;
border-collapse: collapse;
margin: 15px 0;
background: white;
border-radius: 8px;
overflow: hidden;
}
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
td { padding: 12px; border-bottom: 1px solid var(--border); }
tr:hover { background: var(--bg-light); }
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
.footer p { margin: 5px 0; }
.highlight { background: linear-gradient(120deg, #fef3c7 0%, #fef3c7 100%); padding: 2px 6px; border-radius: 4px; font-weight: bold; }
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
.toc h3 { color: var(--primary); margin-bottom: 15px; }
.toc ol { padding-right: 25px; }
.toc li { padding: 6px 0; }
.toc a { color: var(--secondary); text-decoration: none; }
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
.arch-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(280px, 1fr)); gap: 20px; margin: 20px 0; }
.arch-card { background: white; padding: 20px; border-radius: 10px; box-shadow: 0 4px 12px rgba(0,0,0,0.08); border-top: 4px solid var(--secondary); }
.arch-card h4 { color: var(--primary); margin-bottom: 12px; }
.arch-card ul { list-style: none; padding-right: 0; }
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; }
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
</style>
</head>
<body>
<div class="container">
<div class="header">
<h1>🔍 پیاده‌سازی OCR با Tesseract</h1>
<div class="subtitle">تکمیل متد PerformOcrAsync در XImageFileContentExtractor برای استخراج متن از تصاویر</div>
</div>
<div class="meta-bar">
<div class="meta-item">👨‍💻 <strong>توسعه‌دهنده:</strong> هادی خزاعی اصل</div>
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
</div>
<div class="content">
<div class="toc">
<h3>📑 فهرست مطالب</h3>
<ol>
<li><a href="#step1">گام ۱: نصب پکیج NuGet Tesseract</a></li>
<li><a href="#step2">گام ۲: دانلود فایل‌های Trained Data</a></li>
<li><a href="#step3">گام ۳: پیکربندی در appsettings.json</a></li>
<li><a href="#step4">گام ۴: بازنویسی کلاس XImageFileContentExtractor</a></li>
<li><a href="#step5">گام ۵: پیاده‌سازی متد PerformOcrAsync</a></li>
<li><a href="#step6">گام ۶: نکات مهم و عیب‌یابی</a></li>
</ol>
</div>
<!-- Section 1: NuGet -->
<div class="section" id="step1">
<h2>📦 گام ۱: نصب پکیج NuGet Tesseract</h2>
<p>برای استفاده از Tesseract در پروژه .NET، پکیج زیر را نصب کنید:</p>
<pre># در Package Manager Console:
Install-Package Tesseract
# یا در .NET CLI:
dotnet add package Tesseract</pre>
<div class="alert alert-info">
<strong>💡 نکته:</strong> پکیج <code>Tesseract</code> به صورت خودکار Native DLL های مورد نیاز را نیز دانلود و در پوشه output کپی می‌کند.
</div>
</div>
<!-- Section 2: Trained Data -->
<div class="section" id="step2">
<h2>📥 گام ۲: دانلود فایل‌های Trained Data</h2>
<p>Tesseract برای تشخیص متن به فایل‌های <code>traineddata</code> نیاز دارد. این فایل‌ها مدل‌های زبانی هستند که برای هر زبان جداگانه آموزش دیده‌اند.</p>
<h3>۲.۱. دانلود فایل‌ها:</h3>
<p>از لینک زیر فایل‌های مورد نیاز را دانلود کنید:</p>
<p><a href="https://github.com/tesseract-ocr/tessdata" target="_blank">https://github.com/tesseract-ocr/tessdata</a></p>
<h3>۲.۲. فایل‌های مورد نیاز:</h3>
<table>
<tr>
<th>فایل</th>
<th>زبان</th>
<th>حجم تقریبی</th>
<th>کاربرد</th>
</tr>
<tr>
<td><code>fas.traineddata</code></td>
<td>فارسی</td>
<td>~۱۰ MB</td>
<td>تشخیص متن فارسی</td>
</tr>
<tr>
<td><code>eng.traineddata</code></td>
<td>انگلیسی</td>
<td>~۴ MB</td>
<td>تشخیص متن انگلیسی</td>
</tr>
<tr>
<td><code>ara.traineddata</code></td>
<td>عربی</td>
<td>~۵ MB</td>
<td>تشخیص متن عربی (اختیاری)</td>
</tr>
</table>
<h3>۲.۳. ساختار پوشه‌ها:</h3>
<p>فایل‌های دانلود شده را در مسیر زیر قرار دهید:</p>
<pre>xAiApi/
└── tessdata/
├── fas.traineddata
├── eng.traineddata
└── ara.traineddata (اختیاری)</pre>
<div class="alert alert-warning">
<strong>⚠️ نکته مهم:</strong> مسیر <code>tessdata</code> باید در زمان اجرا در دسترس باشد. می‌توانید آن را در <code>appsettings.json</code> پیکربندی کنید.
</div>
</div>
<!-- Section 3: Configuration -->
<div class="section" id="step3">
<h2>⚙️ گام ۳: پیکربندی در appsettings.json</h2>
<p>تنظیمات Tesseract را در فایل <code>appsettings.json</code> اضافه کنید:</p>
<pre>{
"AiApiConfiguration": {
"FileExtraction": {
"OCR": {
"TessDataPath": "tessdata",
"DefaultLanguage": "fas+eng",
"EngineMode": "LstmOnly"
}
},
"Models": [ ... ],
"Prompts": [ ... ]
}
}</pre>
<h3>توضیح پارامترها:</h3>
<table>
<tr>
<th>پارامتر</th>
<th>توضیح</th>
<th>مقادیر مجاز</th>
</tr>
<tr>
<td><code>TessDataPath</code></td>
<td>مسیر پوشه فایل‌های traineddata</td>
<td>مسیر نسبی یا مطلق</td>
</tr>
<tr>
<td><code>DefaultLanguage</code></td>
<td>زبان‌های پیش‌فرض برای OCR</td>
<td>fas, eng, ara (ترکیب با +)</td>
</tr>
<tr>
<td><code>EngineMode</code></td>
<td>حالت موتور Tesseract</td>
<td>TesseractOnly, LstmOnly, TesseractAndLstm, Default</td>
</tr>
</table>
</div>
<!-- Section 4: Rewrite Class -->
<div class="section" id="step4">
<h2>🔄 گام ۴: بازنویسی کلاس XImageFileContentExtractor</h2>
<p>اکنون کلاس <code>XImageFileContentExtractor</code> را بازنویسی می‌کنیم تا از Tesseract برای OCR استفاده کند:</p>
<div class="file-change">
<span class="badge-modify">MODIFY</span>
<span class="path">xAiApi/Providers/Extractors/XImageFileContentExtractor.cs</span>
</div>
<pre>using System;
using System.IO;
using System.Linq;
using System.Threading;
using Tesseract;
using xAiModels.Models;
using System.Threading.Tasks;
using Microsoft.Extensions.Logging;
using xAiApi.Interfaces.Extractors;
namespace xAiApi.Providers.Extractors
{
/// &lt;summary&gt;
/// Extracts content from image files using Tesseract OCR ...
/// &lt;/summary&gt;
public class XImageFileContentExtractor : IXImageFileContentExtractor
{
/// &lt;summary&gt;
/// Supported MIME Types ...
/// &lt;/summary&gt;
private static readonly string[] SupportedMimeTypes =
[
"image/png",
"image/jpeg",
"image/jpg",
"image/gif",
"image/webp",
"image/bmp"
];
/// &lt;summary&gt;
/// مسیر پوشه tessdata ...
/// &lt;/summary&gt;
private readonly string _tessDataPath;
/// &lt;summary&gt;
/// زبان پیش‌فرض برای OCR ...
/// &lt;/summary&gt;
private readonly string _defaultLanguage;
/// &lt;summary&gt;
/// حالت موتور Tesseract ...
/// &lt;/summary&gt;
private readonly OcrEngineMode _engineMode;
/// &lt;summary&gt;
/// Logger ...
/// &lt;/summary&gt;
private readonly ILogger&lt;XImageFileContentExtractor&gt; _logger;
/// &lt;summary&gt;
/// Constructor با پیکربندی پیش‌فرض ...
/// &lt;/summary&gt;
public XImageFileContentExtractor(
ILogger&lt;XImageFileContentExtractor&gt; logger,
string tessDataPath = "tessdata",
string defaultLanguage = "fas+eng",
OcrEngineMode engineMode = OcrEngineMode.LstmOnly
)
{
_logger = logger;
_tessDataPath = tessDataPath;
_defaultLanguage = defaultLanguage;
_engineMode = engineMode;
}
/// &lt;summary&gt;
/// Check if this extractor supports the specified MIME type ...
/// &lt;/summary&gt;
public bool CanExtract(string mimeType)
{
return SupportedMimeTypes.Contains(
mimeType?.ToLowerInvariant() ?? string.Empty
);
}
/// &lt;summary&gt;
/// Extract text content from file stream ...
/// &lt;/summary&gt;
public async Task&lt;string&gt; ExtractAsync(
Stream fileStream,
string mimeType,
CancellationToken cancellationToken = default
)
{
var result = await ExtractRichAsync(
fileStream,
"image",
mimeType,
cancellationToken
);
return result.Text;
}
/// &lt;summary&gt;
/// Extract content from stream as Rich Result ...
/// &lt;/summary&gt;
public async Task&lt;XFileExtractionResult&gt; ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
CancellationToken cancellationToken = default
)
{
var result = new XFileExtractionResult
{
FileName = fileName,
MimeType = mimeType
};
// خواندن بایت‌های تصویر
using var memoryStream = new MemoryStream();
await fileStream.CopyToAsync(memoryStream, cancellationToken);
var imageBytes = memoryStream.ToArray();
// افزودن تصویر به عنوان DataContent (برای Vision Models)
result.Images.Add(new XExtractedImage
{
Bytes = imageBytes,
MimeType = mimeType,
Description = $"Attached image: {fileName}"
});
// انجام OCR برای استخراج متن
try
{
var ocrText = await PerformOcrAsync(imageBytes, cancellationToken);
if (!string.IsNullOrWhiteSpace(ocrText))
{
result.Text = ocrText;
_logger.LogInformation(
"OCR completed successfully for file: {FileName}, extracted {Length} characters",
fileName,
ocrText.Length
);
}
else
{
_logger.LogWarning("OCR completed but no text was extracted from file: {FileName}", fileName);
}
}
catch (Exception ex)
{
_logger.LogError(ex, "OCR failed for file: {FileName}", fileName);
result.ErrorMessage = $"OCR failed: {ex.Message}";
// تصویر همچنان برای Vision Models در دسترس است
}
return result;
}
/// &lt;summary&gt;
/// انجام OCR روی تصویر با استفاده از Tesseract ...
/// &lt;/summary&gt;
private async Task&lt;string&gt; PerformOcrAsync(
byte[] imageBytes,
CancellationToken cancellationToken
)
{
return await Task.Run(() =&gt;
{
// بررسی وجود مسیر tessdata
if (!Directory.Exists(_tessDataPath))
{
throw new DirectoryNotFoundException(
$"TessData directory not found at: {_tessDataPath}. " +
"Please download traineddata files from https://github.com/tesseract-ocr/tessdata"
);
}
// ایجاد Tesseract Engine
using var engine = new TesseractEngine(
datapath: _tessDataPath,
language: _defaultLanguage,
mode: _engineMode
);
// بارگذاری تصویر از byte array
using var pix = Pix.LoadFromMemory(imageBytes);
// تنظیم CancellationToken
cancellationToken.ThrowIfCancellationRequested();
// انجام OCR
using var page = engine.Process(pix);
// استخراج متن
var text = page.GetText();
// پاکسازی متن (حذف فاصله‌های اضافی)
text = text?.Trim();
return text ?? string.Empty;
}, cancellationToken);
}
}
}</pre>
</div>
<!-- Section 5: Implementation Details -->
<div class="section" id="step5">
<h2>🔧 گام ۵: جزئیات پیاده‌سازی متد PerformOcrAsync</h2>
<h3>۵.۱. مراحل کار متد:</h3>
<div class="arch-grid">
<div class="arch-card">
<h4>۱. بررسی مسیر tessdata</h4>
<ul>
<li>اطمینان از وجود پوشه tessdata</li>
<li>پرتاب خطا در صورت عدم وجود</li>
</ul>
</div>
<div class="arch-card">
<h4>۲. ایجاد Tesseract Engine</h4>
<ul>
<li>بارگذاری مدل زبانی</li>
<li>تنظیم حالت موتور (LSTM)</li>
</ul>
</div>
<div class="arch-card">
<h4>۳. بارگذاری تصویر</h4>
<ul>
<li>تبدیل byte[] به Pix object</li>
<li>پشتیبانی از فرمت‌های مختلف</li>
</ul>
</div>
<div class="arch-card">
<h4>۴. انجام OCR</h4>
<ul>
<li>پردازش تصویر</li>
<li>استخراج متن</li>
</ul>
</div>
<div class="arch-card">
<h4>۵. پاکسازی متن</h4>
<ul>
<li>حذف فاصله‌های اضافی</li>
<li>بررسی CancellationToken</li>
</ul>
</div>
</div>
<h3>۵.۲. ویژگی‌های کلیدی پیاده‌سازی:</h3>
<table>
<tr>
<th>ویژگی</th>
<th>توضیح</th>
</tr>
<tr>
<td><strong>پشتیبانی چند زبانه</strong></td>
<td>استفاده از <code>fas+eng</code> برای تشخیص همزمان فارسی و انگلیسی</td>
</tr>
<tr>
<td><strong>حالت LSTM</strong></td>
<td>استفاده از موتور LSTM برای دقت بالاتر در متون فارسی</td>
</tr>
<tr>
<td><strong>مدیریت خطا</strong></td>
<td>بررسی وجود tessdata و مدیریت استثناها</td>
</tr>
<tr>
<td><strong>Logging</strong></td>
<td>ثبت موفقیت/شکست OCR با جزئیات</td>
</tr>
<tr>
<td><strong>Cancellation Support</strong></td>
<td>پشتیبانی از لغو عملیات در میانه کار</td>
</tr>
<tr>
<td><strong>Resource Management</strong></td>
<td>استفاده از <code>using</code> برای آزادسازی منابع</td>
</tr>
</table>
</div>
<!-- Section 6: Tips -->
<div class="section" id="step6">
<h2>💡 گام ۶: نکات مهم و عیب‌یابی</h2>
<h3>۶.۱. مشکلات رایج و راه‌حل‌ها:</h3>
<table>
<tr>
<th>مشکل</th>
<th>علت</th>
<th>راه‌حل</th>
</tr>
<tr>
<td><code>DirectoryNotFoundException</code></td>
<td>پوشه tessdata یافت نشد</td>
<td>دانلود traineddata و قرار دادن در مسیر صحیح</td>
</tr>
<tr>
<td><code>TesseractException</code></td>
<td>فایل traineddata خراب یا ناسازگار</td>
<td>دانلود مجدد از منبع رسمی</td>
</tr>
<tr>
<td>دقت پایین OCR</td>
<td>کیفیت پایین تصویر</td>
<td>پیش‌پردازش تصویر (افزایش کنتراست، resize)</td>
</tr>
<tr>
<td>تشخیص نادرست فارسی</td>
<td>ترکیب نامناسب زبان‌ها</td>
<td>استفاده از <code>fas</code> به تنهایی یا <code>fas+eng</code></td>
</tr>
<tr>
<td>کندی عملکرد</td>
<td>تصاویر بزرگ</td>
<td>تغییر EngineMode به <code>TesseractOnly</code></td>
</tr>
</table>
<h3>۶.۲. بهینه‌سازی عملکرد:</h3>
<div class="arch-grid">
<div class="arch-card">
<h4>🚀 افزایش سرعت</h4>
<ul>
<li>استفاده از <code>OcrEngineMode.TesseractOnly</code></li>
<li>کاهش DPI تصویر (مثلاً 150 به جای 300)</li>
<li>استفاده از یک زبان به جای چند زبان</li>
</ul>
</div>
<div class="arch-card">
<h4>🎯 افزایش دقت</h4>
<ul>
<li>استفاده از <code>OcrEngineMode.LstmOnly</code></li>
<li>افزایش DPI تصویر (300 یا بالاتر)</li>
<li>پیش‌پردازش تصویر (binarization)</li>
</ul>
</div>
</div>
<h3>۶.۳. پیش‌پردازش تصویر (اختیاری):</h3>
<p>برای افزایش دقت OCR، می‌توانید قبل از پردازش، تصویر را پیش‌پردازش کنید:</p>
<pre>// مثال: افزایش کنتراست و تبدیل به سیاه و سفید
using var pix = Pix.LoadFromMemory(imageBytes);
using var enhanced = pix.ConvertTo1(); // تبدیل به 1-bit
using var page = engine.Process(enhanced);</pre>
<h3>۶.۴. ثبت در Startup.cs:</h3>
<p>اکنون <code>XImageFileContentExtractor</code> را با پیکربندی صحیح ثبت کنید:</p>
<pre>public void ConfigureServices(IServiceCollection services)
{
// ... سایر ثبت‌ها
// ثبت XImageFileContentExtractor با پیکربندی
services.AddSingleton&lt;IXFileContentExtractor&gt;(sp =&gt;
{
var logger = sp.GetRequiredService&lt;ILogger&lt;XImageFileContentExtractor&gt;&gt;();
var configuration = sp.GetRequiredService&lt;IConfiguration&gt;();
var tessDataPath = configuration.GetValue&lt;string&gt;(
"AiApiConfiguration:FileExtraction:OCR:TessDataPath"
) ?? "tessdata";
var defaultLanguage = configuration.GetValue&lt;string&gt;(
"AiApiConfiguration:FileExtraction:OCR:DefaultLanguage"
) ?? "fas+eng";
return new XImageFileContentExtractor(
logger: logger,
tessDataPath: tessDataPath,
defaultLanguage: defaultLanguage,
engineMode: OcrEngineMode.LstmOnly
);
});
// ... سایر ثبت‌ها
}</pre>
<div class="alert alert-success">
<strong>✅ نتیجه نهایی:</strong>
<ul style="padding-right: 25px; margin-top: 10px;">
<li>🔍 <strong>OCR کامل:</strong> استخراج متن از تصاویر با دقت بالا</li>
<li>🌐 <strong>چند زبانه:</strong> پشتیبانی از فارسی، انگلیسی و عربی</li>
<li>⚡ <strong>بهینه:</strong> استفاده از موتور LSTM برای دقت بالاتر</li>
<li>🛡️ <strong>مقاوم:</strong> مدیریت خطا و logging کامل</li>
<li>🔄 <strong>قابل لغو:</strong> پشتیبانی از CancellationToken</li>
<li>📦 <strong>منعطف:</strong> پیکربندی از طریق appsettings.json</li>
</ul>
</div>
<div class="alert alert-info">
<strong>💡 نکته تکمیلی:</strong>
<p>اگر نیاز به دقت بالاتر برای متون فارسی دارید، می‌توانید از مدل‌های آموزش‌دیده‌شده خاص فارسی استفاده کنید یا از ترکیب Tesseract با مدل‌های deep learning (مانند CRNN) بهره ببرید.</p>
</div>
</div>
</div>
<div class="footer">
<p><strong>👨‍💻 توسعه‌دهنده:</strong> هادی خزاعی اصل</p>
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
🔍 مستند فنی پیاده‌سازی OCR با Tesseract - تمامی حقوق محفوظ است
</p>
</div>
</div>
</body>
</html>