Files
xSaherelmWorkspace/Documents/Docs/Developer/14.html
T
2026-10-03 18:04:36 +03:30

682 lines
28 KiB
HTML
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
<!DOCTYPE html>
<html lang="fa" dir="rtl">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>پیاده‌سازی OCR با Tesseract در XImageFileContentExtractor - فن آوران ساحر علم</title>
<style>
:root {
--primary: #1e3a8a;
--secondary: #3b82f6;
--accent: #f59e0b;
--success: #10b981;
--danger: #ef4444;
--warning: #f97316;
--bg-light: #f8fafc;
--bg-code: #1e293b;
--text-dark: #0f172a;
--text-muted: #64748b;
--border: #e2e8f0;
}
* { box-sizing: border-box; margin: 0; padding: 0; }
body {
font-family: 'Tahoma', 'Segoe UI', sans-serif;
background: linear-gradient(135deg, #f8fafc 0%, #e0e7ff 100%);
color: var(--text-dark);
line-height: 1.8;
padding: 20px;
}
.container {
max-width: 1200px;
margin: 0 auto;
background: white;
border-radius: 16px;
box-shadow: 0 20px 60px rgba(0,0,0,0.1);
overflow: hidden;
}
.header {
background: linear-gradient(135deg, var(--primary) 0%, var(--secondary) 100%);
color: white;
padding: 40px;
text-align: center;
}
.header h1 { font-size: 2.1em; margin-bottom: 10px; }
.header .subtitle { font-size: 1.1em; opacity: 0.95; }
.meta-bar {
display: flex;
justify-content: space-between;
background: var(--bg-light);
padding: 15px 30px;
border-bottom: 2px solid var(--border);
flex-wrap: wrap;
gap: 15px;
}
.meta-item { display: flex; align-items: center; gap: 8px; font-size: 0.9em; color: var(--text-muted); }
.meta-item strong { color: var(--primary); }
.content { padding: 40px; }
.section {
margin-bottom: 35px;
padding: 25px;
background: var(--bg-light);
border-radius: 12px;
border-right: 5px solid var(--secondary);
}
.section h2 {
color: var(--primary);
font-size: 1.5em;
margin-bottom: 20px;
padding-bottom: 10px;
border-bottom: 2px solid var(--border);
}
.section h3 { color: var(--secondary); font-size: 1.2em; margin: 20px 0 12px; }
pre {
background: var(--bg-code);
color: #e2e8f0;
padding: 18px;
border-radius: 8px;
overflow-x: auto;
direction: ltr;
text-align: left;
font-family: 'Consolas', monospace;
font-size: 0.85em;
margin: 15px 0;
border-right: 4px solid var(--accent);
}
code {
background: #fef3c7;
color: #92400e;
padding: 2px 8px;
border-radius: 4px;
font-family: 'Consolas', monospace;
font-size: 0.9em;
direction: ltr;
display: inline-block;
}
table {
width: 100%;
border-collapse: collapse;
margin: 15px 0;
background: white;
border-radius: 8px;
overflow: hidden;
}
th { background: var(--primary); color: white; padding: 12px; text-align: right; }
td { padding: 12px; border-bottom: 1px solid var(--border); }
tr:hover { background: var(--bg-light); }
.alert { padding: 15px 20px; border-radius: 8px; margin: 15px 0; border-right: 4px solid; }
.alert-info { background: #dbeafe; border-color: var(--secondary); color: #1e40af; }
.alert-success { background: #d1fae5; border-color: var(--success); color: #065f46; }
.alert-warning { background: #fef3c7; border-color: var(--accent); color: #92400e; }
.alert-danger { background: #fee2e2; border-color: var(--danger); color: #991b1b; }
.footer { background: var(--primary); color: white; padding: 25px; text-align: center; }
.footer p { margin: 5px 0; }
.highlight { background: linear-gradient(120deg, #fef3c7 0%, #fef3c7 100%); padding: 2px 6px; border-radius: 4px; font-weight: bold; }
.toc { background: white; padding: 20px; border-radius: 10px; margin-bottom: 25px; border: 2px solid var(--border); }
.toc h3 { color: var(--primary); margin-bottom: 15px; }
.toc ol { padding-right: 25px; }
.toc li { padding: 6px 0; }
.toc a { color: var(--secondary); text-decoration: none; }
.file-change { background: #f0f9ff; border-right: 4px solid var(--secondary); padding: 15px; margin: 10px 0; border-radius: 8px; }
.file-change .path { font-family: 'Consolas', monospace; color: var(--primary); font-weight: bold; direction: ltr; display: inline-block; }
.badge-modify { background: var(--warning); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
.badge-new { background: var(--success); color: white; padding: 2px 8px; border-radius: 4px; font-size: 0.75em; margin-right: 8px; }
.arch-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(280px, 1fr)); gap: 20px; margin: 20px 0; }
.arch-card { background: white; padding: 20px; border-radius: 10px; box-shadow: 0 4px 12px rgba(0,0,0,0.08); border-top: 4px solid var(--secondary); }
.arch-card h4 { color: var(--primary); margin-bottom: 12px; }
.arch-card ul { list-style: none; padding-right: 0; }
.arch-card li { padding: 6px 0; padding-right: 20px; position: relative; }
.arch-card li::before { content: '▸'; position: absolute; right: 0; color: var(--accent); font-weight: bold; }
</style>
</head>
<body>
<div class="container">
<div class="header">
<h1>🔍 پیاده‌سازی OCR با Tesseract</h1>
<div class="subtitle">تکمیل متد PerformOcrAsync در XImageFileContentExtractor برای استخراج متن از تصاویر</div>
</div>
<div class="meta-bar">
<div class="meta-item">👨‍💻 <strong>توسعه‌دهنده:</strong> هادی خزاعی اصل</div>
<div class="meta-item">🏢 <strong>شرکت:</strong> فن آوران ساحر علم</div>
<div class="meta-item">📅 <strong>تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</div>
<div class="meta-item">📦 <strong>پروژه:</strong> xAiApi</div>
</div>
<div class="content">
<div class="toc">
<h3>📑 فهرست مطالب</h3>
<ol>
<li><a href="#step1">گام ۱: نصب پکیج NuGet Tesseract</a></li>
<li><a href="#step2">گام ۲: دانلود فایل‌های Trained Data</a></li>
<li><a href="#step3">گام ۳: پیکربندی در appsettings.json</a></li>
<li><a href="#step4">گام ۴: بازنویسی کلاس XImageFileContentExtractor</a></li>
<li><a href="#step5">گام ۵: پیاده‌سازی متد PerformOcrAsync</a></li>
<li><a href="#step6">گام ۶: نکات مهم و عیب‌یابی</a></li>
</ol>
</div>
<!-- Section 1: NuGet -->
<div class="section" id="step1">
<h2>📦 گام ۱: نصب پکیج NuGet Tesseract</h2>
<p>برای استفاده از Tesseract در پروژه .NET، پکیج زیر را نصب کنید:</p>
<pre># در Package Manager Console:
Install-Package Tesseract
# یا در .NET CLI:
dotnet add package Tesseract</pre>
<div class="alert alert-info">
<strong>💡 نکته:</strong> پکیج <code>Tesseract</code> به صورت خودکار Native DLL های مورد نیاز را نیز دانلود و در پوشه output کپی می‌کند.
</div>
</div>
<!-- Section 2: Trained Data -->
<div class="section" id="step2">
<h2>📥 گام ۲: دانلود فایل‌های Trained Data</h2>
<p>Tesseract برای تشخیص متن به فایل‌های <code>traineddata</code> نیاز دارد. این فایل‌ها مدل‌های زبانی هستند که برای هر زبان جداگانه آموزش دیده‌اند.</p>
<h3>۲.۱. دانلود فایل‌ها:</h3>
<p>از لینک زیر فایل‌های مورد نیاز را دانلود کنید:</p>
<p><a href="https://github.com/tesseract-ocr/tessdata" target="_blank">https://github.com/tesseract-ocr/tessdata</a></p>
<h3>۲.۲. فایل‌های مورد نیاز:</h3>
<table>
<tr>
<th>فایل</th>
<th>زبان</th>
<th>حجم تقریبی</th>
<th>کاربرد</th>
</tr>
<tr>
<td><code>fas.traineddata</code></td>
<td>فارسی</td>
<td>~۱۰ MB</td>
<td>تشخیص متن فارسی</td>
</tr>
<tr>
<td><code>eng.traineddata</code></td>
<td>انگلیسی</td>
<td>~۴ MB</td>
<td>تشخیص متن انگلیسی</td>
</tr>
<tr>
<td><code>ara.traineddata</code></td>
<td>عربی</td>
<td>~۵ MB</td>
<td>تشخیص متن عربی (اختیاری)</td>
</tr>
</table>
<h3>۲.۳. ساختار پوشه‌ها:</h3>
<p>فایل‌های دانلود شده را در مسیر زیر قرار دهید:</p>
<pre>xAiApi/
└── tessdata/
├── fas.traineddata
├── eng.traineddata
└── ara.traineddata (اختیاری)</pre>
<div class="alert alert-warning">
<strong>⚠️ نکته مهم:</strong> مسیر <code>tessdata</code> باید در زمان اجرا در دسترس باشد. می‌توانید آن را در <code>appsettings.json</code> پیکربندی کنید.
</div>
</div>
<!-- Section 3: Configuration -->
<div class="section" id="step3">
<h2>⚙️ گام ۳: پیکربندی در appsettings.json</h2>
<p>تنظیمات Tesseract را در فایل <code>appsettings.json</code> اضافه کنید:</p>
<pre>{
"AiApiConfiguration": {
"FileExtraction": {
"OCR": {
"TessDataPath": "tessdata",
"DefaultLanguage": "fas+eng",
"EngineMode": "LstmOnly"
}
},
"Models": [ ... ],
"Prompts": [ ... ]
}
}</pre>
<h3>توضیح پارامترها:</h3>
<table>
<tr>
<th>پارامتر</th>
<th>توضیح</th>
<th>مقادیر مجاز</th>
</tr>
<tr>
<td><code>TessDataPath</code></td>
<td>مسیر پوشه فایل‌های traineddata</td>
<td>مسیر نسبی یا مطلق</td>
</tr>
<tr>
<td><code>DefaultLanguage</code></td>
<td>زبان‌های پیش‌فرض برای OCR</td>
<td>fas, eng, ara (ترکیب با +)</td>
</tr>
<tr>
<td><code>EngineMode</code></td>
<td>حالت موتور Tesseract</td>
<td>TesseractOnly, LstmOnly, TesseractAndLstm, Default</td>
</tr>
</table>
</div>
<!-- Section 4: Rewrite Class -->
<div class="section" id="step4">
<h2>🔄 گام ۴: بازنویسی کلاس XImageFileContentExtractor</h2>
<p>اکنون کلاس <code>XImageFileContentExtractor</code> را بازنویسی می‌کنیم تا از Tesseract برای OCR استفاده کند:</p>
<div class="file-change">
<span class="badge-modify">MODIFY</span>
<span class="path">xAiApi/Providers/Extractors/XImageFileContentExtractor.cs</span>
</div>
<pre>using System;
using System.IO;
using System.Linq;
using System.Threading;
using Tesseract;
using xAiModels.Models;
using System.Threading.Tasks;
using Microsoft.Extensions.Logging;
using xAiApi.Interfaces.Extractors;
namespace xAiApi.Providers.Extractors
{
/// &lt;summary&gt;
/// Extracts content from image files using Tesseract OCR ...
/// &lt;/summary&gt;
public class XImageFileContentExtractor : IXImageFileContentExtractor
{
/// &lt;summary&gt;
/// Supported MIME Types ...
/// &lt;/summary&gt;
private static readonly string[] SupportedMimeTypes =
[
"image/png",
"image/jpeg",
"image/jpg",
"image/gif",
"image/webp",
"image/bmp"
];
/// &lt;summary&gt;
/// مسیر پوشه tessdata ...
/// &lt;/summary&gt;
private readonly string _tessDataPath;
/// &lt;summary&gt;
/// زبان پیش‌فرض برای OCR ...
/// &lt;/summary&gt;
private readonly string _defaultLanguage;
/// &lt;summary&gt;
/// حالت موتور Tesseract ...
/// &lt;/summary&gt;
private readonly OcrEngineMode _engineMode;
/// &lt;summary&gt;
/// Logger ...
/// &lt;/summary&gt;
private readonly ILogger&lt;XImageFileContentExtractor&gt; _logger;
/// &lt;summary&gt;
/// Constructor با پیکربندی پیش‌فرض ...
/// &lt;/summary&gt;
public XImageFileContentExtractor(
ILogger&lt;XImageFileContentExtractor&gt; logger,
string tessDataPath = "tessdata",
string defaultLanguage = "fas+eng",
OcrEngineMode engineMode = OcrEngineMode.LstmOnly
)
{
_logger = logger;
_tessDataPath = tessDataPath;
_defaultLanguage = defaultLanguage;
_engineMode = engineMode;
}
/// &lt;summary&gt;
/// Check if this extractor supports the specified MIME type ...
/// &lt;/summary&gt;
public bool CanExtract(string mimeType)
{
return SupportedMimeTypes.Contains(
mimeType?.ToLowerInvariant() ?? string.Empty
);
}
/// &lt;summary&gt;
/// Extract text content from file stream ...
/// &lt;/summary&gt;
public async Task&lt;string&gt; ExtractAsync(
Stream fileStream,
string mimeType,
CancellationToken cancellationToken = default
)
{
var result = await ExtractRichAsync(
fileStream,
"image",
mimeType,
cancellationToken
);
return result.Text;
}
/// &lt;summary&gt;
/// Extract content from stream as Rich Result ...
/// &lt;/summary&gt;
public async Task&lt;XFileExtractionResult&gt; ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
CancellationToken cancellationToken = default
)
{
var result = new XFileExtractionResult
{
FileName = fileName,
MimeType = mimeType
};
// خواندن بایت‌های تصویر
using var memoryStream = new MemoryStream();
await fileStream.CopyToAsync(memoryStream, cancellationToken);
var imageBytes = memoryStream.ToArray();
// افزودن تصویر به عنوان DataContent (برای Vision Models)
result.Images.Add(new XExtractedImage
{
Bytes = imageBytes,
MimeType = mimeType,
Description = $"Attached image: {fileName}"
});
// انجام OCR برای استخراج متن
try
{
var ocrText = await PerformOcrAsync(imageBytes, cancellationToken);
if (!string.IsNullOrWhiteSpace(ocrText))
{
result.Text = ocrText;
_logger.LogInformation(
"OCR completed successfully for file: {FileName}, extracted {Length} characters",
fileName,
ocrText.Length
);
}
else
{
_logger.LogWarning("OCR completed but no text was extracted from file: {FileName}", fileName);
}
}
catch (Exception ex)
{
_logger.LogError(ex, "OCR failed for file: {FileName}", fileName);
result.ErrorMessage = $"OCR failed: {ex.Message}";
// تصویر همچنان برای Vision Models در دسترس است
}
return result;
}
/// &lt;summary&gt;
/// انجام OCR روی تصویر با استفاده از Tesseract ...
/// &lt;/summary&gt;
private async Task&lt;string&gt; PerformOcrAsync(
byte[] imageBytes,
CancellationToken cancellationToken
)
{
return await Task.Run(() =&gt;
{
// بررسی وجود مسیر tessdata
if (!Directory.Exists(_tessDataPath))
{
throw new DirectoryNotFoundException(
$"TessData directory not found at: {_tessDataPath}. " +
"Please download traineddata files from https://github.com/tesseract-ocr/tessdata"
);
}
// ایجاد Tesseract Engine
using var engine = new TesseractEngine(
datapath: _tessDataPath,
language: _defaultLanguage,
mode: _engineMode
);
// بارگذاری تصویر از byte array
using var pix = Pix.LoadFromMemory(imageBytes);
// تنظیم CancellationToken
cancellationToken.ThrowIfCancellationRequested();
// انجام OCR
using var page = engine.Process(pix);
// استخراج متن
var text = page.GetText();
// پاکسازی متن (حذف فاصله‌های اضافی)
text = text?.Trim();
return text ?? string.Empty;
}, cancellationToken);
}
}
}</pre>
</div>
<!-- Section 5: Implementation Details -->
<div class="section" id="step5">
<h2>🔧 گام ۵: جزئیات پیاده‌سازی متد PerformOcrAsync</h2>
<h3>۵.۱. مراحل کار متد:</h3>
<div class="arch-grid">
<div class="arch-card">
<h4>۱. بررسی مسیر tessdata</h4>
<ul>
<li>اطمینان از وجود پوشه tessdata</li>
<li>پرتاب خطا در صورت عدم وجود</li>
</ul>
</div>
<div class="arch-card">
<h4>۲. ایجاد Tesseract Engine</h4>
<ul>
<li>بارگذاری مدل زبانی</li>
<li>تنظیم حالت موتور (LSTM)</li>
</ul>
</div>
<div class="arch-card">
<h4>۳. بارگذاری تصویر</h4>
<ul>
<li>تبدیل byte[] به Pix object</li>
<li>پشتیبانی از فرمت‌های مختلف</li>
</ul>
</div>
<div class="arch-card">
<h4>۴. انجام OCR</h4>
<ul>
<li>پردازش تصویر</li>
<li>استخراج متن</li>
</ul>
</div>
<div class="arch-card">
<h4>۵. پاکسازی متن</h4>
<ul>
<li>حذف فاصله‌های اضافی</li>
<li>بررسی CancellationToken</li>
</ul>
</div>
</div>
<h3>۵.۲. ویژگی‌های کلیدی پیاده‌سازی:</h3>
<table>
<tr>
<th>ویژگی</th>
<th>توضیح</th>
</tr>
<tr>
<td><strong>پشتیبانی چند زبانه</strong></td>
<td>استفاده از <code>fas+eng</code> برای تشخیص همزمان فارسی و انگلیسی</td>
</tr>
<tr>
<td><strong>حالت LSTM</strong></td>
<td>استفاده از موتور LSTM برای دقت بالاتر در متون فارسی</td>
</tr>
<tr>
<td><strong>مدیریت خطا</strong></td>
<td>بررسی وجود tessdata و مدیریت استثناها</td>
</tr>
<tr>
<td><strong>Logging</strong></td>
<td>ثبت موفقیت/شکست OCR با جزئیات</td>
</tr>
<tr>
<td><strong>Cancellation Support</strong></td>
<td>پشتیبانی از لغو عملیات در میانه کار</td>
</tr>
<tr>
<td><strong>Resource Management</strong></td>
<td>استفاده از <code>using</code> برای آزادسازی منابع</td>
</tr>
</table>
</div>
<!-- Section 6: Tips -->
<div class="section" id="step6">
<h2>💡 گام ۶: نکات مهم و عیب‌یابی</h2>
<h3>۶.۱. مشکلات رایج و راه‌حل‌ها:</h3>
<table>
<tr>
<th>مشکل</th>
<th>علت</th>
<th>راه‌حل</th>
</tr>
<tr>
<td><code>DirectoryNotFoundException</code></td>
<td>پوشه tessdata یافت نشد</td>
<td>دانلود traineddata و قرار دادن در مسیر صحیح</td>
</tr>
<tr>
<td><code>TesseractException</code></td>
<td>فایل traineddata خراب یا ناسازگار</td>
<td>دانلود مجدد از منبع رسمی</td>
</tr>
<tr>
<td>دقت پایین OCR</td>
<td>کیفیت پایین تصویر</td>
<td>پیش‌پردازش تصویر (افزایش کنتراست، resize)</td>
</tr>
<tr>
<td>تشخیص نادرست فارسی</td>
<td>ترکیب نامناسب زبان‌ها</td>
<td>استفاده از <code>fas</code> به تنهایی یا <code>fas+eng</code></td>
</tr>
<tr>
<td>کندی عملکرد</td>
<td>تصاویر بزرگ</td>
<td>تغییر EngineMode به <code>TesseractOnly</code></td>
</tr>
</table>
<h3>۶.۲. بهینه‌سازی عملکرد:</h3>
<div class="arch-grid">
<div class="arch-card">
<h4>🚀 افزایش سرعت</h4>
<ul>
<li>استفاده از <code>OcrEngineMode.TesseractOnly</code></li>
<li>کاهش DPI تصویر (مثلاً 150 به جای 300)</li>
<li>استفاده از یک زبان به جای چند زبان</li>
</ul>
</div>
<div class="arch-card">
<h4>🎯 افزایش دقت</h4>
<ul>
<li>استفاده از <code>OcrEngineMode.LstmOnly</code></li>
<li>افزایش DPI تصویر (300 یا بالاتر)</li>
<li>پیش‌پردازش تصویر (binarization)</li>
</ul>
</div>
</div>
<h3>۶.۳. پیش‌پردازش تصویر (اختیاری):</h3>
<p>برای افزایش دقت OCR، می‌توانید قبل از پردازش، تصویر را پیش‌پردازش کنید:</p>
<pre>// مثال: افزایش کنتراست و تبدیل به سیاه و سفید
using var pix = Pix.LoadFromMemory(imageBytes);
using var enhanced = pix.ConvertTo1(); // تبدیل به 1-bit
using var page = engine.Process(enhanced);</pre>
<h3>۶.۴. ثبت در Startup.cs:</h3>
<p>اکنون <code>XImageFileContentExtractor</code> را با پیکربندی صحیح ثبت کنید:</p>
<pre>public void ConfigureServices(IServiceCollection services)
{
// ... سایر ثبت‌ها
// ثبت XImageFileContentExtractor با پیکربندی
services.AddSingleton&lt;IXFileContentExtractor&gt;(sp =&gt;
{
var logger = sp.GetRequiredService&lt;ILogger&lt;XImageFileContentExtractor&gt;&gt;();
var configuration = sp.GetRequiredService&lt;IConfiguration&gt;();
var tessDataPath = configuration.GetValue&lt;string&gt;(
"AiApiConfiguration:FileExtraction:OCR:TessDataPath"
) ?? "tessdata";
var defaultLanguage = configuration.GetValue&lt;string&gt;(
"AiApiConfiguration:FileExtraction:OCR:DefaultLanguage"
) ?? "fas+eng";
return new XImageFileContentExtractor(
logger: logger,
tessDataPath: tessDataPath,
defaultLanguage: defaultLanguage,
engineMode: OcrEngineMode.LstmOnly
);
});
// ... سایر ثبت‌ها
}</pre>
<div class="alert alert-success">
<strong>✅ نتیجه نهایی:</strong>
<ul style="padding-right: 25px; margin-top: 10px;">
<li>🔍 <strong>OCR کامل:</strong> استخراج متن از تصاویر با دقت بالا</li>
<li>🌐 <strong>چند زبانه:</strong> پشتیبانی از فارسی، انگلیسی و عربی</li>
<li>⚡ <strong>بهینه:</strong> استفاده از موتور LSTM برای دقت بالاتر</li>
<li>🛡️ <strong>مقاوم:</strong> مدیریت خطا و logging کامل</li>
<li>🔄 <strong>قابل لغو:</strong> پشتیبانی از CancellationToken</li>
<li>📦 <strong>منعطف:</strong> پیکربندی از طریق appsettings.json</li>
</ul>
</div>
<div class="alert alert-info">
<strong>💡 نکته تکمیلی:</strong>
<p>اگر نیاز به دقت بالاتر برای متون فارسی دارید، می‌توانید از مدل‌های آموزش‌دیده‌شده خاص فارسی استفاده کنید یا از ترکیب Tesseract با مدل‌های deep learning (مانند CRNN) بهره ببرید.</p>
</div>
</div>
</div>
<div class="footer">
<p><strong>👨‍💻 توسعه‌دهنده:</strong> هادی خزاعی اصل</p>
<p><strong>🏢 شرکت:</strong> فن آوران ساحر علم</p>
<p><strong>📅 تاریخ:</strong> شنبه ۱۲ مهر ۱۴۰۵</p>
<p style="margin-top: 15px; opacity: 0.8; font-size: 0.9em;">
🔍 مستند فنی پیاده‌سازی OCR با Tesseract - تمامی حقوق محفوظ است
</p>
</div>
</div>
</body>
</html>