📑 فهرست مطالب
🔍 گام ۱: تحلیل وضعیت فعلی
با بررسی آخرین نسخه کد XAudioFileContentExtractor، موارد زیر شناسایی شد:
| بخش | وضعیت | توضیح |
|---|---|---|
SupportedMimeTypes |
✅ کامل | پشتیبانی از MP3, WAV, OGG, M4A, WebM |
CanExtract() |
✅ کامل | تشخیص صحیح MIME Type |
ExtractAsync() |
✅ کامل | Delegate به ExtractRichAsync |
ExtractRichAsync() |
⚠️ نیمهکاره | ساختار اصلی وجود دارد ولی نیاز به تکمیل دارد |
ConvertToWavAsync() |
❌ خالی (TODO) | فقط placeholder است و باید پیادهسازی شود |
ILogger |
❌ موجود نیست | نیاز به تزریق Logger برای ثبت وقایع |
📦 گام ۲: نصب پکیجهای NuGet
# پکیج اصلی Whisper برای Speech-to-Text Install-Package Whisper.net # Runtime مورد نیاز Whisper (برای CPU) Install-Package Whisper.net.Runtime # کتابخانه NAudio برای تبدیل فرمت صوتی Install-Package NAudio
💡 نکته: اگر از GPU NVIDIA استفاده میکنید، به جای
Whisper.net.Runtime پکیج Whisper.net.Runtime.Cuda را نصب کنید تا سرعت پردازش چندین برابر شود.
📥 گام ۳: دانلود مدل Whisper
مدلهای Whisper در اندازههای مختلف موجود هستند. برای زبان فارسی، مدل base یا small پیشنهاد میشود:
| مدل | حجم | دقت فارسی | سرعت | VRAM مورد نیاز |
|---|---|---|---|---|
ggml-tiny.bin |
~75 MB | ⭐⭐ | ⭐⭐⭐⭐⭐ | ~1 GB |
ggml-base.bin |
~142 MB | ⭐⭐⭐ | ⭐⭐⭐⭐ | ~1 GB |
ggml-small.bin |
~466 MB | ⭐⭐⭐⭐ | ⭐⭐⭐ | ~2 GB |
ggml-medium.bin |
~1.5 GB | ⭐⭐⭐⭐⭐ | ⭐⭐ | ~5 GB |
دانلود از:
# لینک مستقیم دانلود مدل base: https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin # قرار دادن در مسیر: xAiApi/Models/ggml-base.bin
⚙️ گام ۴: پیکربندی appsettings.json
{
"AiApiConfiguration": {
"FileExtraction": {
"Audio": {
"WhisperModelPath": "Models/ggml-base.bin",
"DefaultLanguage": "fa"
}
}
}
}
🛠️ گام ۵: کد کامل XAudioFileContentExtractor
MODIFY
xAiApi/Providers/Extractors/XAudioFileContentExtractor.cs
using System;
using System.IO;
using System.Linq;
using System.Text;
using NAudio.Wave;
using NAudio.Wave.SampleProviders;
using Whisper.net;
using System.Threading;
using xAiModels.Models;
using System.Threading.Tasks;
using Microsoft.Extensions.Logging;
using xAiApi.Interfaces.Extractors;
namespace xAiApi.Providers.Extractors
{
/// <summary>
/// Extracts text content from Audio files using Whisper ...
/// </summary>
public class XAudioFileContentExtractor : IXAudioFileContentExtractor
{
/// <summary>
/// Supported MIME Types ...
/// </summary>
private static readonly string[] SupportedMimeTypes =
[
"audio/mpeg",
"audio/mp3",
"audio/wav",
"audio/wave",
"audio/ogg",
"audio/m4a",
"audio/mp4",
"audio/webm"
];
/// <summary>
/// Language for Whisper STT ...
/// </summary>
private readonly string language;
/// <summary>
/// Path to Whisper GGML model file ...
/// </summary>
private readonly string whisperModelPath;
/// <summary>
/// Logger ...
/// </summary>
private readonly ILogger<XAudioFileContentExtractor> logger;
/// <summary>
/// Default Constructor ...
/// </summary>
public XAudioFileContentExtractor()
: this(
logger: null,
whisperModelPath: "Models/ggml-base.bin",
language: "fa"
)
{ }
/// <summary>
/// Constructor with configuration ...
/// </summary>
public XAudioFileContentExtractor(
ILogger<XAudioFileContentExtractor> logger,
string whisperModelPath = "Models/ggml-base.bin",
string language = "fa"
)
{
this.logger = logger;
this.language = language;
this.whisperModelPath = whisperModelPath;
}
/// <summary>
/// Check if this extractor supports the specified MIME type ...
/// </summary>
public bool CanExtract(string mimeType)
{
return SupportedMimeTypes.Contains(
mimeType?.ToLowerInvariant() ?? string.Empty
);
}
/// <summary>
/// Extract text content from file stream ...
/// </summary>
public async Task<string> ExtractAsync(
Stream fileStream,
string mimeType,
CancellationToken cancellationToken = default
)
{
var result = await ExtractRichAsync(
fileStream,
"audio",
mimeType,
cancellationToken
);
return result.Text;
}
/// <summary>
/// Extract content from stream as Rich Result ...
/// </summary>
public async Task<XFileExtractionResult> ExtractRichAsync(
Stream fileStream,
string fileName,
string mimeType,
CancellationToken cancellationToken = default
)
{
var result = new XFileExtractionResult
{
FileName = fileName,
MimeType = mimeType
};
try
{
// ۱. کپی Stream ورودی به MemoryStream
using var memoryStream = new MemoryStream();
await fileStream.CopyToAsync(memoryStream, cancellationToken);
memoryStream.Position = 0;
// ۲. تبدیل فرمت صوتی به WAV 16kHz Mono (مورد نیاز Whisper)
using var wavStream = await ConvertToWavAsync(
inputStream: memoryStream,
mimeType: mimeType,
cancellationToken: cancellationToken
);
// ۳. بررسی وجود فایل مدل Whisper
if (!File.Exists(whisperModelPath))
{
throw new FileNotFoundException(
$"Whisper model not found at: {whisperModelPath}. " +
"Please download from https://huggingface.co/ggerganov/whisper.cpp/tree/main"
);
}
// ۴. انجام Speech-to-Text با Whisper
using var factory = WhisperFactory.FromPath(whisperModelPath);
using var processor = factory.CreateBuilder()
.WithLanguage(language)
.Build();
var segments = new StringBuilder();
await foreach (var segment in processor.ProcessAsync(wavStream, cancellationToken))
{
cancellationToken.ThrowIfCancellationRequested();
segments.Append(segment.Text);
}
// ۵. تنظیم نتیجه
result.AudioTranscript = segments.ToString().Trim();
result.Text = result.AudioTranscript;
logger?.LogInformation(
"Audio extraction completed for {FileName}: {Length} characters extracted",
fileName,
result.Text.Length
);
}
catch (OperationCanceledException)
{
logger?.LogWarning("Audio extraction cancelled for file: {FileName}", fileName);
throw;
}
catch (Exception ex)
{
logger?.LogError(ex, "Audio extraction failed for file: {FileName}", fileName);
result.ErrorMessage = $"Audio extraction failed: {ex.Message}";
}
return result;
}
/// <summary>
/// تبدیل فرمت صوتی به WAV 16kHz 16-bit Mono ...
/// فرمت استاندارد مورد نیاز موتور Whisper
/// </summary>
private async Task<Stream> ConvertToWavAsync(
Stream inputStream,
string mimeType,
CancellationToken cancellationToken
)
{
return await Task.Run(() =>
{
WaveStream reader = null;
try
{
inputStream.Position = 0;
// انتخاب Reader مناسب بر اساس MIME Type
reader = CreateWaveReader(inputStream, mimeType);
// فرمت هدف: 16kHz, 16-bit, Mono
var targetFormat = new WaveFormat(16000, 16, 1);
// تبدیل به SampleProvider برای پردازش
ISampleProvider sampleProvider = reader.ToSampleProvider();
// تبدیل به Mono اگر چند کاناله باشد
if (reader.WaveFormat.Channels > 1)
{
sampleProvider = sampleProvider.ToMono();
}
// Resample به 16kHz اگر نرخ نمونهبرداری متفاوت باشد
if (reader.WaveFormat.SampleRate != 16000)
{
sampleProvider = new WdlResamplingSampleProvider(
sampleProvider,
16000
);
}
// تبدیل به WaveProvider 16-bit
var waveProvider = sampleProvider.ToWaveProvider16();
// نوشتن به MemoryStream موقت
var tempStream = new MemoryStream();
using (var writer = new WaveFileWriter(tempStream, targetFormat))
{
var buffer = new byte[4096];
int bytesRead;
while ((bytesRead = waveProvider.Read(buffer, 0, buffer.Length)) > 0)
{
writer.Write(buffer, 0, bytesRead);
}
}
// ساخت MemoryStream جدید از دادههای نهایی
// (چون WaveFileWriter پس از Dispose، stream را میبندد)
var resultStream = new MemoryStream(tempStream.ToArray());
tempStream.Dispose();
return resultStream;
}
finally
{
reader?.Dispose();
}
}, cancellationToken);
}
/// <summary>
/// ایجاد WaveStream Reader مناسب بر اساس نوع فایل صوتی ...
/// </summary>
private WaveStream CreateWaveReader(Stream stream, string mimeType)
{
return mimeType?.ToLowerInvariant() switch
{
"audio/mpeg" or "audio/mp3"
=> new Mp3FileReader(stream),
"audio/wav" or "audio/wave"
=> new WaveFileReader(stream),
// برای سایر فرمتها، ابتدا به عنوان MP3 تلاش میکنیم
// در صورت خطا، باید از FFmpeg استفاده شود
_ => new Mp3FileReader(stream)
};
}
}
}
🔌 گام ۶: ثبت در Startup.cs
MODIFY
xAiApi/Startup.cs
public void ConfigureServices(IServiceCollection services)
{
// ... (سایر ثبتها)
// ✅ ثبت XAudioFileContentExtractor با پیکربندی
services.AddSingleton<IXFileContentExtractorBase>(sp =>
{
var logger = sp.GetRequiredService<ILogger<XAudioFileContentExtractor>>();
var configuration = sp.GetRequiredService<IConfiguration>();
var modelPath = configuration.GetValue<string>(
"AiApiConfiguration:FileExtraction:Audio:WhisperModelPath"
) ?? "Models/ggml-base.bin";
var language = configuration.GetValue<string>(
"AiApiConfiguration:FileExtraction:Audio:DefaultLanguage"
) ?? "fa";
return new XAudioFileContentExtractor(
logger: logger,
whisperModelPath: modelPath,
language: language
);
});
// ... (سایر ثبتها)
}
💡 گام ۷: نکات و عیبیابی
۷.۱. مشکلات رایج:
| مشکل | علت | راهحل |
|---|---|---|
FileNotFoundException |
فایل مدل Whisper یافت نشد | دانلود مدل و قرار دادن در مسیر Models/ggml-base.bin |
InvalidOperationException در NAudio |
فرمت صوتی پشتیبانی نمیشود | نصب MediaFoundation یا استفاده از FFmpeg برای فرمتهای OGG/M4A |
| دقت پایین STT فارسی | مدل base برای فارسی محدود است | استفاده از مدل ggml-small.bin یا ggml-medium.bin |
| کندی پردازش | اجرا روی CPU | نصب Whisper.net.Runtime.Cuda برای GPU |
OutOfMemoryException |
فایل صوتی بسیار بزرگ | محدود کردن حجم فایل ورودی یا تقسیم به بخشهای کوچکتر |
۷.۲. پشتیبانی از فرمتهای OGG و M4A:
کتابخانه NAudio به صورت پیشفرض از فرمتهای OGG و M4A پشتیبانی نمیکند. برای این فرمتها دو راهحل وجود دارد:
راهحل ۱: استفاده از FFmpeg
- نصب FFmpeg روی سرور
- تبدیل فرمت با Process.Start
- پشتیبانی از تمام فرمتها
راهحل ۲: NAudio + MediaFoundation
- فقط در Windows کار میکند
- پشتیبانی از M4A و برخی فرمتها
- بدون نیاز به نصب اضافی
۷.۳. ساختار پوشه نهایی:
xAiApi/
├── Models/
│ └── ggml-base.bin ← مدل Whisper
├── tessdata/
│ ├── fas.traineddata ← مدل Tesseract فارسی
│ └── eng.traineddata ← مدل Tesseract انگلیسی
├── appsettings.json ← پیکربندی
└── Providers/
└── Extractors/
├── XAudioFileContentExtractor.cs ← ✅ تکمیل شده
├── XImageFileContentExtractor.cs ← ✅ تکمیل شده
├── XPdfFileContentExtractor.cs
├── XDocxFileContentExtractor.cs
├── XExcelFileContentExtractor.cs
├── XPlainTextFileContentExtractor.cs
├── XVisionFileContentExtractor.cs
└── XFileContentExtractor.cs
✅ خلاصه تغییرات:
- 🎙️ STT کامل: تبدیل صوت به متن با Whisper
- 🔄 تبدیل فرمت: تبدیل خودکار MP3/WAV به 16kHz Mono با NAudio
- 🌐 پشتیبانی فارسی: زبان پیشفرض فارسی
- 🛡️ مدیریت خطا: بررسی وجود مدل و مدیریت استثناها
- 📝 Logging: ثبت موفقیت/شکست با جزئیات
- ⚡ بهینه: اجرای تبدیل فرمت در Thread جداگانه
- 🔄 قابل لغو: پشتیبانی از CancellationToken