From eb1574b813e230b156fc09eaaf03227b1b0b4134 Mon Sep 17 00:00:00 2001
From: weilikai <jasper@talkus.fun>
Date: 星期六, 20 九月 2025 22:41:05 +0800
Subject: [PATCH] fix: support loading .pcm (16k 1c 16bit) audio files in load_utils.py (#2667) (#2668)

---
 funasr/models/whisper/model.py |    8 +++++---
 1 files changed, 5 insertions(+), 3 deletions(-)

diff --git a/funasr/models/whisper/model.py b/funasr/models/whisper/model.py
index a332100..398eea3 100644
--- a/funasr/models/whisper/model.py
+++ b/funasr/models/whisper/model.py
@@ -9,6 +9,7 @@
 from torch import nn
 
 import whisper
+
 # import whisper_timestamped as whisper
 
 from funasr.utils.load_utils import load_audio_text_image_video, extract_fbank
@@ -27,6 +28,7 @@
 @tables.register("model_classes", "Whisper-large-v1")
 @tables.register("model_classes", "Whisper-large-v2")
 @tables.register("model_classes", "Whisper-large-v3")
+@tables.register("model_classes", "Whisper-large-v3-turbo")
 @tables.register("model_classes", "WhisperWarp")
 class WhisperWarp(nn.Module):
     def __init__(self, *args, **kwargs):
@@ -111,10 +113,10 @@
 
         # decode the audio
         options = whisper.DecodingOptions(**kwargs.get("DecodingOptions", {}))
-        
-        result = whisper.decode(self.model, speech, language='english')
+
+        result = whisper.decode(self.model, speech, options=options)
         # result = whisper.transcribe(self.model, speech)
-        
+
         results = []
         result_i = {"key": key[0], "text": result.text}
 

--
Gitblit v1.9.1