From e9d2cfc3a134b00f4e98271fbee3838d1ccecbcc Mon Sep 17 00:00:00 2001
From: VirtuosoQ <2416050435@qq.com>
Date: 星期五, 26 四月 2024 14:59:30 +0800
Subject: [PATCH] FunASR java http  client

---
 funasr/tokenizer/whisper_tokenizer.py |   22 ++++++++++++++++++++++
 1 files changed, 22 insertions(+), 0 deletions(-)

diff --git a/funasr/tokenizer/whisper_tokenizer.py b/funasr/tokenizer/whisper_tokenizer.py
index 6684f25..0a34d19 100644
--- a/funasr/tokenizer/whisper_tokenizer.py
+++ b/funasr/tokenizer/whisper_tokenizer.py
@@ -22,3 +22,25 @@
 	
 	return tokenizer
 
+
+@tables.register("tokenizer_classes", "SenseVoiceTokenizer")
+def SenseVoiceTokenizer(**kwargs):
+	try:
+		from funasr.models.sense_voice.whisper_lib.tokenizer import get_tokenizer
+	except:
+		print("Notice: If you want to use whisper, please `pip install -U openai-whisper`")
+	
+	language = kwargs.get("language", None)
+	task = kwargs.get("task", None)
+	is_multilingual = kwargs.get("is_multilingual", True)
+	num_languages = kwargs.get("num_languages", 8749)
+	vocab_path = kwargs.get("vocab_path", None)
+	tokenizer = get_tokenizer(
+		multilingual=is_multilingual,
+		num_languages=num_languages,
+		language=language,
+		task=task,
+		vocab_path=vocab_path,
+	)
+	
+	return tokenizer

--
Gitblit v1.9.1