From fce4e1d1b48f23cd8332e60afce3df8d6209a6a7 Mon Sep 17 00:00:00 2001
From: gaochangfeng <54253717+gaochangfeng@users.noreply.github.com>
Date: 星期四, 11 四月 2024 14:59:22 +0800
Subject: [PATCH] SenseVoice对富文本解码的参数 (#1608)
---
funasr/tokenizer/char_tokenizer.py | 12 ++++++++----
1 files changed, 8 insertions(+), 4 deletions(-)
diff --git a/funasr/tokenizer/char_tokenizer.py b/funasr/tokenizer/char_tokenizer.py
index 0f40b5e..2efc0b0 100644
--- a/funasr/tokenizer/char_tokenizer.py
+++ b/funasr/tokenizer/char_tokenizer.py
@@ -36,6 +36,7 @@
self.remove_non_linguistic_symbols = remove_non_linguistic_symbols
self.split_with_space = split_with_space
self.seg_dict = None
+ seg_dict = seg_dict if seg_dict is not None else kwargs.get("seg_dict_file", None)
if seg_dict is not None:
self.seg_dict = load_seg_dict(seg_dict)
@@ -50,10 +51,11 @@
def text2tokens(self, line: Union[str, list]) -> List[str]:
- if self.split_with_space:
+ # if self.split_with_space:
+
+ if self.seg_dict is not None:
tokens = line.strip().split(" ")
- if self.seg_dict is not None:
- tokens = seg_tokenize(tokens, self.seg_dict)
+ tokens = seg_tokenize(tokens, self.seg_dict)
else:
tokens = []
while len(line) != 0:
@@ -66,7 +68,9 @@
else:
t = line[0]
if t == " ":
- t = "<space>"
+ # t = "<space>"
+ line = line[1:]
+ continue
tokens.append(t)
line = line[1:]
return tokens
--
Gitblit v1.9.1