From e9d2cfc3a134b00f4e98271fbee3838d1ccecbcc Mon Sep 17 00:00:00 2001
From: VirtuosoQ <2416050435@qq.com>
Date: 星期五, 26 四月 2024 14:59:30 +0800
Subject: [PATCH] FunASR java http client
---
funasr/tokenizer/char_tokenizer.py | 4 +++-
1 files changed, 3 insertions(+), 1 deletions(-)
diff --git a/funasr/tokenizer/char_tokenizer.py b/funasr/tokenizer/char_tokenizer.py
index 28b4f40..92c6e67 100644
--- a/funasr/tokenizer/char_tokenizer.py
+++ b/funasr/tokenizer/char_tokenizer.py
@@ -36,6 +36,7 @@
self.remove_non_linguistic_symbols = remove_non_linguistic_symbols
self.split_with_space = split_with_space
self.seg_dict = None
+ seg_dict = seg_dict if seg_dict is not None else kwargs.get("seg_dict_file", None)
if seg_dict is not None:
self.seg_dict = load_seg_dict(seg_dict)
@@ -92,7 +93,8 @@
return seg_dict
def seg_tokenize(txt, seg_dict):
- pattern = re.compile(r'^[\u4E00-\u9FA50-9]+$')
+ # pattern = re.compile(r'^[\u4E00-\u9FA50-9]+$')
+ pattern = re.compile(r"([\u4E00-\u9FA5A-Za-z0-9])")
out_txt = ""
for word in txt:
word = word.lower()
--
Gitblit v1.9.1