From 3a71c7a227ebd8b61995e2fb8ee0c3c25a8c19c4 Mon Sep 17 00:00:00 2001
From: hnluo <haoneng.lhn@alibaba-inc.com>
Date: 星期二, 15 八月 2023 19:56:47 +0800
Subject: [PATCH] Merge pull request #857 from alibaba-damo-academy/dev_wjm_tmp
---
funasr/datasets/large_datasets/utils/tokenize.py | 2 +-
1 files changed, 1 insertions(+), 1 deletions(-)
diff --git a/funasr/datasets/large_datasets/utils/tokenize.py b/funasr/datasets/large_datasets/utils/tokenize.py
index c16e1dc..34a97c1 100644
--- a/funasr/datasets/large_datasets/utils/tokenize.py
+++ b/funasr/datasets/large_datasets/utils/tokenize.py
@@ -54,9 +54,9 @@
length = len(text)
if 'hw_tag' in data:
+ pre_index = None
if hw_config['pre_hwlist'] is not None and hw_config['pre_prob'] > 0:
# enable preset hotword detect in sampling
- pre_index = None
for hw in hw_config['pre_hwlist']:
hw = " ".join(seg_tokenize(hw, seg_dict))
_find = " ".join(text).find(hw)
--
Gitblit v1.9.1