From d43f77408b8f3e169c59dfb6b6d82e45e6b91714 Mon Sep 17 00:00:00 2001
From: 游雁 <zhifu.gzf@alibaba-inc.com>
Date: 星期二, 11 六月 2024 19:19:06 +0800
Subject: [PATCH] decoding

---
 examples/industrial_data_pretraining/llm_asr/demo_speech2text.sh |   42 +++++++++++++++++++++---------------------
 1 files changed, 21 insertions(+), 21 deletions(-)

diff --git a/examples/industrial_data_pretraining/llm_asr/demo_speech2text.sh b/examples/industrial_data_pretraining/llm_asr/demo_speech2text.sh
index bb63d44..57299fc 100644
--- a/examples/industrial_data_pretraining/llm_asr/demo_speech2text.sh
+++ b/examples/industrial_data_pretraining/llm_asr/demo_speech2text.sh
@@ -42,24 +42,24 @@
 }&
 done
 wait
-#for data_set in "s2tt_en2zh.v20240605.test.jsonl"; do
-#    jsonl=${jsonl_dir}/${data_set}
-#    output_dir=${out_dir}/${data_set}
-#    mkdir -p ${output_dir}
-#    pred_file=${output_dir}/1best_recog/text_tn
-#    ref_file=${output_dir}/1best_recog/label
-#
-#    python ./demo_speech2text.py ${ckpt_dir} ${ckpt_id} ${jsonl} ${output_dir} ${device}
-#
-#done
-#
-#for data_set in "s2tt_zh2en.v20240605.test.jsonl"; do
-#    jsonl=${jsonl_dir}/${data_set}
-#    output_dir=${out_dir}/${data_set}
-#    mkdir -p ${output_dir}
-#    pred_file=${output_dir}/1best_recog/text_tn
-#    ref_file=${output_dir}/1best_recog/label
-#
-#    python ./demo_speech2text.py ${ckpt_dir} ${ckpt_id} ${jsonl} ${output_dir} ${device}
-#
-#done
\ No newline at end of file
+
+for data_set in "common_voice_zh-CN_speech2text.jsonl" "common_voice_en_speech2text.jsonl"; do
+{
+    jsonl=${jsonl_dir}/${data_set}
+    output_dir=${out_dir}/${data_set}
+    mkdir -p ${output_dir}
+    pred_file=${output_dir}/1best_recog/text_tn
+    ref_file=${output_dir}/1best_recog/label
+
+    python ./demo_speech2text.py ${ckpt_dir} ${ckpt_id} ${jsonl} ${output_dir} ${device}
+
+    cn_postprocess=false
+    if [ $data_set = "common_voice_zh-CN_speech2text.jsonl" ];then
+      cn_postprocess=true
+    fi
+
+    python /mnt/workspace/zhifu.gzf/codebase/FunASR/funasr/metrics/wer.py ++ref_file=${ref_file} ++hyp_file=${pred_file} ++cer_file=${pred_file}.cer ++cn_postprocess=${cn_postprocess}
+
+}&
+done
+wait
\ No newline at end of file

--
Gitblit v1.9.1