| | |
| | | set -u |
| | | set -o pipefail |
| | | |
| | | set=L |
| | | train_set=train_l |
| | | valid_set=dev |
| | | test_sets="dev test_net test_meeting" |
| | |
| | | fi |
| | | |
| | | if [ ${stage} -le 0 ] && [ ${stop_stage} -ge 0 ]; then |
| | | # echo "stage 0: Data preparation" |
| | | # # Data preparation |
| | | # local/wenetspeech_data_prep.sh $raw_data $feats_dir |
| | | echo "stage 0: Data preparation" |
| | | # Data preparation |
| | | # local/data.sh --set ${set} --nj $nj --data_dir $feats_dir --WENETSPEECH $raw_data --train_cmd $train_cmd |
| | | mkdir $feats_dir/data |
| | | mv $feats_dir/$train_set $feats_dir/data/$train_set |
| | | for x in $test_sets; do |
| | | mv mv $feats_dir/$x $feats_dir/data/ |
| | | done |
| | | fi |
| | | mv $feats_dir/$x $feats_dir/data/ |
| | | done |
| | | fi |
| | | |
| | | if [ ${stage} -le 1 ] && [ ${stop_stage} -ge 1 ]; then |
| | | echo "stage 1: Feature and CMVN Generation" |
| | | utils/compute_cmvn.sh --fbankdir ${feats_dir}/data/${train_set} --cmd "$train_cmd" --nj $nj --feats_dim ${feats_dim} --config_file "$asr_config" --scale 0.1 |
| | | fi |
| | | |
| | | token_list=${feats_dir}/data/${lang}_token_list/$token_type/tokens.txt |
| | | echo "dictionary: ${token_list}" |
| | | if [ ${stage} -le 2 ] && [ ${stop_stage} -ge 2 ]; then |
| | | echo "stage 2: Dictionary Preparation" |
| | | mkdir -p ${feats_dir}/data/${lang}_token_list/$token_type/ |
| | | |
| | | echo "make a dictionary" |
| | | echo "<blank>" > ${token_list} |
| | | echo "<s>" >> ${token_list} |
| | | echo "</s>" >> ${token_list} |
| | | utils/text2token.py -s 1 -n 1 --space "" ${feats_dir}/data/$train_set/text | cut -f 2- -d" " | tr " " "\n" \ |
| | | | sort | uniq | grep -a -v -e '^\s*$' | awk '{print $0}' >> ${token_list} |
| | | echo "<unk>" >> ${token_list} |
| | | fi |