python/FunASR-XL.git

			@@ -3,7 +3,7 @@
			. ./path.sh \|\| exit 1;

			# machines configuration
			CUDA_VISIBLE_DEVICES="0,1"
			CUDA_VISIBLE_DEVICES="2,3"
			gpu_num=2
			count=1
			gpu_inference=true # Whether to perform gpu decoding, set false for cpu decoding
			@@ -22,20 +22,18 @@
			scp=wav.scp
			type=sound
			stage=1
			stop_stage=1
			stop_stage=3

			# feature configuration
			feats_dim=80
			sample_frequency=16000
			nj=64
			speed_perturb="0.9,1.0,1.1"

			# data
			raw_data=
			data_url=www.openslr.org/resources/33

			# exp tag
			tag=""
			tag="exp2"

			. utils/parse_options.sh \|\| exit 1;

			@@ -86,7 +84,6 @@
			done
			fi

			feat_train_dir=${feats_dir}/${dumpdir}/train; mkdir -p ${feat_train_dir}
			if [ ${stage} -le 1 ] && [ ${stop_stage} -ge 1 ]; then
			echo "stage 1: Feature and CMVN Generation"
			utils/compute_cmvn.sh --cmd "$train_cmd" --nj $nj --feats_dim ${feats_dim} ${feats_dir}/data/${train_set}
			@@ -102,11 +99,9 @@
			echo "<blank>" > ${token_list}
			echo "<s>" >> ${token_list}
			echo "</s>" >> ${token_list}
			utils/text2token.py -s 1 -n 1 --space "" ${feats_dir}/data/train/text \| cut -f 2- -d" " \| tr " " "\n" \
			utils/text2token.py -s 1 -n 1 --space "" ${feats_dir}/data/$train_set/text \| cut -f 2- -d" " \| tr " " "\n" \
			\| sort \| uniq \| grep -a -v -e '^\s*$' \| awk '{print $0}' >> ${token_list}
			num_token=$(cat ${token_list} \| wc -l)
			echo "<unk>" >> ${token_list}
			vocab_size=$(cat ${token_list} \| wc -l)
			fi

			# Training Stage
			@@ -135,7 +130,7 @@
			--data_dir ${feats_dir}/data \
			--train_set ${train_set} \
			--valid_set ${valid_set} \
			--cmvn_file ${feats_dir}/cmvn/cmvn.mvn \
			--cmvn_file ${feats_dir}/data/${train_set}/cmvn/cmvn.mvn \
			--resume true \
			--output_dir ${exp_dir}/exp/${model_dir} \
			--config $asr_config \