From 9fcb3cc06b4e324f0913d2f61b89becc2baeef1b Mon Sep 17 00:00:00 2001
From: hnluo <haoneng.lhn@alibaba-inc.com>
Date: 星期一, 11 九月 2023 17:40:03 +0800
Subject: [PATCH] Merge pull request #932 from alibaba-damo-academy/dev_lhn
---
egs/aishell/rnnt/conf/train_conformer_rnnt_unified.yaml | 65 +++++++++++++++++++-------------
1 files changed, 39 insertions(+), 26 deletions(-)
diff --git a/egs/aishell/rnnt/conf/train_conformer_rnnt_unified.yaml b/egs/aishell/rnnt/conf/train_conformer_rnnt_unified.yaml
index 60f796c..a1f27a3 100644
--- a/egs/aishell/rnnt/conf/train_conformer_rnnt_unified.yaml
+++ b/egs/aishell/rnnt/conf/train_conformer_rnnt_unified.yaml
@@ -1,54 +1,57 @@
+encoder: chunk_conformer
encoder_conf:
- main_conf:
- pos_wise_act_type: swish
- pos_enc_dropout_rate: 0.5
- conv_mod_act_type: swish
+ activation_type: swish
+ positional_dropout_rate: 0.5
time_reduction_factor: 2
unified_model_training: true
default_chunk_size: 16
jitter_range: 4
- left_chunk_size: 0
- input_conf:
- block_type: conv2d
- conv_size: 512
+ left_chunk_size: 1
+ embed_vgg_like: false
subsampling_factor: 4
- num_frame: 1
- body_conf:
- - block_type: conformer
- linear_size: 2048
- hidden_size: 512
- heads: 8
+ linear_units: 2048
+ output_size: 512
+ attention_heads: 8
dropout_rate: 0.5
- pos_wise_dropout_rate: 0.5
- att_dropout_rate: 0.5
- conv_mod_kernel_size: 15
+ positional_dropout_rate: 0.5
+ attention_dropout_rate: 0.5
+ cnn_module_kernel: 15
num_blocks: 12
# decoder related
-decoder: rnn
-decoder_conf:
+rnnt_decoder: rnnt
+rnnt_decoder_conf:
embed_size: 512
hidden_size: 512
embed_dropout_rate: 0.5
dropout_rate: 0.5
-
joint_network_conf:
joint_space_size: 512
+# frontend related
+frontend: wav_frontend
+frontend_conf:
+ fs: 16000
+ window: hamming
+ n_mels: 80
+ frame_length: 25
+ frame_shift: 10
+ lfr_m: 1
+ lfr_n: 1
+
+
# Auxiliary CTC
+model: rnnt_unified
model_conf:
auxiliary_ctc_weight: 0.0
# minibatch related
use_amp: true
-batch_type: unsorted
-batch_size: 16
-num_workers: 16
# optimization related
accum_grad: 1
grad_clip: 5
-max_epoch: 200
+max_epoch: 120
val_scheduler_criterion:
- valid
- loss
@@ -65,8 +68,6 @@
scheduler_conf:
warmup_steps: 25000
-normalize: None
-
specaug: specaug
specaug_conf:
apply_time_warp: true
@@ -83,4 +84,16 @@
- 50
num_time_mask: 5
+dataset_conf:
+ data_names: speech,text
+ data_types: sound,text
+ shuffle: True
+ shuffle_conf:
+ shuffle_size: 2048
+ sort_size: 500
+ batch_conf:
+ batch_type: token
+ batch_size: 16000
+ num_workers: 8
+
log_interval: 50
--
Gitblit v1.9.1