From 82e5ca37a8bd80f56c99f9d790a03b458ced716b Mon Sep 17 00:00:00 2001
From: 游雁 <zhifu.gzf@alibaba-inc.com>
Date: 星期二, 25 二月 2025 14:28:34 +0800
Subject: [PATCH] Large-Scale Data Training
---
examples/industrial_data_pretraining/llm_asr/conf/template.yaml | 2 +-
1 files changed, 1 insertions(+), 1 deletions(-)
diff --git a/examples/industrial_data_pretraining/llm_asr/conf/template.yaml b/examples/industrial_data_pretraining/llm_asr/conf/template.yaml
index 3c51ff4..c64c886 100644
--- a/examples/industrial_data_pretraining/llm_asr/conf/template.yaml
+++ b/examples/industrial_data_pretraining/llm_asr/conf/template.yaml
@@ -73,7 +73,7 @@
dataset: AudioLLMDataset
dataset_conf:
index_ds: IndexDSJsonl
- batch_sampler: RankFullLocalShuffleBatchSampler
+ batch_sampler: BatchSampler
batch_type: example # example or length
batch_size: 8 # if batch_type is example, batch_size is the numbers of samples; if length, batch_size is source_token_len+target_token_len;
max_token_length: 2048 # filter samples if source_token_len+target_token_len > max_token_length,
--
Gitblit v1.9.1