python/FunASR-XL.git

			@@ -684,6 +684,11 @@
			# audio encoder
			speech = batch["speech"]
			speech_lengths = batch["speech_lengths"][:, 0]
			# fp16
			if kwargs.get("fp16", False):
			speech = speech.to(torch.float16)
			elif kwargs.get("bf16", False):
			speech = speech.to(torch.bfloat16)
			encoder_out, encoder_out_lens = self.audio_encoder(speech.permute(0, 2, 1), speech_lengths)

			# audio_adaptor
			@@ -707,12 +712,18 @@
			]

			llm_dtype = kwargs.get("llm_dtype", "fp32")
			if llm_dtype == "fp32":
			llm_dtype = "fp16" if kwargs.get("fp16", False) else llm_dtype
			llm_dtype = "bf16" if kwargs.get("bf16", False) else llm_dtype

			dtype_map = {"bf16": torch.bfloat16, "fp16": torch.float16, "fp32": torch.float32}
			with torch.cuda.amp.autocast(dtype=dtype_map[llm_dtype]):
			with torch.cuda.amp.autocast(
			enabled=True if llm_dtype != "fp32" else False, dtype=dtype_map[llm_dtype]
			):
			label = contents["assistant"][0]
			self.llm = self.llm.to(dtype_map[llm_dtype])
			inputs_embeds = inputs_embeds.to(dtype_map[llm_dtype])
			attention_mask = attention_mask.to(dtype_map[llm_dtype])

			if not kwargs.get("tearchforing", False):

			generated_ids = self.llm.generate(
			@@ -732,6 +743,7 @@
			labels_ids = batch["labels_ids"]
			labels_ids[labels_ids == -1] = -100
			attention_mask = batch.get("attention_mask", None)
			# attention_mask = attention_mask.to(dtype_map[llm_dtype])
			model_outputs = self.llm(
			inputs_embeds=inputs_embeds, attention_mask=attention_mask, labels=labels_ids
			)