python/FunASR-XL.git

			@@ -684,6 +684,13 @@
			# audio encoder
			speech = batch["speech"]
			speech_lengths = batch["speech_lengths"][:, 0]
			# fp16
			if kwargs.get("fp16", False):
			speech = speech.to(torch.float16)
			encoder_out_lens = encoder_out_lens.to(torch.float16)
			elif kwargs.get("bf16", False):
			speech = speech.to(torch.bfloat16)
			encoder_out_lens = encoder_out_lens.to(torch.bfloat16)
			encoder_out, encoder_out_lens = self.audio_encoder(speech.permute(0, 2, 1), speech_lengths)

			# audio_adaptor
			@@ -710,9 +717,9 @@
			dtype_map = {"bf16": torch.bfloat16, "fp16": torch.float16, "fp32": torch.float32}
			with torch.cuda.amp.autocast(dtype=dtype_map[llm_dtype]):
			label = contents["assistant"][0]
			self.llm = self.llm.to(dtype_map[llm_dtype])
			inputs_embeds = inputs_embeds.to(dtype_map[llm_dtype])
			attention_mask = attention_mask.to(dtype_map[llm_dtype])
			# self.llm = self.llm.to(dtype_map[llm_dtype])
			# inputs_embeds = inputs_embeds.to(dtype_map[llm_dtype])

			if not kwargs.get("tearchforing", False):

			generated_ids = self.llm.generate(
			@@ -732,6 +739,7 @@
			labels_ids = batch["labels_ids"]
			labels_ids[labels_ids == -1] = -100
			attention_mask = batch.get("attention_mask", None)
			# attention_mask = attention_mask.to(dtype_map[llm_dtype])
			model_outputs = self.llm(
			inputs_embeds=inputs_embeds, attention_mask=attention_mask, labels=labels_ids
			)