From 2ebcda008d265ae83ebda34ed1be6f7a30e30b00 Mon Sep 17 00:00:00 2001 From: Yanyu Ren Date: Tue, 17 Dec 2024 15:46:07 +0800 Subject: [PATCH] Fix issue #8 and #4 --- README.md | 1 + deepseek_vl/models/modeling_deepseek.py | 5 ++++- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index f427054..b55a7da 100644 --- a/README.md +++ b/README.md @@ -156,6 +156,7 @@ inputs_embeds = vl_gpt.prepare_inputs_embeds(**prepare_inputs) # run the model to get the response outputs = vl_gpt.language.generate( + input_ids = prepare_inputs["input_ids"].to(vl_gpt.device), inputs_embeds=inputs_embeds, attention_mask=prepare_inputs.attention_mask, pad_token_id=tokenizer.eos_token_id, diff --git a/deepseek_vl/models/modeling_deepseek.py b/deepseek_vl/models/modeling_deepseek.py index 6df4fed..925c214 100644 --- a/deepseek_vl/models/modeling_deepseek.py +++ b/deepseek_vl/models/modeling_deepseek.py @@ -1753,13 +1753,16 @@ class DeepseekV2ForCausalLM(DeepseekV2PreTrainedModel): output = (logits,) + outputs[1:] return (loss,) + output if loss is not None else output - return CausalLMOutputWithPast( + device = input_ids.device if input_ids is not None else inputs_embeds.device + output = CausalLMOutputWithPast( loss=loss, logits=logits, past_key_values=outputs.past_key_values, hidden_states=outputs.hidden_states, attentions=outputs.attentions, ) + output['logits'] = output['logits'].to(device) + return output def prepare_inputs_for_generation( self,