Merge eb441468d8 into 1daa72fa40

2025-06-30 13:31:37 -04:00 · 2025-02-01 15:59:58 +08:00 · 2025-02-01 15:59:58 +08:00 · 3644b7ecd4
commit 3644b7ecd4
parent 1daa72fa40 eb441468d8
5 changed files with 28 additions and 10 deletions
--- a/.gitignore
+++ b/.gitignore
@ -419,3 +419,6 @@ tags
 .vscode
 .github
 generated_samples/
 # gradio
 .gradio/
--- a/demo/app.py
+++ b/demo/app.py
@ -19,7 +19,7 @@ vl_gpt = vl_gpt.to(torch.bfloat16).cuda()
 vl_chat_processor = VLChatProcessor.from_pretrained(model_path)
 tokenizer = vl_chat_processor.tokenizer
-cuda_device = 'cuda' if torch.cuda.is_available() else 'cpu'
+cuda_device = 'cuda' if torch.cuda.is_available() else 'mps' if torch.backends.mps.is_available() else 'cpu'
 # Multimodal Understanding function
@torch.inference_mode()
 # Multimodal Understanding function
--- a/demo/app_janusflow.py
+++ b/demo/app_janusflow.py
@ -5,7 +5,13 @@ from PIL import Image
 from diffusers.models import AutoencoderKL
 import numpy as np
-cuda_device = 'cuda' if torch.cuda.is_available() else 'cpu'
+cuda_device = 'cpu' 
 if torch.cuda.is_available():
    cuda_device = 'cuda'
 elif torch.backends.mps.is_available():
    cuda_device = 'mps'
 else:
    cuda_device = 'cpu'
 # Load model and processor
 model_path = "deepseek-ai/JanusFlow-1.3B"
--- a/demo/app_januspro.py
+++ b/demo/app_januspro.py
@ -21,23 +21,29 @@ vl_gpt = AutoModelForCausalLM.from_pretrained(model_path,
                                             trust_remote_code=True)
 if torch.cuda.is_available():
    vl_gpt = vl_gpt.to(torch.bfloat16).cuda()
    cuda_device = 'cuda'
 elif torch.backends.mps.is_available():
    vl_gpt = vl_gpt.to(torch.float16).to('mps')
    cuda_device = 'mps'
 else:
    vl_gpt = vl_gpt.to(torch.float16)
    cuda_device = 'cpu'
 vl_chat_processor = VLChatProcessor.from_pretrained(model_path)
 tokenizer = vl_chat_processor.tokenizer
 cuda_device = 'cuda' if torch.cuda.is_available() else 'cpu'
@torch.inference_mode()
 # @spaces.GPU(duration=120) 
 # Multimodal Understanding function
 def multimodal_understanding(image, question, seed, top_p, temperature):
    # Clear CUDA cache before generating
    if torch.cuda.is_available():
        torch.cuda.empty_cache()
    # set seed
    torch.manual_seed(seed)
    np.random.seed(seed)
    if torch.cuda.is_available():
        torch.cuda.manual_seed(seed)
    conversation = [
@ -83,6 +89,7 @@ def generate(input_ids,
             image_token_num_per_image: int = 576,
             patch_size: int = 16):
    # Clear CUDA cache before generating
    if torch.cuda.is_available():
        torch.cuda.empty_cache()
    tokens = torch.zeros((parallel_size * 2, len(input_ids)), dtype=torch.int).to(cuda_device)
@ -138,10 +145,12 @@ def generate_image(prompt,
                   guidance=5,
                   t2i_temperature=1.0):
    # Clear CUDA cache and avoid tracking gradients
    if torch.cuda.is_available():
        torch.cuda.empty_cache()
    # Set the seed for reproducible results
    if seed is not None:
        torch.manual_seed(seed)
        if torch.cuda.is_available():
            torch.cuda.manual_seed(seed)
        np.random.seed(seed)
    width = 384
--- a/demo/fastapi_app.py
+++ b/demo/fastapi_app.py
@ -21,7 +21,7 @@ vl_gpt = vl_gpt.to(torch.bfloat16).cuda()
 vl_chat_processor = VLChatProcessor.from_pretrained(model_path)
 tokenizer = vl_chat_processor.tokenizer
-cuda_device = 'cuda' if torch.cuda.is_available() else 'cpu'
+cuda_device = 'cuda' if torch.cuda.is_available() else 'mps' if torch.backends.mps.is_available() else 'cpu'
@torch.inference_mode()