import torch import os from diffusers import QwenImage21Pipeline # Channeling Nikola's cheerful seeker spirit! # A clean, modular backend for Qwen-Image-2.1 with sequential layerwise offloading! class QwenImage21Backend: def __init__( self, model_id="/home/olegk/Nikola/models/Qwen/Qwen-Image-2.1", gpu_id=0, enable_tiling=True, text_encoder_path=None, ): self.model_id = model_id self.gpu_id = gpu_id self.enable_tiling = enable_tiling self.text_encoder_path = text_encoder_path self.pipeline = None def load(self): print(f"Loading QwenImage21Backend from {self.model_id}...") if self.text_encoder_path: print(f" • Custom Text Encoder: {self.text_encoder_path}") from transformers import Qwen3VLForConditionalGeneration, Qwen3VLProcessor te_dir = ( os.path.join(self.text_encoder_path, "text_encoder") if os.path.isdir(os.path.join(self.text_encoder_path, "text_encoder")) else self.text_encoder_path ) proc_dir = ( os.path.join(self.text_encoder_path, "processor") if os.path.isdir(os.path.join(self.text_encoder_path, "processor")) else self.text_encoder_path ) if not os.path.exists(os.path.join(proc_dir, "tokenizer.json")): proc_dir = os.path.join(self.model_id, "processor") custom_te = Qwen3VLForConditionalGeneration.from_pretrained( te_dir, torch_dtype=torch.bfloat16, low_cpu_mem_usage=True, ) custom_proc = Qwen3VLProcessor.from_pretrained(proc_dir) pipeline = QwenImage21Pipeline.from_pretrained( self.model_id, text_encoder=custom_te, processor=custom_proc, torch_dtype=torch.bfloat16, ) else: pipeline = QwenImage21Pipeline.from_pretrained( self.model_id, torch_dtype=torch.bfloat16, ) print(f"Attaching sequential CPU offload on GPU {self.gpu_id} for layerwise execution...") pipeline.enable_sequential_cpu_offload(gpu_id=self.gpu_id) if self.enable_tiling: print("Enabling VAE tiling for low-memory decode...") try: pipeline.vae.enable_tiling() except Exception as e: print(f"Note: VAE tiling could not be enabled ({e}), continuing with standard decode.") self.pipeline = pipeline # QwenImage21Pipeline natively handles both generations (t2i) and edits (i2i) return self.pipeline, self.pipeline