Instructions to use Zero-Vision/Llama-3-MixSenseV1_1 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Zero-Vision/Llama-3-MixSenseV1_1 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="Zero-Vision/Llama-3-MixSenseV1_1", trust_remote_code=True)# pip install -U transformers accelerate # Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("Zero-Vision/Llama-3-MixSenseV1_1", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use Zero-Vision/Llama-3-MixSenseV1_1 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "Zero-Vision/Llama-3-MixSenseV1_1" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Zero-Vision/Llama-3-MixSenseV1_1", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/Zero-Vision/Llama-3-MixSenseV1_1
- SGLang
How to use Zero-Vision/Llama-3-MixSenseV1_1 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "Zero-Vision/Llama-3-MixSenseV1_1" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Zero-Vision/Llama-3-MixSenseV1_1", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "Zero-Vision/Llama-3-MixSenseV1_1" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Zero-Vision/Llama-3-MixSenseV1_1", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use Zero-Vision/Llama-3-MixSenseV1_1 with Docker Model Runner:
docker model run hf.co/Zero-Vision/Llama-3-MixSenseV1_1
Download modeling_mixsense_llama.py from Zero-Vision/Llama-3-MixSenseV1_1: direct link, hf CLI and curl.
- Browser
- Download file 43.6 kB
-
https://huggingface.co/Zero-Vision/Llama-3-MixSenseV1_1/resolve/main/modeling_mixsense_llama.py
- Command line
-
hf download hf://Zero-Vision/Llama-3-MixSenseV1_1/modeling_mixsense_llama.py
-
curl -L -o modeling_mixsense_llama.py https://huggingface.co/Zero-Vision/Llama-3-MixSenseV1_1/resolve/main/modeling_mixsense_llama.py
43.6 kB
| import torch | |
| import torch.nn as nn | |
| from transformers import AutoConfig, CLIPImageProcessor | |
| import os | |
| from einops import rearrange | |
| from collections import OrderedDict | |
| from timm.models.layers import DropPath, trunc_normal_ | |
| import torch.nn.functional as F | |
| """ | |
| DaViT : Copied from https://huggingface.co/microsoft/Florence-2-large/blob/main/modeling_florence2.py | |
| """ | |
| class MySequential(nn.Sequential): | |
| def forward(self, *inputs): | |
| for module in self._modules.values(): | |
| if type(inputs) == tuple: | |
| inputs = module(*inputs) | |
| else: | |
| inputs = module(inputs) | |
| return inputs | |
| class PreNorm(nn.Module): | |
| def __init__(self, norm, fn, drop_path=None): | |
| super().__init__() | |
| self.norm = norm | |
| self.fn = fn | |
| self.drop_path = drop_path | |
| def forward(self, x, *args, **kwargs): | |
| shortcut = x | |
| if self.norm != None: | |
| x, size = self.fn(self.norm(x), *args, **kwargs) | |
| else: | |
| x, size = self.fn(x, *args, **kwargs) | |
| if self.drop_path: | |
| x = self.drop_path(x) | |
| x = shortcut + x | |
| return x, size | |
| class Mlp(nn.Module): | |
| def __init__( | |
| self, | |
| in_features, | |
| hidden_features=None, | |
| out_features=None, | |
| act_layer=nn.GELU, | |
| ): | |
| super().__init__() | |
| out_features = out_features or in_features | |
| hidden_features = hidden_features or in_features | |
| self.net = nn.Sequential( | |
| OrderedDict( | |
| [ | |
| ("fc1", nn.Linear(in_features, hidden_features)), | |
| ("act", act_layer()), | |
| ("fc2", nn.Linear(hidden_features, out_features)), | |
| ] | |
| ) | |
| ) | |
| def forward(self, x, size): | |
| return self.net(x), size | |
| class DepthWiseConv2d(nn.Module): | |
| def __init__( | |
| self, | |
| dim_in, | |
| kernel_size, | |
| padding, | |
| stride, | |
| bias=True, | |
| ): | |
| super().__init__() | |
| self.dw = nn.Conv2d( | |
| dim_in, | |
| dim_in, | |
| kernel_size=kernel_size, | |
| padding=padding, | |
| groups=dim_in, | |
| stride=stride, | |
| bias=bias, | |
| ) | |
| def forward(self, x, size): | |
| B, N, C = x.shape | |
| H, W = size | |
| assert N == H * W | |
| x = self.dw(x.transpose(1, 2).view(B, C, H, W)) | |
| size = (x.size(-2), x.size(-1)) | |
| x = x.flatten(2).transpose(1, 2) | |
| return x, size | |
| class ConvEmbed(nn.Module): | |
| """Image to Patch Embedding""" | |
| def __init__( | |
| self, | |
| patch_size=7, | |
| in_chans=3, | |
| embed_dim=64, | |
| stride=4, | |
| padding=2, | |
| norm_layer=None, | |
| pre_norm=True, | |
| ): | |
| super().__init__() | |
| self.patch_size = patch_size | |
| self.proj = nn.Conv2d( | |
| in_chans, embed_dim, kernel_size=patch_size, stride=stride, padding=padding | |
| ) | |
| dim_norm = in_chans if pre_norm else embed_dim | |
| self.norm = norm_layer(dim_norm) if norm_layer else None | |
| self.pre_norm = pre_norm | |
| def forward(self, x, size): | |
| H, W = size | |
| if len(x.size()) == 3: | |
| if self.norm and self.pre_norm: | |
| x = self.norm(x) | |
| x = rearrange(x, "b (h w) c -> b c h w", h=H, w=W) | |
| x = self.proj(x) | |
| _, _, H, W = x.shape | |
| x = rearrange(x, "b c h w -> b (h w) c") | |
| if self.norm and not self.pre_norm: | |
| x = self.norm(x) | |
| return x, (H, W) | |
| class ChannelAttention(nn.Module): | |
| def __init__(self, dim, groups=8, qkv_bias=True): | |
| super().__init__() | |
| self.groups = groups | |
| self.qkv = nn.Linear(dim, dim * 3, bias=qkv_bias) | |
| self.proj = nn.Linear(dim, dim) | |
| def forward(self, x, size): | |
| B, N, C = x.shape | |
| qkv = ( | |
| self.qkv(x) | |
| .reshape(B, N, 3, self.groups, C // self.groups) | |
| .permute(2, 0, 3, 1, 4) | |
| ) | |
| q, k, v = qkv[0], qkv[1], qkv[2] | |
| q = q * (float(N) ** -0.5) | |
| attention = q.transpose(-1, -2) @ k | |
| attention = attention.softmax(dim=-1) | |
| x = (attention @ v.transpose(-1, -2)).transpose(-1, -2) | |
| x = x.transpose(1, 2).reshape(B, N, C) | |
| x = self.proj(x) | |
| return x, size | |
| class ChannelBlock(nn.Module): | |
| def __init__( | |
| self, | |
| dim, | |
| groups, | |
| mlp_ratio=4.0, | |
| qkv_bias=True, | |
| drop_path_rate=0.0, | |
| act_layer=nn.GELU, | |
| norm_layer=nn.LayerNorm, | |
| conv_at_attn=True, | |
| conv_at_ffn=True, | |
| ): | |
| super().__init__() | |
| drop_path = DropPath(drop_path_rate) if drop_path_rate > 0.0 else nn.Identity() | |
| self.conv1 = ( | |
| PreNorm(None, DepthWiseConv2d(dim, 3, 1, 1)) if conv_at_attn else None | |
| ) | |
| self.channel_attn = PreNorm( | |
| norm_layer(dim), | |
| ChannelAttention(dim, groups=groups, qkv_bias=qkv_bias), | |
| drop_path, | |
| ) | |
| self.conv2 = ( | |
| PreNorm(None, DepthWiseConv2d(dim, 3, 1, 1)) if conv_at_ffn else None | |
| ) | |
| self.ffn = PreNorm( | |
| norm_layer(dim), | |
| Mlp( | |
| in_features=dim, | |
| hidden_features=int(dim * mlp_ratio), | |
| act_layer=act_layer, | |
| ), | |
| drop_path, | |
| ) | |
| def forward(self, x, size): | |
| if self.conv1: | |
| x, size = self.conv1(x, size) | |
| x, size = self.channel_attn(x, size) | |
| if self.conv2: | |
| x, size = self.conv2(x, size) | |
| x, size = self.ffn(x, size) | |
| return x, size | |
| def window_partition(x, window_size: int): | |
| B, H, W, C = x.shape | |
| x = x.view(B, H // window_size, window_size, W // window_size, window_size, C) | |
| windows = ( | |
| x.permute(0, 1, 3, 2, 4, 5).contiguous().view(-1, window_size, window_size, C) | |
| ) | |
| return windows | |
| def window_reverse(windows, batch_size: int, window_size: int, H: int, W: int): | |
| B = batch_size | |
| # this will cause onnx conversion failed for dynamic axis, because treated as constant | |
| # int(windows.shape[0] / (H * W / window_size / window_size)) | |
| x = windows.view( | |
| B, H // window_size, W // window_size, window_size, window_size, -1 | |
| ) | |
| x = x.permute(0, 1, 3, 2, 4, 5).contiguous().view(B, H, W, -1) | |
| return x | |
| class WindowAttention(nn.Module): | |
| def __init__(self, dim, num_heads, window_size, qkv_bias=True): | |
| super().__init__() | |
| self.dim = dim | |
| self.window_size = window_size | |
| self.num_heads = num_heads | |
| head_dim = dim // num_heads | |
| self.scale = float(head_dim) ** -0.5 | |
| self.qkv = nn.Linear(dim, dim * 3, bias=qkv_bias) | |
| self.proj = nn.Linear(dim, dim) | |
| self.softmax = nn.Softmax(dim=-1) | |
| def forward(self, x, size): | |
| H, W = size | |
| B, L, C = x.shape | |
| assert L == H * W, "input feature has wrong size" | |
| x = x.view(B, H, W, C) | |
| pad_l = pad_t = 0 | |
| pad_r = (self.window_size - W % self.window_size) % self.window_size | |
| pad_b = (self.window_size - H % self.window_size) % self.window_size | |
| x = F.pad(x, (0, 0, pad_l, pad_r, pad_t, pad_b)) | |
| _, Hp, Wp, _ = x.shape | |
| x = window_partition(x, self.window_size) | |
| x = x.view(-1, self.window_size * self.window_size, C) | |
| # W-MSA/SW-MSA | |
| # attn_windows = self.attn(x_windows) | |
| B_, N, C = x.shape | |
| qkv = ( | |
| self.qkv(x) | |
| .reshape(B_, N, 3, self.num_heads, C // self.num_heads) | |
| .permute(2, 0, 3, 1, 4) | |
| ) | |
| q, k, v = qkv[0], qkv[1], qkv[2] | |
| q = q * self.scale | |
| attn = q @ k.transpose(-2, -1) | |
| attn = self.softmax(attn) | |
| x = (attn @ v).transpose(1, 2).reshape(B_, N, C) | |
| x = self.proj(x) | |
| # merge windows | |
| x = x.view(-1, self.window_size, self.window_size, C) | |
| x = window_reverse(x, B, self.window_size, Hp, Wp) | |
| if pad_r > 0 or pad_b > 0: | |
| x = x[:, :H, :W, :].contiguous() | |
| x = x.view(B, H * W, C) | |
| return x, size | |
| class SpatialBlock(nn.Module): | |
| def __init__( | |
| self, | |
| dim, | |
| num_heads, | |
| window_size, | |
| mlp_ratio=4.0, | |
| qkv_bias=True, | |
| drop_path_rate=0.0, | |
| act_layer=nn.GELU, | |
| norm_layer=nn.LayerNorm, | |
| conv_at_attn=True, | |
| conv_at_ffn=True, | |
| ): | |
| super().__init__() | |
| drop_path = DropPath(drop_path_rate) if drop_path_rate > 0.0 else nn.Identity() | |
| self.conv1 = ( | |
| PreNorm(None, DepthWiseConv2d(dim, 3, 1, 1)) if conv_at_attn else None | |
| ) | |
| self.window_attn = PreNorm( | |
| norm_layer(dim), | |
| WindowAttention(dim, num_heads, window_size, qkv_bias=qkv_bias), | |
| drop_path, | |
| ) | |
| self.conv2 = ( | |
| PreNorm(None, DepthWiseConv2d(dim, 3, 1, 1)) if conv_at_ffn else None | |
| ) | |
| self.ffn = PreNorm( | |
| norm_layer(dim), | |
| Mlp( | |
| in_features=dim, | |
| hidden_features=int(dim * mlp_ratio), | |
| act_layer=act_layer, | |
| ), | |
| drop_path, | |
| ) | |
| def forward(self, x, size): | |
| if self.conv1: | |
| x, size = self.conv1(x, size) | |
| x, size = self.window_attn(x, size) | |
| if self.conv2: | |
| x, size = self.conv2(x, size) | |
| x, size = self.ffn(x, size) | |
| return x, size | |
| class DaViT(nn.Module): | |
| """DaViT: Dual-Attention Transformer | |
| Args: | |
| in_chans (int): Number of input image channels. Default: 3. | |
| num_classes (int): Number of classes for classification head. Default: 1000. | |
| patch_size (tuple(int)): Patch size of convolution in different stages. Default: (7, 2, 2, 2). | |
| patch_stride (tuple(int)): Patch stride of convolution in different stages. Default: (4, 2, 2, 2). | |
| patch_padding (tuple(int)): Patch padding of convolution in different stages. Default: (3, 0, 0, 0). | |
| patch_prenorm (tuple(bool)): If True, perform norm before convlution layer. Default: (True, False, False, False). | |
| embed_dims (tuple(int)): Patch embedding dimension in different stages. Default: (64, 128, 192, 256). | |
| num_heads (tuple(int)): Number of spatial attention heads in different stages. Default: (4, 8, 12, 16). | |
| num_groups (tuple(int)): Number of channel groups in different stages. Default: (4, 8, 12, 16). | |
| window_size (int): Window size. Default: 7. | |
| mlp_ratio (float): Ratio of mlp hidden dim to embedding dim. Default: 4. | |
| qkv_bias (bool): If True, add a learnable bias to query, key, value. Default: True. | |
| drop_path_rate (float): Stochastic depth rate. Default: 0.1. | |
| norm_layer (nn.Module): Normalization layer. Default: nn.LayerNorm. | |
| enable_checkpoint (bool): If True, enable checkpointing. Default: False. | |
| conv_at_attn (bool): If True, performe depthwise convolution before attention layer. Default: True. | |
| conv_at_ffn (bool): If True, performe depthwise convolution before ffn layer. Default: True. | |
| """ | |
| def __init__( | |
| self, | |
| in_chans=3, | |
| num_classes=1000, | |
| depths=(1, 1, 3, 1), | |
| patch_size=(7, 2, 2, 2), | |
| patch_stride=(4, 2, 2, 2), | |
| patch_padding=(3, 0, 0, 0), | |
| patch_prenorm=(False, False, False, False), | |
| embed_dims=(64, 128, 192, 256), | |
| num_heads=(3, 6, 12, 24), | |
| num_groups=(3, 6, 12, 24), | |
| window_size=7, | |
| mlp_ratio=4.0, | |
| qkv_bias=True, | |
| drop_path_rate=0.1, | |
| norm_layer=nn.LayerNorm, | |
| enable_checkpoint=False, | |
| conv_at_attn=True, | |
| conv_at_ffn=True, | |
| ): | |
| super().__init__() | |
| self.num_classes = num_classes | |
| self.embed_dims = embed_dims | |
| self.num_heads = num_heads | |
| self.num_groups = num_groups | |
| self.num_stages = len(self.embed_dims) | |
| self.enable_checkpoint = enable_checkpoint | |
| assert self.num_stages == len(self.num_heads) == len(self.num_groups) | |
| num_stages = len(embed_dims) | |
| dpr = [x.item() for x in torch.linspace(0, drop_path_rate, sum(depths) * 2)] | |
| depth_offset = 0 | |
| convs = [] | |
| blocks = [] | |
| for i in range(num_stages): | |
| conv_embed = ConvEmbed( | |
| patch_size=patch_size[i], | |
| stride=patch_stride[i], | |
| padding=patch_padding[i], | |
| in_chans=in_chans if i == 0 else self.embed_dims[i - 1], | |
| embed_dim=self.embed_dims[i], | |
| norm_layer=norm_layer, | |
| pre_norm=patch_prenorm[i], | |
| ) | |
| convs.append(conv_embed) | |
| block = MySequential( | |
| *[ | |
| MySequential( | |
| OrderedDict( | |
| [ | |
| ( | |
| "spatial_block", | |
| SpatialBlock( | |
| embed_dims[i], | |
| num_heads[i], | |
| window_size, | |
| drop_path_rate=dpr[depth_offset + j * 2], | |
| qkv_bias=qkv_bias, | |
| mlp_ratio=mlp_ratio, | |
| conv_at_attn=conv_at_attn, | |
| conv_at_ffn=conv_at_ffn, | |
| ), | |
| ), | |
| ( | |
| "channel_block", | |
| ChannelBlock( | |
| embed_dims[i], | |
| num_groups[i], | |
| drop_path_rate=dpr[depth_offset + j * 2 + 1], | |
| qkv_bias=qkv_bias, | |
| mlp_ratio=mlp_ratio, | |
| conv_at_attn=conv_at_attn, | |
| conv_at_ffn=conv_at_ffn, | |
| ), | |
| ), | |
| ] | |
| ) | |
| ) | |
| for j in range(depths[i]) | |
| ] | |
| ) | |
| blocks.append(block) | |
| depth_offset += depths[i] * 2 | |
| self.convs = nn.ModuleList(convs) | |
| self.blocks = nn.ModuleList(blocks) | |
| self.norms = norm_layer(self.embed_dims[-1]) | |
| self.avgpool = nn.AdaptiveAvgPool1d(1) | |
| self.head = ( | |
| nn.Linear(self.embed_dims[-1], num_classes) | |
| if num_classes > 0 | |
| else nn.Identity() | |
| ) | |
| # self.apply(self._init_weights) | |
| def dim_out(self): | |
| return self.embed_dims[-1] | |
| def _init_weights(self, m): | |
| if isinstance(m, nn.Linear): | |
| trunc_normal_(m.weight, std=0.02) | |
| if m.bias is not None: | |
| nn.init.constant_(m.bias, 0) | |
| elif isinstance(m, nn.Conv2d): | |
| nn.init.normal_(m.weight, std=0.02) | |
| for name, _ in m.named_parameters(): | |
| if name in ["bias"]: | |
| nn.init.constant_(m.bias, 0) | |
| elif isinstance(m, nn.LayerNorm): | |
| nn.init.constant_(m.weight, 1.0) | |
| nn.init.constant_(m.bias, 0) | |
| elif isinstance(m, nn.BatchNorm2d): | |
| nn.init.constant_(m.weight, 1.0) | |
| nn.init.constant_(m.bias, 0) | |
| def forward_features_unpool(self, x): | |
| """ | |
| forward until avg pooling | |
| Args: | |
| x (_type_): input image tensor | |
| """ | |
| input_size = (x.size(2), x.size(3)) | |
| for conv, block in zip(self.convs, self.blocks): | |
| x, input_size = conv(x, input_size) | |
| if self.enable_checkpoint: | |
| x, input_size = checkpoint.checkpoint(block, x, input_size) | |
| else: | |
| x, input_size = block(x, input_size) | |
| return x | |
| def forward_features(self, x): | |
| x = self.forward_features_unpool(x) | |
| # (batch_size, num_tokens, token_dim) | |
| x = self.avgpool(x.transpose(1, 2)) | |
| # (batch_size, 1, num_tokens) | |
| x = torch.flatten(x, 1) | |
| x = self.norms(x) | |
| return x | |
| def forward(self, x): | |
| x = self.forward_features(x) | |
| x = self.head(x) | |
| return x | |
| def from_config(cls, config): | |
| return cls( | |
| depths=config.depths, | |
| embed_dims=config.dim_embed, | |
| num_heads=config.num_heads, | |
| num_groups=config.num_groups, | |
| patch_size=config.patch_size, | |
| patch_stride=config.patch_stride, | |
| patch_padding=config.patch_padding, | |
| patch_prenorm=config.patch_prenorm, | |
| drop_path_rate=config.drop_path_rate, | |
| window_size=config.window_size, | |
| ) | |
| class DaViTVisionTower(nn.Module): | |
| def __init__(self, vision_tower, args, delay_load=False): | |
| super().__init__() | |
| self.is_loaded = False | |
| self.image_size = getattr(args, "size", 768) | |
| self.vision_tower_name = vision_tower | |
| self.select_layer = args.mm_vision_select_layer | |
| self.select_feature = getattr(args, "mm_vision_select_feature", "patch") | |
| self.config = AutoConfig.from_pretrained( | |
| self.vision_tower_name, trust_remote_code=True | |
| ).vision_config | |
| if not delay_load: | |
| self.load_model() | |
| elif getattr(args, "unfreeze_mm_vision_tower", False): | |
| self.load_model() | |
| else: | |
| self.cfg_only = self.config | |
| def load_model(self, device_map=None): | |
| if self.is_loaded: | |
| print( | |
| "{} is already loaded, `load_model` called again, skipping.".format( | |
| self.vision_tower_name | |
| ) | |
| ) | |
| return | |
| self.vision_tower = DaViT.from_config(config=self.config) | |
| del self.vision_tower.head | |
| del self.vision_tower.norms | |
| self.image_processor = CLIPImageProcessor.from_pretrained( | |
| self.vision_tower_name | |
| ) | |
| self.vision_tower.requires_grad_(False) | |
| self.is_loaded = True | |
| def feature_select(self, image_forward_outs): | |
| image_features = image_forward_outs | |
| if self.select_feature == "patch": | |
| image_features = image_features[:, 1:] | |
| elif self.select_feature == "cls_patch": | |
| image_features = image_features | |
| else: | |
| raise ValueError(f"Unexpected select feature: {self.select_feature}") | |
| return image_features | |
| def forward(self, images): | |
| if type(images) is list: | |
| image_features = [] | |
| for image in images: | |
| image_forward_out = self.vision_tower.forward_features_unpool( | |
| image.to(device=self.device, dtype=self.dtype).unsqueeze(0) | |
| ) | |
| image_feature = self.feature_select(image_forward_out).to(image.dtype) | |
| image_features.append(image_forward_out) | |
| else: | |
| image_forward_outs = self.vision_tower.forward_features_unpool( | |
| images.to(device=self.device, dtype=self.dtype) | |
| ) | |
| image_features = self.feature_select(image_forward_outs).to(images.dtype) | |
| return image_features | |
| def dummy_feature(self): | |
| return torch.zeros(1, self.hidden_size, device=self.device, dtype=self.dtype) | |
| def dtype(self): | |
| for n, p in self.vision_tower.named_parameters(): | |
| dtype = p.dtype | |
| break | |
| return dtype | |
| def device(self): | |
| for n, p in self.vision_tower.named_parameters(): | |
| device = p.device | |
| break | |
| return device | |
| def hidden_size(self): | |
| return self.config.dim_embed[-1] | |
| def num_patches_per_side(self): | |
| return self.config.image_size // self.config.patch_size | |
| def num_patches(self): | |
| return (self.config.image_size // self.config.patch_size) ** 2 | |
| from abc import ABC, abstractmethod | |
| IGNORE_INDEX = -100 | |
| IMAGE_TOKEN_INDEX = -200 | |
| DEFAULT_IMAGE_PATCH_TOKEN = "<im_patch>" | |
| DEFAULT_IM_START_TOKEN = "<im_start>" | |
| DEFAULT_IM_END_TOKEN = "<im_end>" | |
| def build_vision_tower(vision_tower_cfg, **kwargs): | |
| vision_tower = getattr( | |
| vision_tower_cfg, | |
| "mm_vision_tower", | |
| getattr(vision_tower_cfg, "vision_tower", None), | |
| ) | |
| return DaViTVisionTower(vision_tower, args=vision_tower_cfg, **kwargs) | |
| import re | |
| def build_vision_projector(config, delay_load=False, **kwargs): | |
| projector_type = getattr(config, "mm_projector_type", "linear") | |
| mlp_gelu_match = re.match(r"^mlp(\d+)x_gelu$", projector_type) | |
| if mlp_gelu_match: | |
| mlp_depth = int(mlp_gelu_match.group(1)) | |
| modules = [nn.Linear(config.mm_hidden_size, config.hidden_size)] | |
| for _ in range(1, mlp_depth): | |
| modules.append(nn.GELU()) | |
| modules.append(nn.Linear(config.hidden_size, config.hidden_size)) | |
| return nn.Sequential(*modules) | |
| class MixsenseMetaModel: | |
| def __init__(self, config): | |
| super(MixsenseMetaModel, self).__init__(config) | |
| if hasattr(config, "mm_vision_tower"): | |
| self.vision_tower = build_vision_tower(config, delay_load=True) | |
| self.mm_projector = build_vision_projector(config) | |
| if "unpad" in getattr(config, "mm_patch_merge_type", ""): | |
| self.image_newline = nn.Parameter( | |
| torch.empty(config.hidden_size, dtype=self.dtype) | |
| ) | |
| def get_vision_tower(self): | |
| vision_tower = getattr(self, "vision_tower", None) | |
| if type(vision_tower) is list: | |
| vision_tower = vision_tower[0] | |
| return vision_tower | |
| def initialize_vision_modules(self, model_args, fsdp=None): | |
| vision_tower = model_args.vision_tower | |
| mm_vision_select_layer = model_args.mm_vision_select_layer | |
| mm_vision_select_feature = model_args.mm_vision_select_feature | |
| pretrain_mm_mlp_adapter = model_args.pretrain_mm_mlp_adapter | |
| mm_patch_merge_type = model_args.mm_patch_merge_type | |
| self.config.mm_vision_tower = vision_tower | |
| if self.get_vision_tower() is None: | |
| vision_tower = build_vision_tower(model_args) | |
| if fsdp is not None and len(fsdp) > 0: | |
| self.vision_tower = [vision_tower] | |
| else: | |
| self.vision_tower = vision_tower | |
| else: | |
| if fsdp is not None and len(fsdp) > 0: | |
| vision_tower = self.vision_tower[0] | |
| else: | |
| vision_tower = self.vision_tower | |
| vision_tower.load_model() | |
| self.config.use_mm_proj = True | |
| self.config.mm_projector_type = getattr( | |
| model_args, "mm_projector_type", "linear" | |
| ) | |
| self.config.mm_hidden_size = vision_tower.hidden_size | |
| self.config.mm_vision_select_layer = mm_vision_select_layer | |
| self.config.mm_vision_select_feature = mm_vision_select_feature | |
| self.config.mm_patch_merge_type = mm_patch_merge_type | |
| if getattr(self, "mm_projector", None) is None: | |
| self.mm_projector = build_vision_projector(self.config) | |
| if "unpad" in mm_patch_merge_type: | |
| embed_std = 1 / torch.sqrt( | |
| torch.tensor(self.config.hidden_size, dtype=self.dtype) | |
| ) | |
| self.image_newline = nn.Parameter( | |
| torch.randn(self.config.hidden_size, dtype=self.dtype) * embed_std | |
| ) | |
| else: | |
| # In case it is frozen by LoRA | |
| for p in self.mm_projector.parameters(): | |
| p.requires_grad = True | |
| if pretrain_mm_mlp_adapter is not None: | |
| mm_projector_weights = torch.load( | |
| pretrain_mm_mlp_adapter, map_location="cpu" | |
| ) | |
| def get_w(weights, keyword): | |
| return { | |
| k.split(keyword + ".")[1]: v | |
| for k, v in weights.items() | |
| if keyword in k | |
| } | |
| self.mm_projector.load_state_dict( | |
| get_w(mm_projector_weights, "mm_projector") | |
| ) | |
| class MixsenseMetaForCausalLM(ABC): | |
| def get_model(self): | |
| pass | |
| def get_vision_tower(self): | |
| return self.get_model().get_vision_tower() | |
| def encode_images(self, images): | |
| image_features = self.get_model().get_vision_tower()(images) | |
| image_features = self.get_model().mm_projector(image_features) | |
| return image_features | |
| def prepare_inputs_labels_for_multimodal( | |
| self, | |
| input_ids, | |
| position_ids, | |
| attention_mask, | |
| past_key_values, | |
| labels, | |
| images, | |
| image_sizes=None, | |
| ): | |
| vision_tower = self.get_vision_tower() | |
| if vision_tower is None or images is None or input_ids.shape[1] == 1: | |
| return ( | |
| input_ids, | |
| position_ids, | |
| attention_mask, | |
| past_key_values, | |
| None, | |
| labels, | |
| ) | |
| elif type(images) is list or images.ndim == 5: | |
| if type(images) is list: | |
| images = [x.unsqueeze(0) if x.ndim == 3 else x for x in images] | |
| concat_images = torch.cat([image for image in images], dim=0) | |
| image_features = self.encode_images(concat_images) | |
| split_sizes = [image.shape[0] for image in images] | |
| image_features = torch.split(image_features, split_sizes, dim=0) | |
| mm_patch_merge_type = getattr(self.config, "mm_patch_merge_type", "flat") | |
| image_aspect_ratio = getattr(self.config, "image_aspect_ratio", "square") | |
| if mm_patch_merge_type == "flat": | |
| image_features = [x.flatten(0, 1) for x in image_features] | |
| else: | |
| image_features = self.encode_images(images) | |
| # TODO: image start / end is not implemented here to support pretraining. | |
| if getattr(self.config, "tune_mm_mlp_adapter", False) and getattr( | |
| self.config, "mm_use_im_start_end", False | |
| ): | |
| raise NotImplementedError | |
| # Let's just add dummy tensors if they do not exist, | |
| # it is a headache to deal with None all the time. | |
| # But it is not ideal, and if you have a better idea, | |
| # please open an issue / submit a PR, thanks. | |
| _labels = labels | |
| _position_ids = position_ids | |
| _attention_mask = attention_mask | |
| if attention_mask is None: | |
| attention_mask = torch.ones_like(input_ids, dtype=torch.bool) | |
| else: | |
| attention_mask = attention_mask.bool() | |
| if position_ids is None: | |
| position_ids = torch.arange( | |
| 0, input_ids.shape[1], dtype=torch.long, device=input_ids.device | |
| ) | |
| if labels is None: | |
| labels = torch.full_like(input_ids, IGNORE_INDEX) | |
| # remove the padding using attention_mask -- FIXME | |
| _input_ids = input_ids | |
| input_ids = [ | |
| cur_input_ids[cur_attention_mask] | |
| for cur_input_ids, cur_attention_mask in zip(input_ids, attention_mask) | |
| ] | |
| labels = [ | |
| cur_labels[cur_attention_mask] | |
| for cur_labels, cur_attention_mask in zip(labels, attention_mask) | |
| ] | |
| new_input_embeds = [] | |
| new_labels = [] | |
| cur_image_idx = 0 | |
| for batch_idx, cur_input_ids in enumerate(input_ids): | |
| num_images = (cur_input_ids == IMAGE_TOKEN_INDEX).sum() | |
| if num_images == 0: | |
| cur_image_features = image_features[cur_image_idx] | |
| cur_input_embeds_1 = self.get_model().embed_tokens(cur_input_ids) | |
| cur_input_embeds = torch.cat( | |
| [cur_input_embeds_1, cur_image_features[0:0]], dim=0 | |
| ) | |
| new_input_embeds.append(cur_input_embeds) | |
| new_labels.append(labels[batch_idx]) | |
| cur_image_idx += 1 | |
| continue | |
| image_token_indices = ( | |
| [-1] | |
| + torch.where(cur_input_ids == IMAGE_TOKEN_INDEX)[0].tolist() | |
| + [cur_input_ids.shape[0]] | |
| ) | |
| cur_input_ids_noim = [] | |
| cur_labels = labels[batch_idx] | |
| cur_labels_noim = [] | |
| for i in range(len(image_token_indices) - 1): | |
| cur_input_ids_noim.append( | |
| cur_input_ids[ | |
| image_token_indices[i] + 1 : image_token_indices[i + 1] | |
| ] | |
| ) | |
| cur_labels_noim.append( | |
| cur_labels[image_token_indices[i] + 1 : image_token_indices[i + 1]] | |
| ) | |
| split_sizes = [x.shape[0] for x in cur_labels_noim] | |
| cur_input_embeds = self.get_model().embed_tokens( | |
| torch.cat(cur_input_ids_noim) | |
| ) | |
| cur_input_embeds_no_im = torch.split(cur_input_embeds, split_sizes, dim=0) | |
| cur_new_input_embeds = [] | |
| cur_new_labels = [] | |
| for i in range(num_images + 1): | |
| cur_new_input_embeds.append(cur_input_embeds_no_im[i]) | |
| cur_new_labels.append(cur_labels_noim[i]) | |
| if i < num_images: | |
| cur_image_features = image_features[cur_image_idx] | |
| cur_image_idx += 1 | |
| cur_new_input_embeds.append(cur_image_features) | |
| cur_new_labels.append( | |
| torch.full( | |
| (cur_image_features.shape[0],), | |
| IGNORE_INDEX, | |
| device=cur_labels.device, | |
| dtype=cur_labels.dtype, | |
| ) | |
| ) | |
| cur_new_input_embeds = [x.to(self.device) for x in cur_new_input_embeds] | |
| cur_new_input_embeds = torch.cat(cur_new_input_embeds) | |
| cur_new_labels = torch.cat(cur_new_labels) | |
| new_input_embeds.append(cur_new_input_embeds) | |
| new_labels.append(cur_new_labels) | |
| # Truncate sequences to max length as image embeddings can make the sequence longer | |
| tokenizer_model_max_length = getattr( | |
| self.config, "tokenizer_model_max_length", None | |
| ) | |
| if tokenizer_model_max_length is not None: | |
| new_input_embeds = [ | |
| x[:tokenizer_model_max_length] for x in new_input_embeds | |
| ] | |
| new_labels = [x[:tokenizer_model_max_length] for x in new_labels] | |
| # Combine them | |
| max_len = max(x.shape[0] for x in new_input_embeds) | |
| batch_size = len(new_input_embeds) | |
| new_input_embeds_padded = [] | |
| new_labels_padded = torch.full( | |
| (batch_size, max_len), | |
| IGNORE_INDEX, | |
| dtype=new_labels[0].dtype, | |
| device=new_labels[0].device, | |
| ) | |
| attention_mask = torch.zeros( | |
| (batch_size, max_len), | |
| dtype=attention_mask.dtype, | |
| device=attention_mask.device, | |
| ) | |
| position_ids = torch.zeros( | |
| (batch_size, max_len), dtype=position_ids.dtype, device=position_ids.device | |
| ) | |
| for i, (cur_new_embed, cur_new_labels) in enumerate( | |
| zip(new_input_embeds, new_labels) | |
| ): | |
| cur_len = cur_new_embed.shape[0] | |
| if getattr(self.config, "tokenizer_padding_side", "right") == "left": | |
| new_input_embeds_padded.append( | |
| torch.cat( | |
| ( | |
| torch.zeros( | |
| (max_len - cur_len, cur_new_embed.shape[1]), | |
| dtype=cur_new_embed.dtype, | |
| device=cur_new_embed.device, | |
| ), | |
| cur_new_embed, | |
| ), | |
| dim=0, | |
| ) | |
| ) | |
| if cur_len > 0: | |
| new_labels_padded[i, -cur_len:] = cur_new_labels | |
| attention_mask[i, -cur_len:] = True | |
| position_ids[i, -cur_len:] = torch.arange( | |
| 0, cur_len, dtype=position_ids.dtype, device=position_ids.device | |
| ) | |
| else: | |
| new_input_embeds_padded.append( | |
| torch.cat( | |
| ( | |
| cur_new_embed, | |
| torch.zeros( | |
| (max_len - cur_len, cur_new_embed.shape[1]), | |
| dtype=cur_new_embed.dtype, | |
| device=cur_new_embed.device, | |
| ), | |
| ), | |
| dim=0, | |
| ) | |
| ) | |
| if cur_len > 0: | |
| new_labels_padded[i, :cur_len] = cur_new_labels | |
| attention_mask[i, :cur_len] = True | |
| position_ids[i, :cur_len] = torch.arange( | |
| 0, cur_len, dtype=position_ids.dtype, device=position_ids.device | |
| ) | |
| new_input_embeds = torch.stack(new_input_embeds_padded, dim=0) | |
| if _labels is None: | |
| new_labels = None | |
| else: | |
| new_labels = new_labels_padded | |
| if _attention_mask is None: | |
| attention_mask = None | |
| else: | |
| attention_mask = attention_mask.to(dtype=_attention_mask.dtype) | |
| if _position_ids is None: | |
| position_ids = None | |
| return ( | |
| None, | |
| position_ids, | |
| attention_mask, | |
| past_key_values, | |
| new_input_embeds, | |
| new_labels, | |
| ) | |
| def initialize_vision_tokenizer(self, model_args, tokenizer): | |
| if model_args.mm_use_im_patch_token: | |
| tokenizer.add_tokens([DEFAULT_IMAGE_PATCH_TOKEN], special_tokens=True) | |
| self.resize_token_embeddings(len(tokenizer)) | |
| if model_args.mm_use_im_start_end: | |
| num_new_tokens = tokenizer.add_tokens( | |
| [DEFAULT_IM_START_TOKEN, DEFAULT_IM_END_TOKEN], special_tokens=True | |
| ) | |
| self.resize_token_embeddings(len(tokenizer)) | |
| if num_new_tokens > 0: | |
| input_embeddings = self.get_input_embeddings().weight.data | |
| output_embeddings = self.get_output_embeddings().weight.data | |
| input_embeddings_avg = input_embeddings[:-num_new_tokens].mean( | |
| dim=0, keepdim=True | |
| ) | |
| output_embeddings_avg = output_embeddings[:-num_new_tokens].mean( | |
| dim=0, keepdim=True | |
| ) | |
| input_embeddings[-num_new_tokens:] = input_embeddings_avg | |
| output_embeddings[-num_new_tokens:] = output_embeddings_avg | |
| if model_args.tune_mm_mlp_adapter: | |
| for p in self.get_input_embeddings().parameters(): | |
| p.requires_grad = True | |
| for p in self.get_output_embeddings().parameters(): | |
| p.requires_grad = False | |
| if model_args.pretrain_mm_mlp_adapter: | |
| mm_projector_weights = torch.load( | |
| model_args.pretrain_mm_mlp_adapter, map_location="cpu" | |
| ) | |
| embed_tokens_weight = mm_projector_weights["model.embed_tokens.weight"] | |
| assert num_new_tokens == 2 | |
| if input_embeddings.shape == embed_tokens_weight.shape: | |
| input_embeddings[-num_new_tokens:] = embed_tokens_weight[ | |
| -num_new_tokens: | |
| ] | |
| elif embed_tokens_weight.shape[0] == num_new_tokens: | |
| input_embeddings[-num_new_tokens:] = embed_tokens_weight | |
| else: | |
| raise ValueError( | |
| f"Unexpected embed_tokens_weight shape. Pretrained: {embed_tokens_weight.shape}. Current: {input_embeddings.shape}. Numer of new tokens: {num_new_tokens}." | |
| ) | |
| elif model_args.mm_use_im_patch_token: | |
| if model_args.tune_mm_mlp_adapter: | |
| for p in self.get_input_embeddings().parameters(): | |
| p.requires_grad = False | |
| for p in self.get_output_embeddings().parameters(): | |
| p.requires_grad = False | |
| from typing import List, Optional, Tuple, Union | |
| from transformers import ( | |
| AutoConfig, | |
| AutoModelForCausalLM, | |
| LlamaConfig, | |
| LlamaModel, | |
| LlamaForCausalLM, | |
| ) | |
| from transformers.modeling_outputs import CausalLMOutputWithPast | |
| from transformers.generation.utils import GenerateOutput | |
| class MixsenseConfig(LlamaConfig): | |
| model_type = "mixsense_llama" | |
| class MixsenseLlamaModel(MixsenseMetaModel, LlamaModel): | |
| config_class = MixsenseConfig | |
| def __init__(self, config: LlamaConfig): | |
| super(MixsenseLlamaModel, self).__init__(config) | |
| class MixsenseLlamaForCausalLM(LlamaForCausalLM, MixsenseMetaForCausalLM): | |
| config_class = MixsenseConfig | |
| def __init__(self, config): | |
| super(LlamaForCausalLM, self).__init__(config) | |
| self.model = MixsenseLlamaModel(config) | |
| self.pretraining_tp = config.pretraining_tp | |
| self.vocab_size = config.vocab_size | |
| self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False) | |
| # Initialize weights and apply final processing | |
| self.post_init() | |
| def get_model(self): | |
| return self.model | |
| def forward( | |
| self, | |
| input_ids: torch.LongTensor = None, | |
| attention_mask: Optional[torch.Tensor] = None, | |
| position_ids: Optional[torch.LongTensor] = None, | |
| past_key_values: Optional[List[torch.FloatTensor]] = None, | |
| inputs_embeds: Optional[torch.FloatTensor] = None, | |
| labels: Optional[torch.LongTensor] = None, | |
| use_cache: Optional[bool] = None, | |
| output_attentions: Optional[bool] = None, | |
| output_hidden_states: Optional[bool] = None, | |
| images: Optional[torch.FloatTensor] = None, | |
| image_sizes: Optional[List[List[int]]] = None, | |
| return_dict: Optional[bool] = None, | |
| cache_position: Optional[torch.LongTensor] = None, | |
| ) -> Union[Tuple, CausalLMOutputWithPast]: | |
| if inputs_embeds is None: | |
| ( | |
| input_ids, | |
| position_ids, | |
| attention_mask, | |
| past_key_values, | |
| inputs_embeds, | |
| labels, | |
| ) = self.prepare_inputs_labels_for_multimodal( | |
| input_ids, | |
| position_ids, | |
| attention_mask, | |
| past_key_values, | |
| labels, | |
| images, | |
| image_sizes, | |
| ) | |
| return super().forward( | |
| input_ids=input_ids, | |
| attention_mask=attention_mask, | |
| position_ids=position_ids, | |
| past_key_values=past_key_values, | |
| inputs_embeds=inputs_embeds, | |
| labels=labels, | |
| use_cache=use_cache, | |
| output_attentions=output_attentions, | |
| output_hidden_states=output_hidden_states, | |
| return_dict=return_dict, | |
| cache_position=cache_position, | |
| ) | |
| def generate( | |
| self, | |
| inputs: Optional[torch.Tensor] = None, | |
| images: Optional[torch.Tensor] = None, | |
| image_sizes: Optional[torch.Tensor] = None, | |
| **kwargs, | |
| ) -> Union[GenerateOutput, torch.LongTensor]: | |
| position_ids = kwargs.pop("position_ids", None) | |
| attention_mask = kwargs.pop("attention_mask", None) | |
| if "inputs_embeds" in kwargs: | |
| raise NotImplementedError("`inputs_embeds` is not supported") | |
| if images is not None: | |
| (inputs, position_ids, attention_mask, _, inputs_embeds, _) = ( | |
| self.prepare_inputs_labels_for_multimodal( | |
| inputs, | |
| position_ids, | |
| attention_mask, | |
| None, | |
| None, | |
| images, | |
| image_sizes=image_sizes, | |
| ) | |
| ) | |
| else: | |
| inputs_embeds = self.get_model().embed_tokens(inputs) | |
| output = super().generate( | |
| position_ids=position_ids, | |
| attention_mask=attention_mask, | |
| inputs_embeds=inputs_embeds, | |
| **kwargs, | |
| ) | |
| return output | |
| def prepare_inputs_for_generation( | |
| self, input_ids, past_key_values=None, inputs_embeds=None, cache_position=None, **kwargs | |
| ): | |
| images = kwargs.pop("images", None) | |
| image_sizes = kwargs.pop("image_sizes", None) | |
| inputs = super().prepare_inputs_for_generation( | |
| input_ids, | |
| past_key_values=past_key_values, | |
| inputs_embeds=inputs_embeds, | |
| cache_position=cache_position, | |
| **kwargs, | |
| ) | |
| if images is not None: | |
| inputs["images"] = images | |
| if image_sizes is not None: | |
| inputs["image_sizes"] = image_sizes | |
| return inputs | |
| def image_process(self,images): | |
| vision_tower = self.get_vision_tower() | |
| if not vision_tower.is_loaded: | |
| vision_tower.load_model() | |
| processor = vision_tower.image_processor | |
| def expand2square(pil_img, background_color): | |
| from PIL import Image | |
| width, height = pil_img.size | |
| if width == height: | |
| return pil_img | |
| elif width > height: | |
| result = Image.new(pil_img.mode, (width, width), background_color) | |
| result.paste(pil_img, (0, (width - height) // 2)) | |
| return result | |
| else: | |
| result = Image.new(pil_img.mode, (height, height), background_color) | |
| result.paste(pil_img, ((height - width) // 2, 0)) | |
| return result | |
| processed_images=[] | |
| for image in images: | |
| image = expand2square(image, tuple(int(x*255) for x in processor.image_mean)) | |
| image = processor.preprocess(image, return_tensors='pt')['pixel_values'][0] | |
| processed_images.append(image) | |
| if all(x.shape == processed_images[0].shape for x in processed_images): | |
| processed_images = torch.stack(processed_images, dim=0) | |
| return processed_images | |
| def text_process(self,text,tokenizer): | |
| prompt=f"<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\nYou are a helpful language and vision assistant. You are able to understand the visual content that the user provides, and assist the user with a variety of tasks using natural language.<|eot_id|><|start_header_id|>user<|end_header_id|>\n\n<image>\n{text}<|eot_id|><|start_header_id|>assistant<|end_header_id|>\n\n" | |
| text_chunks = [tokenizer(chunk).input_ids for chunk in prompt.split('<image>')] | |
| input_ids = torch.tensor(text_chunks[0] + [-200] + text_chunks[1][1:], dtype=torch.long).unsqueeze(0) | |
| return input_ids | |
| AutoConfig.register("mixsense_llama", MixsenseConfig) | |
| AutoModelForCausalLM.register(MixsenseConfig, MixsenseLlamaForCausalLM) |