THUDM
/

glm-4v-9b

@@ -6,6 +6,7 @@ from transformers.activations import ACT2FN
 import math
 from torch.nn import LayerNorm
 def standard_attention(query_layer, key_layer, value_layer, scaling_attention_score=True):
     if scaling_attention_score:
         query_layer = query_layer / math.sqrt(query_layer.shape[-1])
@@ -16,11 +17,12 @@ def standard_attention(query_layer, key_layer, value_layer, scaling_attention_sc
     context_layer = torch.matmul(attention_probs, value_layer)
     return context_layer
 def attention_fn_default(query_layer, key_layer, value_layer, scaling_attention_score=True):
     if int(torch.__version__.split('.')[0]) >= 2 and scaling_attention_score:
         # Pytorch 2.0 attention uses very much memory if attention_mask is float, and has NaN bug if attention_mask is None.
         attn_output = torch.nn.functional.scaled_dot_product_attention(
-            query_layer, key_layer, value_layer,
             attn_mask=None,
             dropout_p=0.,
             is_causal=False
@@ -31,10 +33,12 @@ def attention_fn_default(query_layer, key_layer, value_layer, scaling_attention_
             query_layer, key_layer, value_layer, scaling_attention_score=scaling_attention_score
         )
 class PatchEmbedding(nn.Module):
     def __init__(self, config):
         super().__init__()
-        self.proj = nn.Conv2d(config.in_channels, config.hidden_size, kernel_size=config.patch_size, stride=config.patch_size)
         self.cls_embedding = nn.Parameter(torch.zeros(1, config.hidden_size))
         self.position_embedding = nn.Embedding(config.num_positions, config.hidden_size)
@@ -62,11 +66,11 @@ class Attention(nn.Module):
         qkv = self.query_key_value(x)
         qkv = qkv.reshape(B, L, 3, self.num_heads, -1).permute(2, 0, 3, 1, 4)  # 3, B, H, L, D
         q, k, v = qkv[0], qkv[1], qkv[2]
         out = attention_fn_default(
             q, k, v
         )
-        output = self.dense(out.transpose(1, 2).reshape(B, L, -1))
         output = self.output_dropout(output)
         return output
@@ -105,7 +109,9 @@ class TransformerLayer(nn.Module):
         attention_output = self.input_layernorm(self.attention(attention_input))
         hidden_states = attention_input + attention_output
         mlp_input = hidden_states
-        mlp_output = self.post_attention_layernorm(self.mlp(mlp_input))
         output = mlp_input + mlp_output
         return output
@@ -147,7 +153,8 @@ class EVA2CLIPModel(nn.Module):
         self.patch_embedding = PatchEmbedding(vision_config)
         self.transformer = Transformer(vision_config)
         self.linear_proj = GLU(config, in_features=config.hidden_size)
-        self.conv = nn.Conv2d(in_channels=vision_config.hidden_size, out_channels=config.hidden_size, kernel_size=2, stride=2)
         self.boi = nn.Parameter(torch.zeros(1, 1, config.hidden_size))
         self.eoi = nn.Parameter(torch.zeros(1, 1, config.hidden_size))
         self.scaling_factor = vision_config.scaling_factor
@@ -158,14 +165,16 @@ class EVA2CLIPModel(nn.Module):
         x = x[:, 1:]
         b, s, h = x.shape
-        grid_size = int(s**0.5)
         x = x.view(b, grid_size, grid_size, h).permute(0, 3, 1, 2)
         x = self.conv(x)
         x = x.flatten(2).transpose(1, 2)
         x = self.linear_proj(x)
-        boi = self.boi.expand(x.shape[0], -1, -1)
-        eoi = self.eoi.expand(x.shape[0], -1, -1)
         x = torch.cat((boi, x, eoi), dim=1)
         x = x / self.scaling_factor
         return x

 import math
 from torch.nn import LayerNorm
 def standard_attention(query_layer, key_layer, value_layer, scaling_attention_score=True):
     if scaling_attention_score:
         query_layer = query_layer / math.sqrt(query_layer.shape[-1])
     context_layer = torch.matmul(attention_probs, value_layer)
     return context_layer
 def attention_fn_default(query_layer, key_layer, value_layer, scaling_attention_score=True):
     if int(torch.__version__.split('.')[0]) >= 2 and scaling_attention_score:
         # Pytorch 2.0 attention uses very much memory if attention_mask is float, and has NaN bug if attention_mask is None.
         attn_output = torch.nn.functional.scaled_dot_product_attention(
+            query_layer, key_layer, value_layer,
             attn_mask=None,
             dropout_p=0.,
             is_causal=False
             query_layer, key_layer, value_layer, scaling_attention_score=scaling_attention_score
         )
 class PatchEmbedding(nn.Module):
     def __init__(self, config):
         super().__init__()
+        self.proj = nn.Conv2d(config.in_channels, config.hidden_size, kernel_size=config.patch_size,
+                              stride=config.patch_size)
         self.cls_embedding = nn.Parameter(torch.zeros(1, config.hidden_size))
         self.position_embedding = nn.Embedding(config.num_positions, config.hidden_size)
         qkv = self.query_key_value(x)
         qkv = qkv.reshape(B, L, 3, self.num_heads, -1).permute(2, 0, 3, 1, 4)  # 3, B, H, L, D
         q, k, v = qkv[0], qkv[1], qkv[2]
         out = attention_fn_default(
             q, k, v
         )
+        output = self.dense(out.transpose(1, 2).view(B, L, -1))
         output = self.output_dropout(output)
         return output
         attention_output = self.input_layernorm(self.attention(attention_input))
         hidden_states = attention_input + attention_output
         mlp_input = hidden_states
+        # https://github.com/THUDM/GLM-4/issues/350
+        mlp_output = self.post_attention_layernorm(self.mlp(mlp_input)).to(mlp_input.device)
         output = mlp_input + mlp_output
         return output
         self.patch_embedding = PatchEmbedding(vision_config)
         self.transformer = Transformer(vision_config)
         self.linear_proj = GLU(config, in_features=config.hidden_size)
+        self.conv = nn.Conv2d(in_channels=vision_config.hidden_size, out_channels=config.hidden_size, kernel_size=2,
+                              stride=2)
         self.boi = nn.Parameter(torch.zeros(1, 1, config.hidden_size))
         self.eoi = nn.Parameter(torch.zeros(1, 1, config.hidden_size))
         self.scaling_factor = vision_config.scaling_factor
         x = x[:, 1:]
         b, s, h = x.shape
+        grid_size = int(s ** 0.5)
         x = x.view(b, grid_size, grid_size, h).permute(0, 3, 1, 2)
         x = self.conv(x)
         x = x.flatten(2).transpose(1, 2)
         x = self.linear_proj(x)
+        # https://github.com/THUDM/GLM-4/issues/350
+        boi = self.boi.expand(x.shape[0], -1, -1).to(x.device)
+        eoi = self.eoi.expand(x.shape[0], -1, -1).to(x.device)
         x = torch.cat((boi, x, eoi), dim=1)
         x = x / self.scaling_factor
         return x