OpenGVLab
/

InternVL2-Llama3-76B-AWQ

@@ -30,7 +30,7 @@ LMDeploy supports the following NVIDIA GPU for W4A16 inference:
 Before proceeding with the quantization and inference, please ensure that lmdeploy is installed.
 ```shell
-pip install lmdeploy
 ```
 This article comprises the following sections:

 Before proceeding with the quantization and inference, please ensure that lmdeploy is installed.
 ```shell
+pip install lmdeploy==0.5.3
 ```
 This article comprises the following sections:

modeling_intern_vit.py CHANGED Viewed

@@ -20,18 +20,12 @@ from transformers.utils import logging
 from .configuration_intern_vit import InternVisionConfig
 try:
-    try:  # v1
-        from flash_attn.flash_attn_interface import \
-            flash_attn_unpadded_qkvpacked_func
-    except:  # v2
-        from flash_attn.flash_attn_interface import \
-            flash_attn_varlen_qkvpacked_func as flash_attn_unpadded_qkvpacked_func
     from flash_attn.bert_padding import pad_input, unpad_input
     has_flash_attn = True
 except:
-    print('FlashAttention is not installed.')
     has_flash_attn = False
 logger = logging.get_logger(__name__)
@@ -74,7 +68,7 @@ class FlashAttention(nn.Module):
                 max_s = seqlen
                 cu_seqlens = torch.arange(0, (batch_size + 1) * seqlen, step=seqlen, dtype=torch.int32,
                                           device=qkv.device)
-                output = flash_attn_unpadded_qkvpacked_func(
                     qkv, cu_seqlens, max_s, self.dropout_p if self.training else 0.0,
                     softmax_scale=self.softmax_scale, causal=causal
                 )
@@ -84,7 +78,7 @@ class FlashAttention(nn.Module):
                 x = rearrange(qkv, 'b s three h d -> b s (three h d)')
                 x_unpad, indices, cu_seqlens, max_s = unpad_input(x, key_padding_mask)
                 x_unpad = rearrange(x_unpad, 'nnz (three h d) -> nnz three h d', three=3, h=nheads)
-                output_unpad = flash_attn_unpadded_qkvpacked_func(
                     x_unpad, cu_seqlens, max_s, self.dropout_p if self.training else 0.0,
                     softmax_scale=self.softmax_scale, causal=causal
                 )
@@ -93,7 +87,7 @@ class FlashAttention(nn.Module):
                                    'b s (h d) -> b s h d', h=nheads)
         else:
             assert max_s is not None
-            output = flash_attn_unpadded_qkvpacked_func(
                 qkv, cu_seqlens, max_s, self.dropout_p if self.training else 0.0,
                 softmax_scale=self.softmax_scale, causal=causal
             )

 from .configuration_intern_vit import InternVisionConfig
 try:
     from flash_attn.bert_padding import pad_input, unpad_input
+    from flash_attn.flash_attn_interface import \
+        flash_attn_varlen_qkvpacked_func
     has_flash_attn = True
 except:
+    print('FlashAttention2 is not installed.')
     has_flash_attn = False
 logger = logging.get_logger(__name__)
                 max_s = seqlen
                 cu_seqlens = torch.arange(0, (batch_size + 1) * seqlen, step=seqlen, dtype=torch.int32,
                                           device=qkv.device)
+                output = flash_attn_varlen_qkvpacked_func(
                     qkv, cu_seqlens, max_s, self.dropout_p if self.training else 0.0,
                     softmax_scale=self.softmax_scale, causal=causal
                 )
                 x = rearrange(qkv, 'b s three h d -> b s (three h d)')
                 x_unpad, indices, cu_seqlens, max_s = unpad_input(x, key_padding_mask)
                 x_unpad = rearrange(x_unpad, 'nnz (three h d) -> nnz three h d', three=3, h=nheads)
+                output_unpad = flash_attn_varlen_qkvpacked_func(
                     x_unpad, cu_seqlens, max_s, self.dropout_p if self.training else 0.0,
                     softmax_scale=self.softmax_scale, causal=causal
                 )
                                    'b s (h d) -> b s h d', h=nheads)
         else:
             assert max_s is not None
+            output = flash_attn_varlen_qkvpacked_func(
                 qkv, cu_seqlens, max_s, self.dropout_p if self.training else 0.0,
                 softmax_scale=self.softmax_scale, causal=causal
             )