Use attention dropout during training (#1)

Files changed (1) hide show

modeling_bert.py CHANGED Viewed

@@ -282,7 +282,8 @@ class JinaBertSelfAttention(nn.Module):
         self.layer_norm_q = nn.LayerNorm(config.hidden_size, eps=config.layer_norm_eps)
         self.layer_norm_k = nn.LayerNorm(config.hidden_size, eps=config.layer_norm_eps)
-        self.dropout = nn.Dropout(config.attention_probs_dropout_prob)
         self.position_embedding_type = position_embedding_type or getattr(
             config, "position_embedding_type", "absolute"
         )
@@ -357,7 +358,8 @@ class JinaBertSelfAttention(nn.Module):
         if self.attn_implementation == 'torch' and scaled_dot_product_attention is not None:
             b, _, s, _ = query_layer.shape
             new_bias = attention_mask + bias
-            attn = scaled_dot_product_attention(query_layer, key_layer, value_layer, new_bias)
             attn = attn.permute(0, 2, 1, 3).contiguous()
             return (attn.view(b, s, self.all_head_size),)

         self.layer_norm_q = nn.LayerNorm(config.hidden_size, eps=config.layer_norm_eps)
         self.layer_norm_k = nn.LayerNorm(config.hidden_size, eps=config.layer_norm_eps)
+        self.dropout_p = config.attention_probs_dropout_prob
+        self.dropout = nn.Dropout(self.dropout_p)
         self.position_embedding_type = position_embedding_type or getattr(
             config, "position_embedding_type", "absolute"
         )
         if self.attn_implementation == 'torch' and scaled_dot_product_attention is not None:
             b, _, s, _ = query_layer.shape
             new_bias = attention_mask + bias
+            dropout_p = self.dropout_p if self.training else 0.0
+            attn = scaled_dot_product_attention(query_layer, key_layer, value_layer, new_bias, dropout_p=dropout_p)
             attn = attn.permute(0, 2, 1, 3).contiguous()
             return (attn.view(b, s, self.all_head_size),)