Spaces:

Dovakiins
/

qwerrwe

Build error

winglian commited on Jul 20, 2023

Commit

a032c9f

•

1 Parent(s): b06d3e3

fix sdp attention to use the flash/mem-efficient context manaager

Files changed (1) hide show

src/axolotl/monkeypatch/llama_attn_hijack_xformers.py CHANGED Viewed

@@ -184,14 +184,15 @@ def sdp_attention_forward(
     # We only apply sdp attention if we don't need to output the whole attention matrix
     if not output_attentions:
-        attn_output = torch.nn.functional.scaled_dot_product_attention(
-            query_states,
-            key_states,
-            value_states,
-            attn_mask=attention_mask,
-            is_causal=False,
-        )
-        attn_weights = None
     else:
         attn_weights = torch.matmul(
             query_states, key_states.transpose(2, 3)

     # We only apply sdp attention if we don't need to output the whole attention matrix
     if not output_attentions:
+        with torch.backends.cuda.sdp_kernel():
+            attn_output = torch.nn.functional.scaled_dot_product_attention(
+                query_states,
+                key_states,
+                value_states,
+                attn_mask=attention_mask,
+                is_causal=False,
+            )
+            attn_weights = None
     else:
         attn_weights = torch.matmul(
             query_states, key_states.transpose(2, 3)