MCPcopy Create free account

hub / github.com/Repeerc/flash-attention-v2-RDNA3-minimal / functions

Functions124 in github.com/Repeerc/flash-attention-v2-RDNA3-minimal

↓ 29 callersMethodbackward
(ctx, do)
triton_fused_attention.py:480
↓ 19 callersMethodforward
(ctx, q, k, v, causal, sm_scale)
triton_fused_attention.py:443
↓ 4 callersFunctionpad_to_multiple
(tensor, multiple, dim=-1, val = 0)
pure_torch_ver.py:9
↓ 3 callersMethodbackward
(ctx, do)
bench_with_triton_ck_BNHD_linux.py:137
↓ 3 callersFunctionftt_rocm
(q, k, v)
bench_with_sdpa_BNHD.py:104
↓ 3 callersFunctionftt_rocm
(q, k, v)
bench_with_sdpa_bf16.py:98
↓ 3 callersFunctionftt_rocm
(q, k, v)
bench_with_sdpa.py:97
↓ 3 callersFunctionftt_rocm
(q, k, v)
bench_with_sdpa_bf16_BNHD.py:102
↓ 3 callersFunctionftt_triton
(q, k, v)
bench_with_triton.py:154
↓ 3 callersFunctionfttn_rocwmma
(q, k, v=None)
bench_with_triton.py:109
↓ 3 callersFunctionpad_to_multiple
(tensor, multiple, dim=-1, val = 0)
bench_with_triton_ck_BNHD.py:74
↓ 3 callersFunctionpad_to_multiple
(tensor, multiple, dim=-1, val = 0)
bench_with_triton_ck_BNHD_linux.py:161
↓ 3 callersFunctionpad_to_multiple
(tensor, multiple, dim=-1, val = 0)
bench_with_triton_ck.py:73
↓ 3 callersFunctionpad_to_multiple
(tensor, multiple, dim=-1, val = 0)
bench_with_triton_ck_linux.py:72
↓ 3 callersFunctionpad_to_multiple
(tensor, multiple, dim=-1, val = 0)
bench_with_triton.py:130
↓ 3 callersFunctionsdp_pt
(q, k, v=None)
bench_with_sdpa_BNHD.py:63
↓ 3 callersFunctionsdp_pt
(q, k, v=None)
bench_with_sdpa_bf16.py:66
↓ 3 callersFunctionsdp_pt
(q, k, v=None)
sdpa_test.py:61
↓ 3 callersFunctionsdp_pt
(q, k, v=None)
bench_with_sdpa.py:65
↓ 3 callersFunctionsdp_pt
(q, k, v=None)
bench_with_sdpa_bf16_BNHD.py:63
↓ 3 callersFunctionsdp_pt
(q, k, v=None)
bench_with_triton.py:71
↓ 2 callersFunction_attn_bwd_dkdv
(dk, dv, # Q, k, v, sm_scale, # DO, # M, D, #
triton_fused_attention.py:211
↓ 2 callersFunction_attn_bwd_dq
(dq, q, K, V, # do, m, D, # shared by Q/K/V/DO. stride_tok
triton_fused_attention.py:264
↓ 2 callersFunction_attn_fwd_inner
(acc, l_i, m_i, q, # K_block_ptr, V_block_ptr, # start_m, qk_scale,
triton_fused_attention.py:28
↓ 2 callersMethodbackward
(ctx, do)
precision_test_fp32ver.py:79
↓ 2 callersMethodbackward
(ctx, do)
pure_torch_ver.py:94
↓ 2 callersFunctionftt_triton_fwd
(q,k,v)
bench_with_triton.py:129
↓ 2 callersFunctionfttn_ck
(q, k, v=None)
bench_with_triton_ck_BNHD.py:124
↓ 2 callersFunctionfttn_ck
(q, k, v=None)
bench_with_triton_ck_BNHD_linux.py:211
↓ 2 callersFunctionfttn_ck
(q, k, v=None)
bench_with_ck_bf16_BNHD.py:106
↓ 2 callersFunctionfttn_ck
(q, k, v=None)
bench_with_ck_BNHD.py:107
↓ 2 callersFunctionfttn_ck
(q, k, v=None)
bench_with_triton_ck.py:121
↓ 2 callersFunctionfttn_ck
(q, k, v=None)
bench_with_ck.py:121
↓ 2 callersFunctionfttn_ck
(q, k, v=None)
bench_with_triton_ck_linux.py:120
↓ 2 callersFunctionfttn_rocwmma
(q, k, v=None)
bench_with_triton_ck_BNHD.py:118
↓ 2 callersFunctionfttn_rocwmma
(q, k, v=None)
bench_with_triton_ck_BNHD_linux.py:205
↓ 2 callersFunctionfttn_rocwmma
(q, k, v=None)
bench_with_ck_bf16_BNHD.py:96
↓ 2 callersFunctionfttn_rocwmma
(q, k, v=None)
bench_with_ck_BNHD.py:97
↓ 2 callersFunctionfttn_rocwmma
(q, k, v=None)
bench_with_triton_ck.py:115
↓ 2 callersFunctionfttn_rocwmma
(q, k, v=None)
bench_with_ck.py:115
↓ 2 callersFunctionfttn_rocwmma
(q, k, v=None)
bench_with_triton_ck_linux.py:114
↓ 2 callersFunctionis_hip
()
triton_fused_attention.py:23
↓ 2 callersFunctionsdp_pt
(q, k, v=None)
bench_with_triton_ck_BNHD.py:86
↓ 2 callersFunctionsdp_pt
(q, k, v=None)
bench_with_triton_ck_BNHD_linux.py:173
↓ 2 callersFunctionsdp_pt
(q, k, v=None)
bench_with_ck_bf16_BNHD.py:84
↓ 2 callersFunctionsdp_pt
(q, k, v=None)
bench_with_ck_BNHD.py:85
↓ 2 callersFunctionsdp_pt
(q, k, v=None)
bench_with_triton_ck.py:85
↓ 2 callersFunctionsdp_pt
(q, k, v=None)
bench_with_ck.py:85
↓ 2 callersFunctionsdp_pt
(q, k, v=None)
bench_with_triton_ck_linux.py:84
↓ 1 callersFunctionftt_bwd
(q, k, v, O, dO)
bench_with_sdpa_BNHD.py:91
↓ 1 callersFunctionftt_bwd
(q, k, v, O, dO)
bench_with_sdpa_bf16.py:85
↓ 1 callersFunctionftt_bwd
(q, k, v, O, dO)
bench_with_sdpa.py:84
↓ 1 callersFunctionftt_bwd
(q, k, v, O, dO)
bench_with_sdpa_bf16_BNHD.py:89
↓ 1 callersFunctionftt_triton
(q, k, v=None)
bench_with_triton_ck_BNHD.py:157
↓ 1 callersFunctionftt_triton
(q, k, v=None)
bench_with_triton_ck_BNHD_linux.py:244
↓ 1 callersFunctionftt_triton
(q, k, v=None)
bench_with_triton_ck.py:154
↓ 1 callersFunctionftt_triton
(q, k, v=None)
bench_with_triton_ck_linux.py:153
↓ 1 callersFunctionftt_triton_fwd_bwd
(q, k, v, dO)
bench_with_triton.py:116
↓ 1 callersFunctionfttn_rocwmma_fwd_bwd
(q, k, v, dO)
bench_with_triton.py:96
↓ 1 callersFunctionfwd
(q,k,v)
RGP_Capture.py:71
↓ 1 callersFunctionsdp_bwd
(q, k, v, O, dO)
bench_with_sdpa_BNHD.py:75
↓ 1 callersFunctionsdp_bwd
(q, k, v, O, dO)
bench_with_sdpa_bf16.py:75
↓ 1 callersFunctionsdp_bwd
(q, k, v, O, dO)
sdpa_test.py:70
↓ 1 callersFunctionsdp_bwd
(q, k, v, O, dO)
bench_with_sdpa.py:74
↓ 1 callersFunctionsdp_bwd
(q, k, v, O, dO)
bench_with_sdpa_bf16_BNHD.py:73
↓ 1 callersFunctionsdp_fwd_bwd
(q, k, v, dO)
bench_with_triton.py:80
FunctionPYBIND11_MODULE
gemm_test/host.cpp:10
FunctionPYBIND11_MODULE
rocwmma_fattn/host.cpp:60
Function_attn_bwd
(Q, K, V, sm_scale, # DO, # DQ, DK, DV, # M, D, # s
triton_fused_attention.py:311
Function_attn_bwd_preprocess
(O, DO, # Delta, # Z, H, N_CTX, #
triton_fused_attention.py:193
Function_attn_fwd
(Q, K, V, sm_scale, M, Out, # stride_qz, stride_qh, stride_qm, stride_qk, # stri
triton_fused_attention.py:102
Functionbackward
rocwmma_fattn/host.cpp:47
Methodbackward
(ctx, do)
rocwmma_fattn/FlashAttn.py:80
Functioncount_time
(func)
bench_with_triton_ck_BNHD.py:24
Functioncount_time
(func)
bench_with_sdpa_BNHD.py:15
Functioncount_time
(func)
bench_with_triton_ck_BNHD_linux.py:60
Functioncount_time
(func)
bench_with_sdpa_bf16.py:15
Functioncount_time
(func)
sdpa_test.py:14
Functioncount_time
(func)
bench_with_ck_bf16_BNHD.py:24
Functioncount_time
(func)
bench_with_ck_BNHD.py:25
Functioncount_time
(func)
bench_with_triton_ck.py:25
Functioncount_time
(func)
bench_with_sdpa.py:14
Functioncount_time
(func)
bench_with_sdpa_bf16_BNHD.py:13
Functioncount_time
(func)
bench_with_ck.py:26
Functioncount_time
(func)
bench_with_triton_ck_linux.py:25
Functioncount_time
(func)
bench_with_triton.py:20
Functionforward
rocwmma_fattn/host.cpp:30
Methodforward
(ctx, q, k, v, mask=None, causal=None, *args, **kwargs)
bench_with_triton_ck_BNHD_linux.py:108
Methodforward
(ctx, q, k, v, mask=None, causal=None, *args, **kwargs)
precision_test_fp32ver.py:57
Methodforward
(ctx, q, k, v, mask=None, causal=False, Br=64, Bc=256)
pure_torch_ver.py:24
Methodforward
(ctx, q, k, v, mask=None, causal=None, scale=None, BNHD_fmt=False, *args, **kwargs)
rocwmma_fattn/FlashAttn.py:49
Functionftt_triton_bwd
(q, k, v, O, dO, L)
bench_with_triton_ck_BNHD.py:142
Functionftt_triton_bwd
(q, k, v, O, dO, L)
bench_with_triton_ck_BNHD_linux.py:229
Functionftt_triton_bwd
(q, k, v, O, dO, L)
bench_with_triton_ck.py:139
Functionftt_triton_bwd
(q, k, v, O, dO, L)
bench_with_triton_ck_linux.py:138
Functionfttn_rocwmma_bwd
(q, k, v, O, dO)
bench_with_triton_ck_BNHD.py:107
Functionfttn_rocwmma_bwd
(q, k, v, O, dO)
bench_with_triton_ck_BNHD_linux.py:194
Functionfttn_rocwmma_bwd
(q, k, v, O, dO)
bench_with_triton_ck.py:104
Functionfttn_rocwmma_bwd
(q, k, v, O, dO)
bench_with_ck.py:104
Functionfttn_rocwmma_bwd
(q, k, v, O, dO)
bench_with_triton_ck_linux.py:103
next →1–100 of 124, ranked by callers