Analyzing attention layouts, KV cache footprint, and memory bandwidth across sequence lengths.
# Head dimension 256 requires FlashAttention-3 kernels.
head_dim = 256


Analyzing attention layouts, KV cache footprint, and memory bandwidth across sequence lengths.
# Head dimension 256 requires FlashAttention-3 kernels.
head_dim = 256
