Initial K3 snapshot: 0.5B KDA/MLA/MoE train path
Standalone tree split from LLMRL/projects/kda. Includes Triton dt_bias backward fix, train_k3 --preset 0.5b, SFT, Docker runtime, and tests.
This commit is contained in:
@@ -0,0 +1 @@
|
||||
# Vendored FLA common kernels used by KDA.
|
||||
@@ -0,0 +1,806 @@
|
||||
# Copyright (c) 2023-2026, Songlin Yang, Yu Zhang, Zhiyuan Li
|
||||
#
|
||||
# This source code is licensed under the MIT license found in the
|
||||
# LICENSE file in the root directory of this source tree.
|
||||
# For a list of all contributors, visit:
|
||||
# https://github.com/fla-org/flash-linear-attention/graphs/contributors
|
||||
|
||||
import torch
|
||||
import triton
|
||||
import triton.language as tl
|
||||
|
||||
from kda._fla.ops.backends import dispatch
|
||||
from kda._fla.ops.utils import prepare_chunk_indices, prepare_chunk_offsets
|
||||
from kda._fla.ops.utils.cache import fla_cache_autotune
|
||||
from kda._fla.ops.utils.op import exp2
|
||||
from kda._fla.utils import (
|
||||
IS_INTEL,
|
||||
IS_NVIDIA_BLACKWELL,
|
||||
IS_NVIDIA_HOPPER,
|
||||
autotune_cache_kwargs,
|
||||
check_shared_mem,
|
||||
)
|
||||
|
||||
NUM_WARPS = [2, 4] if IS_NVIDIA_HOPPER else [2, 4, 8, 16]
|
||||
|
||||
# TODO: Triton mainline fixes a Blackwell tl.dot recurrence race.
|
||||
# Keep this kernel on num_warps=2 for Blackwell until Triton 3.8 is released
|
||||
# and we re-validate the wider config space.
|
||||
# Intel needs more warps than NVIDIA here: 8 warps is ~1.5x faster than the best
|
||||
# config reachable under the [2, 4] cap.
|
||||
if IS_NVIDIA_BLACKWELL:
|
||||
GATED_DELTA_RULE_FWD_H_NUM_WARPS = [2]
|
||||
elif IS_INTEL:
|
||||
GATED_DELTA_RULE_FWD_H_NUM_WARPS = [2, 4, 8, 16]
|
||||
else:
|
||||
GATED_DELTA_RULE_FWD_H_NUM_WARPS = [2, 4]
|
||||
|
||||
|
||||
@triton.heuristics({
|
||||
'USE_G': lambda args: args['g'] is not None,
|
||||
'USE_GK': lambda args: args['gk'] is not None,
|
||||
'USE_INITIAL_STATE': lambda args: args['h0'] is not None,
|
||||
'STORE_FINAL_STATE': lambda args: args['ht'] is not None,
|
||||
'SAVE_NEW_VALUE': lambda args: args['v_new'] is not None,
|
||||
'IS_VARLEN': lambda args: args['cu_seqlens'] is not None,
|
||||
})
|
||||
@fla_cache_autotune(
|
||||
configs=[
|
||||
triton.Config({'BV': BV}, num_warps=num_warps, num_stages=num_stages)
|
||||
for num_warps in GATED_DELTA_RULE_FWD_H_NUM_WARPS
|
||||
for num_stages in ([2, 3, 4] if check_shared_mem('ampere') else [2, 1])
|
||||
for BV in ([32, 64] if check_shared_mem('ada') else [32])
|
||||
],
|
||||
key=['H', 'HV', 'K', 'V', 'BT', 'STATE_V_FIRST'],
|
||||
**autotune_cache_kwargs,
|
||||
)
|
||||
@triton.jit(do_not_specialize=['T'])
|
||||
def chunk_gated_delta_rule_fwd_kernel_h_blockdim64(
|
||||
k,
|
||||
v,
|
||||
w,
|
||||
v_new,
|
||||
g,
|
||||
gk,
|
||||
h,
|
||||
h0,
|
||||
ht,
|
||||
cu_seqlens,
|
||||
chunk_offsets,
|
||||
T,
|
||||
H: tl.constexpr,
|
||||
HV: tl.constexpr,
|
||||
K: tl.constexpr,
|
||||
V: tl.constexpr,
|
||||
BT: tl.constexpr,
|
||||
BV: tl.constexpr,
|
||||
USE_G: tl.constexpr,
|
||||
USE_GK: tl.constexpr,
|
||||
USE_INITIAL_STATE: tl.constexpr,
|
||||
STORE_FINAL_STATE: tl.constexpr,
|
||||
SAVE_NEW_VALUE: tl.constexpr,
|
||||
STATE_V_FIRST: tl.constexpr,
|
||||
IS_VARLEN: tl.constexpr,
|
||||
):
|
||||
pid = tl.program_id(0)
|
||||
NV = tl.cdiv(V, BV)
|
||||
i_v, i_nh = pid % NV, (pid // NV).to(tl.int64)
|
||||
i_n, i_h = i_nh // HV, i_nh % HV
|
||||
if IS_VARLEN:
|
||||
bos, eos = tl.load(cu_seqlens + i_n).to(tl.int64), tl.load(cu_seqlens + i_n + 1).to(tl.int64)
|
||||
T = eos - bos
|
||||
NT = tl.cdiv(T, BT)
|
||||
boh = tl.load(chunk_offsets + i_n).to(tl.int64)
|
||||
else:
|
||||
bos, eos = i_n * T, i_n * T + T
|
||||
NT = tl.cdiv(T, BT)
|
||||
boh = i_n * NT
|
||||
|
||||
if STATE_V_FIRST:
|
||||
b_h1 = tl.zeros([BV, 64], dtype=tl.float32)
|
||||
if K > 64:
|
||||
b_h2 = tl.zeros([BV, 64], dtype=tl.float32)
|
||||
if K > 128:
|
||||
b_h3 = tl.zeros([BV, 64], dtype=tl.float32)
|
||||
if K > 192:
|
||||
b_h4 = tl.zeros([BV, 64], dtype=tl.float32)
|
||||
else:
|
||||
b_h1 = tl.zeros([64, BV], dtype=tl.float32)
|
||||
if K > 64:
|
||||
b_h2 = tl.zeros([64, BV], dtype=tl.float32)
|
||||
if K > 128:
|
||||
b_h3 = tl.zeros([64, BV], dtype=tl.float32)
|
||||
if K > 192:
|
||||
b_h4 = tl.zeros([64, BV], dtype=tl.float32)
|
||||
|
||||
# calculate offset
|
||||
h += (boh * HV + i_h).to(tl.int64) * K*V
|
||||
v += (bos * HV + i_h).to(tl.int64) * V
|
||||
k += (bos * H + i_h // (HV // H)).to(tl.int64) * K
|
||||
w += (bos * HV + i_h).to(tl.int64) * K
|
||||
if SAVE_NEW_VALUE:
|
||||
v_new += (bos * HV + i_h).to(tl.int64) * V
|
||||
|
||||
if USE_INITIAL_STATE:
|
||||
h0 = h0 + i_nh * K*V
|
||||
if STORE_FINAL_STATE:
|
||||
ht = ht + i_nh * K*V
|
||||
|
||||
# load initial state
|
||||
o_v = i_v * BV + tl.arange(0, BV)
|
||||
m_v = o_v < V
|
||||
o_k1 = tl.arange(0, 64)
|
||||
m_k1 = o_k1 < K
|
||||
o_k2 = 64 + o_k1
|
||||
m_k2 = o_k2 < K
|
||||
o_k3 = 128 + o_k1
|
||||
m_k3 = o_k3 < K
|
||||
o_k4 = 192 + o_k1
|
||||
m_k4 = o_k4 < K
|
||||
if USE_INITIAL_STATE:
|
||||
if STATE_V_FIRST:
|
||||
p_h0_1 = h0 + o_v[:, None] * K + o_k1[None, :]
|
||||
m_h0_1 = m_v[:, None] & m_k1[None, :]
|
||||
else:
|
||||
p_h0_1 = h0 + o_k1[:, None] * V + o_v[None, :]
|
||||
m_h0_1 = m_k1[:, None] & m_v[None, :]
|
||||
b_h1 += tl.load(p_h0_1, mask=m_h0_1, other=0.0).to(tl.float32)
|
||||
if K > 64:
|
||||
if STATE_V_FIRST:
|
||||
p_h0_2 = h0 + o_v[:, None] * K + o_k2[None, :]
|
||||
m_h0_2 = m_v[:, None] & m_k2[None, :]
|
||||
else:
|
||||
p_h0_2 = h0 + o_k2[:, None] * V + o_v[None, :]
|
||||
m_h0_2 = m_k2[:, None] & m_v[None, :]
|
||||
b_h2 += tl.load(p_h0_2, mask=m_h0_2, other=0.0).to(tl.float32)
|
||||
if K > 128:
|
||||
if STATE_V_FIRST:
|
||||
p_h0_3 = h0 + o_v[:, None] * K + o_k3[None, :]
|
||||
m_h0_3 = m_v[:, None] & m_k3[None, :]
|
||||
else:
|
||||
p_h0_3 = h0 + o_k3[:, None] * V + o_v[None, :]
|
||||
m_h0_3 = m_k3[:, None] & m_v[None, :]
|
||||
b_h3 += tl.load(p_h0_3, mask=m_h0_3, other=0.0).to(tl.float32)
|
||||
if K > 192:
|
||||
if STATE_V_FIRST:
|
||||
p_h0_4 = h0 + o_v[:, None] * K + o_k4[None, :]
|
||||
m_h0_4 = m_v[:, None] & m_k4[None, :]
|
||||
else:
|
||||
p_h0_4 = h0 + o_k4[:, None] * V + o_v[None, :]
|
||||
m_h0_4 = m_k4[:, None] & m_v[None, :]
|
||||
b_h4 += tl.load(p_h0_4, mask=m_h0_4, other=0.0).to(tl.float32)
|
||||
|
||||
# main recurrence
|
||||
for i_t in range(NT):
|
||||
i_t_int64 = i_t.to(tl.int64)
|
||||
o_t = i_t * BT + tl.arange(0, BT)
|
||||
m_t = o_t < T
|
||||
if STATE_V_FIRST:
|
||||
p_h1 = h + i_t_int64 * HV*K*V + o_v[:, None] * K + o_k1[None, :]
|
||||
m_h1 = m_v[:, None] & m_k1[None, :]
|
||||
else:
|
||||
p_h1 = h + i_t_int64 * HV*K*V + o_k1[:, None] * V + o_v[None, :]
|
||||
m_h1 = m_k1[:, None] & m_v[None, :]
|
||||
tl.store(p_h1, b_h1.to(p_h1.dtype.element_ty), mask=m_h1)
|
||||
if K > 64:
|
||||
if STATE_V_FIRST:
|
||||
p_h2 = h + i_t_int64 * HV*K*V + o_v[:, None] * K + o_k2[None, :]
|
||||
m_h2 = m_v[:, None] & m_k2[None, :]
|
||||
else:
|
||||
p_h2 = h + i_t_int64 * HV*K*V + o_k2[:, None] * V + o_v[None, :]
|
||||
m_h2 = m_k2[:, None] & m_v[None, :]
|
||||
tl.store(p_h2, b_h2.to(p_h2.dtype.element_ty), mask=m_h2)
|
||||
if K > 128:
|
||||
if STATE_V_FIRST:
|
||||
p_h3 = h + i_t_int64 * HV*K*V + o_v[:, None] * K + o_k3[None, :]
|
||||
m_h3 = m_v[:, None] & m_k3[None, :]
|
||||
else:
|
||||
p_h3 = h + i_t_int64 * HV*K*V + o_k3[:, None] * V + o_v[None, :]
|
||||
m_h3 = m_k3[:, None] & m_v[None, :]
|
||||
tl.store(p_h3, b_h3.to(p_h3.dtype.element_ty), mask=m_h3)
|
||||
if K > 192:
|
||||
if STATE_V_FIRST:
|
||||
p_h4 = h + i_t_int64 * HV*K*V + o_v[:, None] * K + o_k4[None, :]
|
||||
m_h4 = m_v[:, None] & m_k4[None, :]
|
||||
else:
|
||||
p_h4 = h + i_t_int64 * HV*K*V + o_k4[:, None] * V + o_v[None, :]
|
||||
m_h4 = m_k4[:, None] & m_v[None, :]
|
||||
tl.store(p_h4, b_h4.to(p_h4.dtype.element_ty), mask=m_h4)
|
||||
|
||||
p_w = w + o_t[:, None] * (HV*K) + o_k1[None, :]
|
||||
b_w = tl.load(p_w, mask=m_t[:, None] & m_k1[None, :], other=0.0)
|
||||
if STATE_V_FIRST:
|
||||
b_v = tl.dot(b_w, tl.trans(b_h1).to(b_w.dtype))
|
||||
else:
|
||||
b_v = tl.dot(b_w, b_h1.to(b_w.dtype))
|
||||
if K > 64:
|
||||
p_w = w + o_t[:, None] * (HV*K) + o_k2[None, :]
|
||||
b_w = tl.load(p_w, mask=m_t[:, None] & m_k2[None, :], other=0.0)
|
||||
if STATE_V_FIRST:
|
||||
b_v += tl.dot(b_w, tl.trans(b_h2).to(b_w.dtype))
|
||||
else:
|
||||
b_v += tl.dot(b_w, b_h2.to(b_w.dtype))
|
||||
if K > 128:
|
||||
p_w = w + o_t[:, None] * (HV*K) + o_k3[None, :]
|
||||
b_w = tl.load(p_w, mask=m_t[:, None] & m_k3[None, :], other=0.0)
|
||||
if STATE_V_FIRST:
|
||||
b_v += tl.dot(b_w, tl.trans(b_h3).to(b_w.dtype))
|
||||
else:
|
||||
b_v += tl.dot(b_w, b_h3.to(b_w.dtype))
|
||||
if K > 192:
|
||||
p_w = w + o_t[:, None] * (HV*K) + o_k4[None, :]
|
||||
b_w = tl.load(p_w, mask=m_t[:, None] & m_k4[None, :], other=0.0)
|
||||
if STATE_V_FIRST:
|
||||
b_v += tl.dot(b_w, tl.trans(b_h4).to(b_w.dtype))
|
||||
else:
|
||||
b_v += tl.dot(b_w, b_h4.to(b_w.dtype))
|
||||
p_v = v + o_t[:, None] * (HV*V) + o_v[None, :]
|
||||
b_v = tl.load(p_v, mask=m_t[:, None] & m_v[None, :], other=0.0) - b_v
|
||||
|
||||
if SAVE_NEW_VALUE:
|
||||
p_v = v_new + o_t[:, None] * (HV*V) + o_v[None, :]
|
||||
tl.store(p_v, b_v.to(p_v.dtype.element_ty), mask=m_t[:, None] & m_v[None, :])
|
||||
|
||||
last_idx = min((i_t + 1) * BT, T) - 1
|
||||
if USE_G:
|
||||
b_g_last = tl.load(g + (bos * HV + last_idx * HV + i_h).to(tl.int64)).to(tl.float32)
|
||||
p_g = g + (bos * HV + i_h).to(tl.int64) + o_t * HV
|
||||
b_g = tl.load(p_g, mask=m_t, other=0.0).to(tl.float32)
|
||||
b_v = b_v * tl.where(m_t, exp2(b_g_last - b_g), 0)[:, None]
|
||||
b_g_last = exp2(b_g_last)
|
||||
b_h1 *= b_g_last
|
||||
if K > 64:
|
||||
b_h2 *= b_g_last
|
||||
if K > 128:
|
||||
b_h3 *= b_g_last
|
||||
if K > 192:
|
||||
b_h4 *= b_g_last
|
||||
|
||||
if USE_GK:
|
||||
o_k1 = tl.arange(0, 64)
|
||||
b_gk_last1 = tl.load(gk + (bos + last_idx) * HV*K + i_h * K + o_k1, mask=(o_k1 < K), other=0.).to(tl.float32)
|
||||
if STATE_V_FIRST:
|
||||
b_h1 *= exp2(b_gk_last1)[None, :]
|
||||
else:
|
||||
b_h1 *= exp2(b_gk_last1)[:, None]
|
||||
if K > 64:
|
||||
o_k2 = 64 + o_k1
|
||||
b_gk_last2 = tl.load(gk + (bos + last_idx) * HV*K + i_h * K + o_k2, mask=(o_k2 < K), other=0.).to(tl.float32)
|
||||
if STATE_V_FIRST:
|
||||
b_h2 *= exp2(b_gk_last2)[None, :]
|
||||
else:
|
||||
b_h2 *= exp2(b_gk_last2)[:, None]
|
||||
if K > 128:
|
||||
o_k3 = 128 + o_k1
|
||||
b_gk_last3 = tl.load(gk + (bos + last_idx) * HV*K + i_h * K + o_k3, mask=(o_k3 < K), other=0.).to(tl.float32)
|
||||
if STATE_V_FIRST:
|
||||
b_h3 *= exp2(b_gk_last3)[None, :]
|
||||
else:
|
||||
b_h3 *= exp2(b_gk_last3)[:, None]
|
||||
if K > 192:
|
||||
o_k4 = 192 + o_k1
|
||||
b_gk_last4 = tl.load(gk + (bos + last_idx) * HV*K + i_h * K + o_k4, mask=(o_k4 < K), other=0.).to(tl.float32)
|
||||
if STATE_V_FIRST:
|
||||
b_h4 *= exp2(b_gk_last4)[None, :]
|
||||
else:
|
||||
b_h4 *= exp2(b_gk_last4)[:, None]
|
||||
b_v = b_v.to(k.dtype.element_ty)
|
||||
|
||||
p_k = k + o_k1[:, None] + o_t[None, :] * (H*K)
|
||||
b_k = tl.load(p_k, mask=m_k1[:, None] & m_t[None, :], other=0.0)
|
||||
if STATE_V_FIRST:
|
||||
b_h1 += tl.trans(tl.dot(b_k, b_v))
|
||||
else:
|
||||
b_h1 += tl.dot(b_k, b_v)
|
||||
if K > 64:
|
||||
p_k = k + o_k2[:, None] + o_t[None, :] * (H*K)
|
||||
b_k = tl.load(p_k, mask=m_k2[:, None] & m_t[None, :], other=0.0)
|
||||
if STATE_V_FIRST:
|
||||
b_h2 += tl.trans(tl.dot(b_k, b_v))
|
||||
else:
|
||||
b_h2 += tl.dot(b_k, b_v)
|
||||
if K > 128:
|
||||
p_k = k + o_k3[:, None] + o_t[None, :] * (H*K)
|
||||
b_k = tl.load(p_k, mask=m_k3[:, None] & m_t[None, :], other=0.0)
|
||||
if STATE_V_FIRST:
|
||||
b_h3 += tl.trans(tl.dot(b_k, b_v))
|
||||
else:
|
||||
b_h3 += tl.dot(b_k, b_v)
|
||||
if K > 192:
|
||||
p_k = k + o_k4[:, None] + o_t[None, :] * (H*K)
|
||||
b_k = tl.load(p_k, mask=m_k4[:, None] & m_t[None, :], other=0.0)
|
||||
if STATE_V_FIRST:
|
||||
b_h4 += tl.trans(tl.dot(b_k, b_v))
|
||||
else:
|
||||
b_h4 += tl.dot(b_k, b_v)
|
||||
|
||||
if STORE_FINAL_STATE:
|
||||
if STATE_V_FIRST:
|
||||
p_ht = ht + o_v[:, None] * K + o_k1[None, :]
|
||||
m_ht = m_v[:, None] & m_k1[None, :]
|
||||
else:
|
||||
p_ht = ht + o_k1[:, None] * V + o_v[None, :]
|
||||
m_ht = m_k1[:, None] & m_v[None, :]
|
||||
tl.store(p_ht, b_h1.to(p_ht.dtype.element_ty), mask=m_ht)
|
||||
if K > 64:
|
||||
if STATE_V_FIRST:
|
||||
p_ht = ht + o_v[:, None] * K + o_k2[None, :]
|
||||
m_ht = m_v[:, None] & m_k2[None, :]
|
||||
else:
|
||||
p_ht = ht + o_k2[:, None] * V + o_v[None, :]
|
||||
m_ht = m_k2[:, None] & m_v[None, :]
|
||||
tl.store(p_ht, b_h2.to(p_ht.dtype.element_ty), mask=m_ht)
|
||||
if K > 128:
|
||||
if STATE_V_FIRST:
|
||||
p_ht = ht + o_v[:, None] * K + o_k3[None, :]
|
||||
m_ht = m_v[:, None] & m_k3[None, :]
|
||||
else:
|
||||
p_ht = ht + o_k3[:, None] * V + o_v[None, :]
|
||||
m_ht = m_k3[:, None] & m_v[None, :]
|
||||
tl.store(p_ht, b_h3.to(p_ht.dtype.element_ty), mask=m_ht)
|
||||
if K > 192:
|
||||
if STATE_V_FIRST:
|
||||
p_ht = ht + o_v[:, None] * K + o_k4[None, :]
|
||||
m_ht = m_v[:, None] & m_k4[None, :]
|
||||
else:
|
||||
p_ht = ht + o_k4[:, None] * V + o_v[None, :]
|
||||
m_ht = m_k4[:, None] & m_v[None, :]
|
||||
tl.store(p_ht, b_h4.to(p_ht.dtype.element_ty), mask=m_ht)
|
||||
|
||||
|
||||
@triton.heuristics({
|
||||
'USE_G': lambda args: args['g'] is not None,
|
||||
'USE_GK': lambda args: args['gk'] is not None,
|
||||
'USE_INITIAL_STATE': lambda args: args['dh0'] is not None,
|
||||
'USE_FINAL_STATE_GRADIENT': lambda args: args['dht'] is not None,
|
||||
'IS_VARLEN': lambda args: args['cu_seqlens'] is not None,
|
||||
})
|
||||
@fla_cache_autotune(
|
||||
configs=[
|
||||
triton.Config({'BV': BV}, num_warps=num_warps, num_stages=num_stages)
|
||||
for num_warps in [2, 4]
|
||||
for num_stages in ([2, 3, 4] if check_shared_mem('ampere') else [1])
|
||||
for BV in ([32, 64] if check_shared_mem('ada') else [32])
|
||||
],
|
||||
key=['H', 'HV', 'K', 'V', 'BT', 'BV', 'USE_G', 'STATE_V_FIRST'],
|
||||
**autotune_cache_kwargs,
|
||||
)
|
||||
@triton.jit(do_not_specialize=['T'])
|
||||
def chunk_gated_delta_rule_bwd_kernel_dhu_blockdim64(
|
||||
q,
|
||||
k,
|
||||
w,
|
||||
g,
|
||||
gk,
|
||||
dht,
|
||||
dh0,
|
||||
do,
|
||||
dh,
|
||||
dv,
|
||||
dv2,
|
||||
cu_seqlens,
|
||||
chunk_offsets,
|
||||
scale,
|
||||
T,
|
||||
H: tl.constexpr,
|
||||
HV: tl.constexpr,
|
||||
K: tl.constexpr,
|
||||
V: tl.constexpr,
|
||||
BT: tl.constexpr,
|
||||
BV: tl.constexpr,
|
||||
USE_G: tl.constexpr,
|
||||
USE_GK: tl.constexpr,
|
||||
USE_INITIAL_STATE: tl.constexpr,
|
||||
USE_FINAL_STATE_GRADIENT: tl.constexpr,
|
||||
STATE_V_FIRST: tl.constexpr,
|
||||
IS_VARLEN: tl.constexpr,
|
||||
):
|
||||
pid = tl.program_id(0)
|
||||
NV = tl.cdiv(V, BV)
|
||||
i_v, i_nh = pid % NV, (pid // NV).to(tl.int64)
|
||||
i_n, i_h = i_nh // HV, i_nh % HV
|
||||
if IS_VARLEN:
|
||||
bos, eos = tl.load(cu_seqlens + i_n).to(tl.int64), tl.load(cu_seqlens + i_n + 1).to(tl.int64)
|
||||
T = eos - bos
|
||||
NT = tl.cdiv(T, BT)
|
||||
boh = tl.load(chunk_offsets + i_n).to(tl.int64)
|
||||
else:
|
||||
bos, eos = i_n * T, i_n * T + T
|
||||
NT = tl.cdiv(T, BT)
|
||||
boh = i_n * NT
|
||||
|
||||
if STATE_V_FIRST:
|
||||
b_dh1 = tl.zeros([BV, 64], dtype=tl.float32)
|
||||
if K > 64:
|
||||
b_dh2 = tl.zeros([BV, 64], dtype=tl.float32)
|
||||
if K > 128:
|
||||
b_dh3 = tl.zeros([BV, 64], dtype=tl.float32)
|
||||
if K > 192:
|
||||
b_dh4 = tl.zeros([BV, 64], dtype=tl.float32)
|
||||
else:
|
||||
b_dh1 = tl.zeros([64, BV], dtype=tl.float32)
|
||||
if K > 64:
|
||||
b_dh2 = tl.zeros([64, BV], dtype=tl.float32)
|
||||
if K > 128:
|
||||
b_dh3 = tl.zeros([64, BV], dtype=tl.float32)
|
||||
if K > 192:
|
||||
b_dh4 = tl.zeros([64, BV], dtype=tl.float32)
|
||||
|
||||
# calculate offset
|
||||
q += (bos * H + i_h // (HV // H)).to(tl.int64) * K
|
||||
k += (bos * H + i_h // (HV // H)).to(tl.int64) * K
|
||||
w += (bos * HV + i_h).to(tl.int64) * K
|
||||
do += (bos * HV + i_h).to(tl.int64) * V
|
||||
dv += (bos * HV + i_h).to(tl.int64) * V
|
||||
dv2 += (bos * HV + i_h).to(tl.int64) * V
|
||||
dh += (boh * HV + i_h).to(tl.int64) * K*V
|
||||
if USE_GK:
|
||||
gk += (bos * HV + i_h).to(tl.int64) * K
|
||||
|
||||
if USE_INITIAL_STATE:
|
||||
dh0 += i_nh * K*V
|
||||
if USE_FINAL_STATE_GRADIENT:
|
||||
dht += i_nh * K*V
|
||||
|
||||
o_v = i_v * BV + tl.arange(0, BV)
|
||||
m_v = o_v < V
|
||||
o_k1 = tl.arange(0, 64)
|
||||
m_k1 = o_k1 < K
|
||||
o_k2 = 64 + o_k1
|
||||
m_k2 = o_k2 < K
|
||||
o_k3 = 128 + o_k1
|
||||
m_k3 = o_k3 < K
|
||||
o_k4 = 192 + o_k1
|
||||
m_k4 = o_k4 < K
|
||||
if USE_FINAL_STATE_GRADIENT:
|
||||
if STATE_V_FIRST:
|
||||
p_dht1 = dht + o_v[:, None] * K + o_k1[None, :]
|
||||
m_dht1 = m_v[:, None] & m_k1[None, :]
|
||||
else:
|
||||
p_dht1 = dht + o_k1[:, None] * V + o_v[None, :]
|
||||
m_dht1 = m_k1[:, None] & m_v[None, :]
|
||||
b_dh1 += tl.load(p_dht1, mask=m_dht1, other=0.0)
|
||||
if K > 64:
|
||||
if STATE_V_FIRST:
|
||||
p_dht2 = dht + o_v[:, None] * K + o_k2[None, :]
|
||||
m_dht2 = m_v[:, None] & m_k2[None, :]
|
||||
else:
|
||||
p_dht2 = dht + o_k2[:, None] * V + o_v[None, :]
|
||||
m_dht2 = m_k2[:, None] & m_v[None, :]
|
||||
b_dh2 += tl.load(p_dht2, mask=m_dht2, other=0.0)
|
||||
if K > 128:
|
||||
if STATE_V_FIRST:
|
||||
p_dht3 = dht + o_v[:, None] * K + o_k3[None, :]
|
||||
m_dht3 = m_v[:, None] & m_k3[None, :]
|
||||
else:
|
||||
p_dht3 = dht + o_k3[:, None] * V + o_v[None, :]
|
||||
m_dht3 = m_k3[:, None] & m_v[None, :]
|
||||
b_dh3 += tl.load(p_dht3, mask=m_dht3, other=0.0)
|
||||
if K > 192:
|
||||
if STATE_V_FIRST:
|
||||
p_dht4 = dht + o_v[:, None] * K + o_k4[None, :]
|
||||
m_dht4 = m_v[:, None] & m_k4[None, :]
|
||||
else:
|
||||
p_dht4 = dht + o_k4[:, None] * V + o_v[None, :]
|
||||
m_dht4 = m_k4[:, None] & m_v[None, :]
|
||||
b_dh4 += tl.load(p_dht4, mask=m_dht4, other=0.0)
|
||||
|
||||
for i_t in range(NT - 1, -1, -1):
|
||||
i_t_int64 = i_t.to(tl.int64)
|
||||
o_t = i_t * BT + tl.arange(0, BT)
|
||||
m_t = o_t < T
|
||||
if STATE_V_FIRST:
|
||||
p_dh1 = dh + i_t_int64*HV*K*V + o_v[:, None] * K + o_k1[None, :]
|
||||
m_dh1 = m_v[:, None] & m_k1[None, :]
|
||||
else:
|
||||
p_dh1 = dh + i_t_int64*HV*K*V + o_k1[:, None] * V + o_v[None, :]
|
||||
m_dh1 = m_k1[:, None] & m_v[None, :]
|
||||
tl.store(p_dh1, b_dh1.to(p_dh1.dtype.element_ty), mask=m_dh1)
|
||||
if K > 64:
|
||||
if STATE_V_FIRST:
|
||||
p_dh2 = dh + i_t_int64*HV*K*V + o_v[:, None] * K + o_k2[None, :]
|
||||
m_dh2 = m_v[:, None] & m_k2[None, :]
|
||||
else:
|
||||
p_dh2 = dh + i_t_int64*HV*K*V + o_k2[:, None] * V + o_v[None, :]
|
||||
m_dh2 = m_k2[:, None] & m_v[None, :]
|
||||
tl.store(p_dh2, b_dh2.to(p_dh2.dtype.element_ty), mask=m_dh2)
|
||||
if K > 128:
|
||||
if STATE_V_FIRST:
|
||||
p_dh3 = dh + i_t_int64*HV*K*V + o_v[:, None] * K + o_k3[None, :]
|
||||
m_dh3 = m_v[:, None] & m_k3[None, :]
|
||||
else:
|
||||
p_dh3 = dh + i_t_int64*HV*K*V + o_k3[:, None] * V + o_v[None, :]
|
||||
m_dh3 = m_k3[:, None] & m_v[None, :]
|
||||
tl.store(p_dh3, b_dh3.to(p_dh3.dtype.element_ty), mask=m_dh3)
|
||||
if K > 192:
|
||||
if STATE_V_FIRST:
|
||||
p_dh4 = dh + i_t_int64*HV*K*V + o_v[:, None] * K + o_k4[None, :]
|
||||
m_dh4 = m_v[:, None] & m_k4[None, :]
|
||||
else:
|
||||
p_dh4 = dh + i_t_int64*HV*K*V + o_k4[:, None] * V + o_v[None, :]
|
||||
m_dh4 = m_k4[:, None] & m_v[None, :]
|
||||
tl.store(p_dh4, b_dh4.to(p_dh4.dtype.element_ty), mask=m_dh4)
|
||||
|
||||
last_idx = min((i_t + 1) * BT, T) - 1
|
||||
if USE_G:
|
||||
bg_last = tl.load(g + (bos + last_idx) * HV + i_h).to(tl.float32)
|
||||
p_g = g + bos * HV + i_h + o_t * HV
|
||||
b_g = tl.load(p_g, mask=m_t, other=0.0).to(tl.float32)
|
||||
bg_last_exp = exp2(bg_last)
|
||||
b_g_exp = exp2(b_g)
|
||||
p_dv = dv + o_t[:, None] * (HV*V) + o_v[None, :]
|
||||
p_dv2 = dv2 + o_t[:, None] * (HV*V) + o_v[None, :]
|
||||
p_do = do + o_t[:, None] * (HV*V) + o_v[None, :]
|
||||
|
||||
b_do = tl.load(p_do, mask=m_t[:, None] & m_v[None, :], other=0.0)
|
||||
|
||||
# Update dv
|
||||
p_k = k + o_t[:, None] * (H*K) + o_k1[None, :]
|
||||
b_k = tl.load(p_k, mask=m_t[:, None] & m_k1[None, :], other=0.0)
|
||||
if USE_GK:
|
||||
o_k1 = tl.arange(0, 64)
|
||||
b_gk_last1 = tl.load(gk + last_idx * HV*K + o_k1, mask=(o_k1 < K), other=0.).to(tl.float32)
|
||||
if STATE_V_FIRST:
|
||||
b_dv = tl.dot(b_k, tl.trans(b_dh1).to(b_k.dtype))
|
||||
else:
|
||||
b_dv = tl.dot(b_k, b_dh1.to(b_k.dtype))
|
||||
|
||||
if K > 64:
|
||||
p_k = k + o_t[:, None] * (H*K) + o_k2[None, :]
|
||||
b_k = tl.load(p_k, mask=m_t[:, None] & m_k2[None, :], other=0.0)
|
||||
if USE_GK:
|
||||
b_gk_last2 = tl.load(gk + last_idx * HV*K + o_k2, mask=(o_k2 < K), other=0.).to(tl.float32)
|
||||
if STATE_V_FIRST:
|
||||
b_dv += tl.dot(b_k, tl.trans(b_dh2).to(b_k.dtype))
|
||||
else:
|
||||
b_dv += tl.dot(b_k, b_dh2.to(b_k.dtype))
|
||||
|
||||
if K > 128:
|
||||
p_k = k + o_t[:, None] * (H*K) + o_k3[None, :]
|
||||
b_k = tl.load(p_k, mask=m_t[:, None] & m_k3[None, :], other=0.0)
|
||||
if USE_GK:
|
||||
b_gk_last3 = tl.load(gk + last_idx * HV*K + o_k3, mask=(o_k3 < K), other=0.).to(tl.float32)
|
||||
if STATE_V_FIRST:
|
||||
b_dv += tl.dot(b_k, tl.trans(b_dh3).to(b_k.dtype))
|
||||
else:
|
||||
b_dv += tl.dot(b_k, b_dh3.to(b_k.dtype))
|
||||
|
||||
if K > 192:
|
||||
p_k = k + o_t[:, None] * (H*K) + o_k4[None, :]
|
||||
b_k = tl.load(p_k, mask=m_t[:, None] & m_k4[None, :], other=0.0)
|
||||
if USE_GK:
|
||||
b_gk_last4 = tl.load(gk + last_idx * HV*K + o_k4, mask=(o_k4 < K), other=0.).to(tl.float32)
|
||||
if STATE_V_FIRST:
|
||||
b_dv += tl.dot(b_k, tl.trans(b_dh4).to(b_k.dtype))
|
||||
else:
|
||||
b_dv += tl.dot(b_k, b_dh4.to(b_k.dtype))
|
||||
|
||||
if USE_G:
|
||||
b_dv *= tl.where(m_t, exp2(bg_last - b_g), 0)[:, None]
|
||||
b_dv += tl.load(p_dv, mask=m_t[:, None] & m_v[None, :], other=0.0)
|
||||
|
||||
tl.store(p_dv2, b_dv.to(p_dv.dtype.element_ty), mask=m_t[:, None] & m_v[None, :])
|
||||
# Update dh
|
||||
p_w = w + o_k1[:, None] + o_t[None, :] * (HV*K)
|
||||
p_q = q + o_k1[:, None] + o_t[None, :] * (H*K)
|
||||
b_w = tl.load(p_w, mask=m_k1[:, None] & m_t[None, :], other=0.0)
|
||||
b_q = tl.load(p_q, mask=m_k1[:, None] & m_t[None, :], other=0.0)
|
||||
if USE_G:
|
||||
b_dh1 *= bg_last_exp
|
||||
b_q = b_q * b_g_exp[None, :]
|
||||
if USE_GK:
|
||||
if STATE_V_FIRST:
|
||||
b_dh1 *= exp2(b_gk_last1)[None, :]
|
||||
else:
|
||||
b_dh1 *= exp2(b_gk_last1[:, None])
|
||||
if STATE_V_FIRST:
|
||||
b_dh1 += tl.trans(tl.dot(b_q.to(b_q.dtype), b_do.to(b_q.dtype)) * scale - tl.dot(b_w, b_dv.to(b_w.dtype)))
|
||||
else:
|
||||
b_dh1 += tl.dot(b_q.to(b_q.dtype), b_do.to(b_q.dtype)) * scale - tl.dot(b_w, b_dv.to(b_w.dtype))
|
||||
if K > 64:
|
||||
p_q = q + o_k2[:, None] + o_t[None, :] * (H*K)
|
||||
p_w = w + o_k2[:, None] + o_t[None, :] * (HV*K)
|
||||
b_q = tl.load(p_q, mask=m_k2[:, None] & m_t[None, :], other=0.0)
|
||||
b_w = tl.load(p_w, mask=m_k2[:, None] & m_t[None, :], other=0.0)
|
||||
if USE_G:
|
||||
b_dh2 *= bg_last_exp
|
||||
b_q = b_q * b_g_exp[None, :]
|
||||
if USE_GK:
|
||||
if STATE_V_FIRST:
|
||||
b_dh2 *= exp2(b_gk_last2)[None, :]
|
||||
else:
|
||||
b_dh2 *= exp2(b_gk_last2[:, None])
|
||||
if STATE_V_FIRST:
|
||||
b_dh2 += tl.trans(tl.dot(b_q.to(b_q.dtype), b_do.to(b_q.dtype)) * scale - tl.dot(b_w, b_dv.to(b_w.dtype)))
|
||||
else:
|
||||
b_dh2 += tl.dot(b_q.to(b_q.dtype), b_do.to(b_q.dtype)) * scale - tl.dot(b_w, b_dv.to(b_w.dtype))
|
||||
if K > 128:
|
||||
p_q = q + o_k3[:, None] + o_t[None, :] * (H*K)
|
||||
p_w = w + o_k3[:, None] + o_t[None, :] * (HV*K)
|
||||
b_q = tl.load(p_q, mask=m_k3[:, None] & m_t[None, :], other=0.0)
|
||||
b_w = tl.load(p_w, mask=m_k3[:, None] & m_t[None, :], other=0.0)
|
||||
if USE_G:
|
||||
b_dh3 *= bg_last_exp
|
||||
b_q = b_q * b_g_exp[None, :]
|
||||
if USE_GK:
|
||||
if STATE_V_FIRST:
|
||||
b_dh3 *= exp2(b_gk_last3)[None, :]
|
||||
else:
|
||||
b_dh3 *= exp2(b_gk_last3[:, None])
|
||||
if STATE_V_FIRST:
|
||||
b_dh3 += tl.trans(tl.dot(b_q.to(b_q.dtype), b_do.to(b_q.dtype)) * scale - tl.dot(b_w, b_dv.to(b_w.dtype)))
|
||||
else:
|
||||
b_dh3 += tl.dot(b_q.to(b_q.dtype), b_do.to(b_q.dtype)) * scale - tl.dot(b_w, b_dv.to(b_w.dtype))
|
||||
if K > 192:
|
||||
p_q = q + o_k4[:, None] + o_t[None, :] * (H*K)
|
||||
p_w = w + o_k4[:, None] + o_t[None, :] * (HV*K)
|
||||
b_q = tl.load(p_q, mask=m_k4[:, None] & m_t[None, :], other=0.0)
|
||||
b_w = tl.load(p_w, mask=m_k4[:, None] & m_t[None, :], other=0.0)
|
||||
if USE_G:
|
||||
b_dh4 *= bg_last_exp
|
||||
b_q = b_q * b_g_exp[None, :]
|
||||
if USE_GK:
|
||||
if STATE_V_FIRST:
|
||||
b_dh4 *= exp2(b_gk_last4)[None, :]
|
||||
else:
|
||||
b_dh4 *= exp2(b_gk_last4[:, None])
|
||||
if STATE_V_FIRST:
|
||||
b_dh4 += tl.trans(tl.dot(b_q.to(b_q.dtype), b_do.to(b_q.dtype)) * scale - tl.dot(b_w, b_dv.to(b_w.dtype)))
|
||||
else:
|
||||
b_dh4 += tl.dot(b_q.to(b_q.dtype), b_do.to(b_q.dtype)) * scale - tl.dot(b_w, b_dv.to(b_w.dtype))
|
||||
|
||||
if USE_INITIAL_STATE:
|
||||
if STATE_V_FIRST:
|
||||
p_dh0 = dh0 + o_v[:, None] * K + o_k1[None, :]
|
||||
m_dh0 = m_v[:, None] & m_k1[None, :]
|
||||
else:
|
||||
p_dh0 = dh0 + o_k1[:, None] * V + o_v[None, :]
|
||||
m_dh0 = m_k1[:, None] & m_v[None, :]
|
||||
tl.store(p_dh0, b_dh1.to(p_dh0.dtype.element_ty), mask=m_dh0)
|
||||
if K > 64:
|
||||
if STATE_V_FIRST:
|
||||
p_dh1 = dh0 + o_v[:, None] * K + o_k2[None, :]
|
||||
m_dh1 = m_v[:, None] & m_k2[None, :]
|
||||
else:
|
||||
p_dh1 = dh0 + o_k2[:, None] * V + o_v[None, :]
|
||||
m_dh1 = m_k2[:, None] & m_v[None, :]
|
||||
tl.store(p_dh1, b_dh2.to(p_dh1.dtype.element_ty), mask=m_dh1)
|
||||
if K > 128:
|
||||
if STATE_V_FIRST:
|
||||
p_dh2 = dh0 + o_v[:, None] * K + o_k3[None, :]
|
||||
m_dh2 = m_v[:, None] & m_k3[None, :]
|
||||
else:
|
||||
p_dh2 = dh0 + o_k3[:, None] * V + o_v[None, :]
|
||||
m_dh2 = m_k3[:, None] & m_v[None, :]
|
||||
tl.store(p_dh2, b_dh3.to(p_dh2.dtype.element_ty), mask=m_dh2)
|
||||
if K > 192:
|
||||
if STATE_V_FIRST:
|
||||
p_dh3 = dh0 + o_v[:, None] * K + o_k4[None, :]
|
||||
m_dh3 = m_v[:, None] & m_k4[None, :]
|
||||
else:
|
||||
p_dh3 = dh0 + o_k4[:, None] * V + o_v[None, :]
|
||||
m_dh3 = m_k4[:, None] & m_v[None, :]
|
||||
tl.store(p_dh3, b_dh4.to(p_dh3.dtype.element_ty), mask=m_dh3)
|
||||
|
||||
|
||||
@dispatch('common')
|
||||
def chunk_gated_delta_rule_fwd_h(
|
||||
k: torch.Tensor,
|
||||
w: torch.Tensor,
|
||||
u: torch.Tensor,
|
||||
g: torch.Tensor | None = None,
|
||||
gk: torch.Tensor | None = None,
|
||||
initial_state: torch.Tensor | None = None,
|
||||
output_final_state: bool = False,
|
||||
chunk_size: int = 64,
|
||||
save_new_value: bool = True,
|
||||
state_v_first: bool = False,
|
||||
cu_seqlens: torch.LongTensor | None = None,
|
||||
cu_seqlens_cpu: torch.LongTensor | None = None,
|
||||
chunk_indices: torch.LongTensor | None = None,
|
||||
) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor | None]:
|
||||
B, T, H, K, V, HV = *k.shape, u.shape[-1], u.shape[2]
|
||||
BT = chunk_size
|
||||
|
||||
if chunk_indices is None and cu_seqlens is not None:
|
||||
chunk_indices = prepare_chunk_indices(cu_seqlens, chunk_size)
|
||||
# N: the actual number of sequences in the batch with either equal or variable lengths
|
||||
if cu_seqlens is None:
|
||||
N, NT, chunk_offsets = B, triton.cdiv(T, BT), None
|
||||
else:
|
||||
N, NT, chunk_offsets = len(cu_seqlens) - 1, len(chunk_indices), prepare_chunk_offsets(cu_seqlens, BT)
|
||||
assert K <= 256, "current kernel does not support head dimension larger than 256."
|
||||
|
||||
if state_v_first:
|
||||
h = k.new_empty(B, NT, HV, V, K)
|
||||
final_state = k.new_zeros(N, HV, V, K, dtype=torch.float32) if output_final_state else None
|
||||
else:
|
||||
h = k.new_empty(B, NT, HV, K, V)
|
||||
final_state = k.new_zeros(N, HV, K, V, dtype=torch.float32) if output_final_state else None
|
||||
|
||||
v_new = torch.empty_like(u) if save_new_value else None
|
||||
def grid(meta): return (triton.cdiv(V, meta['BV']) * N * HV, )
|
||||
chunk_gated_delta_rule_fwd_kernel_h_blockdim64[grid](
|
||||
k=k,
|
||||
v=u,
|
||||
w=w,
|
||||
v_new=v_new,
|
||||
g=g,
|
||||
gk=gk,
|
||||
h=h,
|
||||
h0=initial_state,
|
||||
ht=final_state,
|
||||
cu_seqlens=cu_seqlens,
|
||||
chunk_offsets=chunk_offsets,
|
||||
T=T,
|
||||
H=H,
|
||||
HV=HV,
|
||||
K=K,
|
||||
V=V,
|
||||
BT=BT,
|
||||
STATE_V_FIRST=state_v_first,
|
||||
)
|
||||
return h, v_new, final_state
|
||||
|
||||
|
||||
@dispatch('common')
|
||||
def chunk_gated_delta_rule_bwd_dhu(
|
||||
q: torch.Tensor,
|
||||
k: torch.Tensor,
|
||||
w: torch.Tensor,
|
||||
do: torch.Tensor,
|
||||
dv: torch.Tensor,
|
||||
g: torch.Tensor | None = None,
|
||||
gk: torch.Tensor | None = None,
|
||||
h0: torch.Tensor | None = None,
|
||||
dht: torch.Tensor | None = None,
|
||||
scale: float | None = None,
|
||||
state_v_first: bool = False,
|
||||
cu_seqlens: torch.LongTensor | None = None,
|
||||
chunk_size: int = 64,
|
||||
chunk_indices: torch.LongTensor | None = None,
|
||||
) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
|
||||
B, T, H, K, V, HV = *q.shape, do.shape[-1], do.shape[2]
|
||||
# N: the actual number of sequences in the batch with either equal or variable lengths
|
||||
BT = chunk_size
|
||||
assert K <= 256, "current kernel does not support head dimension being larger than 256."
|
||||
|
||||
if chunk_indices is None and cu_seqlens is not None:
|
||||
chunk_indices = prepare_chunk_indices(cu_seqlens, chunk_size)
|
||||
if cu_seqlens is None:
|
||||
N, NT, chunk_offsets = B, triton.cdiv(T, BT), None
|
||||
else:
|
||||
N, NT, chunk_offsets = len(cu_seqlens) - 1, len(chunk_indices), prepare_chunk_offsets(cu_seqlens, BT)
|
||||
|
||||
if state_v_first:
|
||||
dh = q.new_empty(B, NT, HV, V, K)
|
||||
else:
|
||||
dh = q.new_empty(B, NT, HV, K, V)
|
||||
dh0 = torch.empty_like(h0, dtype=torch.float32) if h0 is not None else None
|
||||
dv2 = torch.empty_like(dv)
|
||||
|
||||
def grid(meta): return (triton.cdiv(V, meta['BV']) * N * HV, )
|
||||
chunk_gated_delta_rule_bwd_kernel_dhu_blockdim64[grid](
|
||||
q=q,
|
||||
k=k,
|
||||
w=w,
|
||||
g=g,
|
||||
gk=gk,
|
||||
dht=dht,
|
||||
dh0=dh0,
|
||||
do=do,
|
||||
dh=dh,
|
||||
dv=dv,
|
||||
dv2=dv2,
|
||||
cu_seqlens=cu_seqlens,
|
||||
chunk_offsets=chunk_offsets,
|
||||
scale=scale,
|
||||
T=T,
|
||||
H=H,
|
||||
HV=HV,
|
||||
K=K,
|
||||
V=V,
|
||||
BT=BT,
|
||||
STATE_V_FIRST=state_v_first,
|
||||
)
|
||||
return dh, dh0, dv2
|
||||
@@ -0,0 +1,432 @@
|
||||
# Copyright (c) 2023-2026, Songlin Yang, Yu Zhang, Zhiyuan Li
|
||||
#
|
||||
# This source code is licensed under the MIT license found in the
|
||||
# LICENSE file in the root directory of this source tree.
|
||||
# For a list of all contributors, visit:
|
||||
# https://github.com/fla-org/flash-linear-attention/graphs/contributors
|
||||
|
||||
import torch
|
||||
import triton
|
||||
import triton.language as tl
|
||||
|
||||
from kda._fla.ops.backends import dispatch
|
||||
from kda._fla.ops.utils import prepare_chunk_offsets
|
||||
from kda._fla.ops.utils.op import exp2
|
||||
from kda._fla.utils import autotune_cache_kwargs, check_shared_mem
|
||||
|
||||
BKV_LIST = [32, 64] if check_shared_mem() else [16, 32]
|
||||
|
||||
|
||||
@triton.heuristics({
|
||||
'USE_INITIAL_STATE': lambda args: args['h0'] is not None,
|
||||
'STORE_FINAL_STATE': lambda args: args['ht'] is not None,
|
||||
'IS_VARLEN': lambda args: args['cu_seqlens'] is not None,
|
||||
})
|
||||
@triton.autotune(
|
||||
configs=[
|
||||
triton.Config({'BK': BK, 'BV': BV}, num_warps=num_warps, num_stages=num_stages)
|
||||
for BK in BKV_LIST
|
||||
for BV in BKV_LIST
|
||||
for num_warps in [1, 2, 4, 8]
|
||||
for num_stages in [2, 3, 4]
|
||||
],
|
||||
key=['BT', 'USE_G', 'USE_GK', 'USE_GV', 'STATE_V_FIRST'],
|
||||
**autotune_cache_kwargs,
|
||||
)
|
||||
@triton.jit(do_not_specialize=['T'])
|
||||
def chunk_fwd_kernel_h(
|
||||
k,
|
||||
v,
|
||||
h,
|
||||
g,
|
||||
g_gamma,
|
||||
gk,
|
||||
gv,
|
||||
h0,
|
||||
ht,
|
||||
cu_seqlens,
|
||||
split_offsets,
|
||||
T,
|
||||
H: tl.constexpr,
|
||||
K: tl.constexpr,
|
||||
V: tl.constexpr,
|
||||
BT: tl.constexpr,
|
||||
BS: tl.constexpr,
|
||||
BK: tl.constexpr,
|
||||
BV: tl.constexpr,
|
||||
USE_G: tl.constexpr,
|
||||
USE_G_GAMMA: tl.constexpr,
|
||||
USE_GK: tl.constexpr,
|
||||
USE_GV: tl.constexpr,
|
||||
USE_INITIAL_STATE: tl.constexpr,
|
||||
STORE_FINAL_STATE: tl.constexpr,
|
||||
IS_VARLEN: tl.constexpr,
|
||||
STATE_V_FIRST: tl.constexpr,
|
||||
):
|
||||
i_k, i_v, i_nh = tl.program_id(0), tl.program_id(1), tl.program_id(2).to(tl.int64)
|
||||
i_n, i_h = i_nh // H, i_nh % H
|
||||
if IS_VARLEN:
|
||||
bos, eos = tl.load(cu_seqlens + i_n).to(tl.int64), tl.load(cu_seqlens + i_n + 1).to(tl.int64)
|
||||
T = eos - bos
|
||||
NT, NS = tl.cdiv(T, BT), tl.cdiv(T, BS)
|
||||
boh = tl.load(split_offsets + i_n).to(tl.int64)
|
||||
else:
|
||||
bos, eos = i_n * T, i_n * T + T
|
||||
NT, NS = tl.cdiv(T, BT), tl.cdiv(T, BS)
|
||||
boh = i_n * NS
|
||||
NTS = BS // BT
|
||||
|
||||
if USE_G_GAMMA:
|
||||
# decay rate given the head index
|
||||
b_gamma = tl.load(g_gamma + i_h)
|
||||
b_g = b_gamma * (tl.arange(0, BT) + 1)
|
||||
|
||||
# [BK, BV] accumulator; STATE_V_FIRST only flips the stored state's HBM layout to [V, K], applied at the load/store below.
|
||||
b_h = tl.zeros([BK, BV], dtype=tl.float32)
|
||||
o_k = i_k * BK + tl.arange(0, BK)
|
||||
o_v = i_v * BV + tl.arange(0, BV)
|
||||
if USE_INITIAL_STATE:
|
||||
if STATE_V_FIRST:
|
||||
p_h0 = h0 + i_nh * K*V + o_v[:, None] * K + o_k[None, :]
|
||||
b_h = tl.trans(tl.load(p_h0, mask=(o_v[:, None] < V) & (o_k[None, :] < K), other=0.0)).to(tl.float32)
|
||||
else:
|
||||
p_h0 = h0 + i_nh * K*V + o_k[:, None] * V + o_v[None, :]
|
||||
b_h = tl.load(p_h0, mask=(o_k[:, None] < K) & (o_v[None, :] < V), other=0.0).to(tl.float32)
|
||||
|
||||
for i_t in range(NT):
|
||||
i_s = i_t // NTS
|
||||
o_t = i_t * BT + tl.arange(0, BT)
|
||||
m_t = o_t < T
|
||||
p_k = k + (bos*H + i_h) * K + o_k[:, None] + o_t[None, :] * (H*K)
|
||||
p_v = v + (bos*H + i_h) * V + o_t[:, None] * (H*V) + o_v[None, :]
|
||||
|
||||
o_h = ((boh + i_s) * H + i_h).to(tl.int64) * K*V
|
||||
if STATE_V_FIRST:
|
||||
p_h = h + o_h + o_v[:, None] * K + o_k[None, :]
|
||||
m_h = (o_v[:, None] < V) & (o_k[None, :] < K)
|
||||
else:
|
||||
p_h = h + o_h + o_k[:, None] * V + o_v[None, :]
|
||||
m_h = (o_k[:, None] < K) & (o_v[None, :] < V)
|
||||
|
||||
if i_t % NTS == 0:
|
||||
tl.store(p_h, (tl.trans(b_h) if STATE_V_FIRST else b_h).to(p_h.dtype.element_ty), mask=m_h)
|
||||
# [BK, BT]
|
||||
b_k = tl.load(p_k, mask=(o_k[:, None] < K) & m_t[None, :], other=0.0)
|
||||
# [BT, BV]
|
||||
b_v = tl.load(p_v, mask=m_t[:, None] & (o_v < V)[None, :], other=0.0)
|
||||
last_idx = min((i_t + 1) * BT, T) - 1
|
||||
|
||||
# scalar decay
|
||||
if USE_G:
|
||||
b_g_last = tl.load(g + bos * H + last_idx * H + i_h)
|
||||
p_g = g + bos*H + (i_t * BT + tl.arange(0, BT)) * H + i_h
|
||||
b_g = tl.load(p_g, mask=(i_t * BT + tl.arange(0, BT) < T), other=0.)
|
||||
b_h *= exp2(b_g_last)
|
||||
b_v = (b_v * exp2(b_g_last - b_g)[:, None]).to(b_v.dtype)
|
||||
|
||||
if USE_G_GAMMA:
|
||||
b_g_last = b_gamma * min(BT, T - i_t * BT)
|
||||
b_h *= exp2(b_g_last)
|
||||
b_v = (b_v * exp2(b_g_last - b_g)[:, None]).to(b_v.dtype)
|
||||
|
||||
# vector decay, h = Diag(gk) @ h
|
||||
if USE_GK:
|
||||
p_gk = gk + (bos*H + i_h) * K + o_k[:, None] + o_t[None, :] * (H*K)
|
||||
p_gk_last = gk + (bos + last_idx) * H*K + i_h * K + i_k * BK + tl.arange(0, BK)
|
||||
|
||||
b_gk_last = tl.load(p_gk_last, mask=(i_k * BK + tl.arange(0, BK) < K), other=0.)
|
||||
b_gk = tl.load(p_gk, mask=(o_k[:, None] < K) & m_t[None, :], other=0.0)
|
||||
b_h *= exp2(b_gk_last)[:, None]
|
||||
b_k = (b_k * exp2(b_gk_last[:, None] - b_gk)).to(b_k.dtype)
|
||||
|
||||
# vector decay, h = h @ Diag(gv)
|
||||
if USE_GV:
|
||||
p_gv = gv + (bos*H + i_h) * V + o_t[:, None] * (H*V) + o_v[None, :]
|
||||
p_gv_last = gv + (bos + last_idx) * H*V + i_h * V + i_v * BV + tl.arange(0, BV)
|
||||
|
||||
b_gv_last = tl.load(p_gv_last, mask=(i_v * BV + tl.arange(0, BV) < V), other=0.)
|
||||
b_gv = tl.load(p_gv, mask=m_t[:, None] & (o_v < V)[None, :], other=0.0)
|
||||
b_h *= exp2(b_gv_last)[None, :]
|
||||
b_v = (b_v * exp2(b_gv_last[None, :] - b_gv)).to(b_v.dtype)
|
||||
|
||||
b_h += tl.dot(b_k, b_v)
|
||||
|
||||
if STORE_FINAL_STATE:
|
||||
if STATE_V_FIRST:
|
||||
p_ht = ht + i_nh * K*V + o_v[:, None] * K + o_k[None, :]
|
||||
tl.store(p_ht, tl.trans(b_h).to(p_ht.dtype.element_ty), mask=(o_v[:, None] < V) & (o_k[None, :] < K))
|
||||
else:
|
||||
p_ht = ht + i_nh * K*V + o_k[:, None] * V + o_v[None, :]
|
||||
tl.store(p_ht, b_h.to(p_ht.dtype.element_ty), mask=(o_k[:, None] < K) & (o_v[None, :] < V))
|
||||
|
||||
|
||||
@triton.heuristics({
|
||||
'STORE_INITIAL_STATE_GRADIENT': lambda args: args['dh0'] is not None,
|
||||
'USE_FINAL_STATE_GRADIENT': lambda args: args['dht'] is not None,
|
||||
'IS_VARLEN': lambda args: args['cu_seqlens'] is not None,
|
||||
})
|
||||
@triton.autotune(
|
||||
configs=[
|
||||
triton.Config({'BK': BK, 'BV': BV}, num_warps=num_warps, num_stages=num_stages)
|
||||
for BK in BKV_LIST
|
||||
for BV in BKV_LIST
|
||||
for num_warps in [1, 2, 4, 8]
|
||||
for num_stages in [2, 3, 4]
|
||||
],
|
||||
key=['BT', 'USE_G', 'USE_GK', 'USE_GV', 'STATE_V_FIRST'],
|
||||
**autotune_cache_kwargs,
|
||||
)
|
||||
@triton.jit(do_not_specialize=['T'])
|
||||
def chunk_bwd_kernel_dh(
|
||||
q,
|
||||
g,
|
||||
g_gamma,
|
||||
gk,
|
||||
gv,
|
||||
do,
|
||||
dh,
|
||||
dht,
|
||||
dh0,
|
||||
cu_seqlens,
|
||||
split_offsets,
|
||||
scale,
|
||||
T,
|
||||
HQ: tl.constexpr,
|
||||
H: tl.constexpr,
|
||||
K: tl.constexpr,
|
||||
V: tl.constexpr,
|
||||
BT: tl.constexpr,
|
||||
BS: tl.constexpr,
|
||||
BK: tl.constexpr,
|
||||
BV: tl.constexpr,
|
||||
NG: tl.constexpr,
|
||||
USE_G: tl.constexpr,
|
||||
USE_G_GAMMA: tl.constexpr,
|
||||
USE_GK: tl.constexpr,
|
||||
USE_GV: tl.constexpr,
|
||||
STORE_INITIAL_STATE_GRADIENT: tl.constexpr,
|
||||
USE_FINAL_STATE_GRADIENT: tl.constexpr,
|
||||
IS_VARLEN: tl.constexpr,
|
||||
STATE_V_FIRST: tl.constexpr,
|
||||
):
|
||||
i_k, i_v, i_nh = tl.program_id(0), tl.program_id(1), tl.program_id(2).to(tl.int64)
|
||||
i_n, i_hq = i_nh // HQ, i_nh % HQ
|
||||
i_h = i_hq // NG
|
||||
if IS_VARLEN:
|
||||
bos, eos = tl.load(cu_seqlens + i_n).to(tl.int64), tl.load(cu_seqlens + i_n + 1).to(tl.int64)
|
||||
T = eos - bos
|
||||
NT = tl.cdiv(T, BT)
|
||||
NS = tl.cdiv(T, BS)
|
||||
boh = tl.load(split_offsets + i_n).to(tl.int64)
|
||||
else:
|
||||
bos, eos = i_n * T, i_n * T + T
|
||||
NT = tl.cdiv(T, BT)
|
||||
NS = tl.cdiv(T, BS)
|
||||
boh = i_n * NS
|
||||
|
||||
if USE_G_GAMMA:
|
||||
b_gamma = tl.load(g_gamma + i_h)
|
||||
b_g = b_gamma * (tl.arange(0, BT) + 1)
|
||||
|
||||
# [BK, BV] accumulator; STATE_V_FIRST only flips the stored state's HBM layout to [V, K], applied at the load/store below.
|
||||
b_dh = tl.zeros([BK, BV], dtype=tl.float32)
|
||||
o_k = i_k * BK + tl.arange(0, BK)
|
||||
o_v = i_v * BV + tl.arange(0, BV)
|
||||
if USE_FINAL_STATE_GRADIENT:
|
||||
if STATE_V_FIRST:
|
||||
p_dht = dht + i_nh * K*V + o_v[:, None] * K + o_k[None, :]
|
||||
b_dh += tl.trans(tl.load(p_dht, mask=(o_v[:, None] < V) & (o_k[None, :] < K), other=0.0)).to(tl.float32)
|
||||
else:
|
||||
p_dht = dht + i_nh * K*V + o_k[:, None] * V + o_v[None, :]
|
||||
b_dh += tl.load(p_dht, mask=(o_k[:, None] < K) & (o_v[None, :] < V), other=0.0).to(tl.float32)
|
||||
|
||||
for i_t in range(NT - 1, -1, -1):
|
||||
i_s = i_t // (BS // BT)
|
||||
o_dh = ((boh + i_s) * H + i_h).to(tl.int64) * K*V
|
||||
if STATE_V_FIRST:
|
||||
p_dh = dh + o_dh + o_v[:, None] * K + o_k[None, :]
|
||||
m_dh = (o_v[:, None] < V) & (o_k[None, :] < K)
|
||||
else:
|
||||
p_dh = dh + o_dh + o_k[:, None] * V + o_v[None, :]
|
||||
m_dh = (o_k[:, None] < K) & (o_v[None, :] < V)
|
||||
|
||||
if i_t % (BS // BT) == 0:
|
||||
tl.store(p_dh, (tl.trans(b_dh) if STATE_V_FIRST else b_dh).to(p_dh.dtype.element_ty), mask=m_dh)
|
||||
last_idx = min(i_t * BT + BT, T) - 1
|
||||
o_t = i_t * BT + tl.arange(0, BT)
|
||||
m_t = o_t < T
|
||||
# [BK, BT]
|
||||
p_q = q + (bos*HQ + i_hq) * K + o_k[:, None] + o_t[None, :] * (HQ*K)
|
||||
p_do = do + (bos*HQ + i_hq) * V + o_t[:, None] * (HQ*V) + o_v[None, :]
|
||||
b_q = tl.load(p_q, mask=(o_k[:, None] < K) & m_t[None, :], other=0.0)
|
||||
b_q = (b_q * scale).to(b_q.dtype)
|
||||
# [BT, BV]
|
||||
b_do = tl.load(p_do, mask=m_t[:, None] & (o_v < V)[None, :], other=0.0)
|
||||
|
||||
if USE_G:
|
||||
p_g = g + (bos + i_t * BT + tl.arange(0, BT)) * H + i_h
|
||||
b_g_last = tl.load(g + (bos + last_idx) * H + i_h)
|
||||
b_g = tl.load(p_g, mask=(i_t * BT + tl.arange(0, BT) < T), other=0.)
|
||||
b_q = (b_q * exp2(b_g)[None, :]).to(b_q.dtype)
|
||||
b_dh *= exp2(b_g_last)
|
||||
|
||||
if USE_G_GAMMA:
|
||||
b_g_last = b_gamma * min(BT, T - i_t * BT)
|
||||
b_q = (b_q * exp2(b_g)[None, :]).to(b_q.dtype)
|
||||
b_dh *= exp2(b_g_last)
|
||||
|
||||
if USE_GK:
|
||||
p_gk = gk + (bos*H + i_h) * K + o_k[:, None] + o_t[None, :] * (H*K)
|
||||
p_gk_last = gk + (bos + last_idx) * H*K + i_h * K + i_k * BK + tl.arange(0, BK)
|
||||
|
||||
b_gk = tl.load(p_gk, mask=(o_k[:, None] < K) & m_t[None, :], other=0.0)
|
||||
b_gk_last = tl.load(p_gk_last, mask=(i_k * BK + tl.arange(0, BK) < K), other=0.)
|
||||
b_q = (b_q * exp2(b_gk)).to(b_q.dtype)
|
||||
b_dh *= exp2(b_gk_last)[:, None]
|
||||
|
||||
if USE_GV:
|
||||
p_gv = gv + (bos*H + i_h) * V + o_t[:, None] * (H*V) + o_v[None, :]
|
||||
p_gv_last = gv + (bos + last_idx) * H*V + i_h * V + i_v * BV + tl.arange(0, BV)
|
||||
|
||||
b_gv = tl.load(p_gv, mask=m_t[:, None] & (o_v < V)[None, :], other=0.0)
|
||||
b_gv_last = tl.load(p_gv_last, mask=(i_v * BV + tl.arange(0, BV) < V), other=0.)
|
||||
b_do = (b_do * exp2(b_gv))
|
||||
b_dh *= exp2(b_gv_last)[None, :]
|
||||
|
||||
b_dh += tl.dot(b_q, b_do.to(b_q.dtype))
|
||||
|
||||
if STORE_INITIAL_STATE_GRADIENT:
|
||||
if STATE_V_FIRST:
|
||||
p_dh0 = dh0 + i_nh * K*V + o_v[:, None] * K + o_k[None, :]
|
||||
tl.store(p_dh0, tl.trans(b_dh).to(p_dh0.dtype.element_ty), mask=(o_v[:, None] < V) & (o_k[None, :] < K))
|
||||
else:
|
||||
p_dh0 = dh0 + i_nh * K*V + o_k[:, None] * V + o_v[None, :]
|
||||
tl.store(p_dh0, b_dh.to(p_dh0.dtype.element_ty), mask=(o_k[:, None] < K) & (o_v[None, :] < V))
|
||||
|
||||
|
||||
@dispatch('common')
|
||||
def chunk_fwd_h(
|
||||
k: torch.Tensor,
|
||||
v: torch.Tensor,
|
||||
g: torch.Tensor | None = None,
|
||||
g_gamma: torch.Tensor | None = None,
|
||||
gk: torch.Tensor | None = None,
|
||||
gv: torch.Tensor | None = None,
|
||||
h0: torch.Tensor | None = None,
|
||||
output_final_state: bool = False,
|
||||
state_v_first: bool = False,
|
||||
cu_seqlens: torch.Tensor | None = None,
|
||||
chunk_size: int = 64,
|
||||
split_size: int | None = None,
|
||||
states_in_fp32: bool = False,
|
||||
) -> tuple[torch.Tensor, torch.Tensor]:
|
||||
B, T, H, K, V = *k.shape, v.shape[-1]
|
||||
BT = chunk_size
|
||||
BS = BT if split_size is None else split_size
|
||||
assert BS % BT == 0, f"The `split_size` (got {BS}) must be a multiple of `chunk_size` {BT}"
|
||||
# N: the actual number of sequences in the batch with either equal or variable lengths
|
||||
if cu_seqlens is None:
|
||||
N, NS, split_offsets = B, triton.cdiv(T, BS), None
|
||||
else:
|
||||
split_offsets = prepare_chunk_offsets(cu_seqlens, BS)
|
||||
N, NS = len(cu_seqlens) - 1, split_offsets[-1].item()
|
||||
|
||||
# `state_v_first` stores the states in V-first `[V, K]` layout instead of `[K, V]`
|
||||
state_shape = (V, K) if state_v_first else (K, V)
|
||||
h = k.new_empty(B, NS, H, *state_shape, dtype=k.dtype if not states_in_fp32 else torch.float)
|
||||
ht = k.new_empty(N, H, *state_shape, dtype=torch.float) if output_final_state else None
|
||||
def grid(meta): return (triton.cdiv(K, meta['BK']), triton.cdiv(V, meta['BV']), N * H)
|
||||
chunk_fwd_kernel_h[grid](
|
||||
k=k,
|
||||
v=v,
|
||||
h=h,
|
||||
g=g,
|
||||
g_gamma=g_gamma,
|
||||
gk=gk,
|
||||
gv=gv,
|
||||
h0=h0,
|
||||
ht=ht,
|
||||
cu_seqlens=cu_seqlens,
|
||||
split_offsets=split_offsets,
|
||||
T=T,
|
||||
H=H,
|
||||
K=K,
|
||||
V=V,
|
||||
BT=BT,
|
||||
BS=BS,
|
||||
USE_G=g is not None,
|
||||
USE_G_GAMMA=g_gamma is not None,
|
||||
USE_GK=gk is not None,
|
||||
USE_GV=gv is not None,
|
||||
STATE_V_FIRST=state_v_first,
|
||||
)
|
||||
return h, ht
|
||||
|
||||
|
||||
@dispatch('common')
|
||||
def chunk_bwd_dh(
|
||||
q: torch.Tensor,
|
||||
k: torch.Tensor,
|
||||
v: torch.Tensor,
|
||||
do: torch.Tensor,
|
||||
h0: torch.Tensor,
|
||||
dht: torch.Tensor,
|
||||
scale: float,
|
||||
g: torch.Tensor | None = None,
|
||||
g_gamma: torch.Tensor | None = None,
|
||||
gk: torch.Tensor | None = None,
|
||||
gv: torch.Tensor | None = None,
|
||||
state_v_first: bool = False,
|
||||
cu_seqlens: torch.Tensor | None = None,
|
||||
chunk_size: int = 64,
|
||||
split_size: int | None = None,
|
||||
states_in_fp32: bool = False,
|
||||
) -> tuple[torch.Tensor, torch.Tensor]:
|
||||
B, T, H, K, V = *k.shape, v.shape[-1]
|
||||
HQ = q.shape[2]
|
||||
BT = chunk_size
|
||||
BS = BT if split_size is None else split_size
|
||||
assert BS % BT == 0, f"The `split_size` (got {BS}) must be a multiple of `chunk_size` {BT}"
|
||||
# N: the actual number of sequences in the batch with either equal or variable lengths
|
||||
# NG: number of groups in GQA
|
||||
if cu_seqlens is None:
|
||||
N, NS, split_offsets = B, triton.cdiv(T, BS), None
|
||||
else:
|
||||
split_offsets = prepare_chunk_offsets(cu_seqlens, BS)
|
||||
N, NS = len(cu_seqlens) - 1, split_offsets[-1].item()
|
||||
NG = HQ // H
|
||||
|
||||
# `state_v_first` stores the states in V-first `[V, K]` layout instead of `[K, V]`
|
||||
state_shape = (V, K) if state_v_first else (K, V)
|
||||
dh = k.new_empty(B, NS, HQ, *state_shape, dtype=k.dtype if not states_in_fp32 else torch.float)
|
||||
dh0 = torch.empty_like(h0, dtype=torch.float) if h0 is not None else None
|
||||
|
||||
def grid(meta): return (triton.cdiv(K, meta['BK']), triton.cdiv(V, meta['BV']), N * H)
|
||||
chunk_bwd_kernel_dh[grid](
|
||||
q=q,
|
||||
g=g,
|
||||
g_gamma=g_gamma,
|
||||
gk=gk,
|
||||
gv=gv,
|
||||
do=do,
|
||||
dh=dh,
|
||||
dht=dht,
|
||||
dh0=dh0,
|
||||
cu_seqlens=cu_seqlens,
|
||||
split_offsets=split_offsets,
|
||||
scale=scale,
|
||||
T=T,
|
||||
HQ=HQ,
|
||||
H=H,
|
||||
K=K,
|
||||
V=V,
|
||||
BT=BT,
|
||||
BS=BS,
|
||||
NG=NG,
|
||||
USE_G=g is not None,
|
||||
USE_G_GAMMA=g_gamma is not None,
|
||||
USE_GK=gk is not None,
|
||||
USE_GV=gv is not None,
|
||||
STATE_V_FIRST=state_v_first,
|
||||
)
|
||||
return dh, dh0
|
||||
@@ -0,0 +1,110 @@
|
||||
# Copyright (c) 2023-2026, Songlin Yang, Yu Zhang, Zhiyuan Li
|
||||
#
|
||||
# This source code is licensed under the MIT license found in the
|
||||
# LICENSE file in the root directory of this source tree.
|
||||
# For a list of all contributors, visit:
|
||||
# https://github.com/fla-org/flash-linear-attention/graphs/contributors
|
||||
|
||||
# Shared gate helpers reused across delta-rule family ops (KDA, GDN, ...).
|
||||
|
||||
import torch
|
||||
import triton
|
||||
import triton.language as tl
|
||||
|
||||
from kda._fla.ops.backends import dispatch
|
||||
from kda._fla.utils import autocast_custom_bwd, autocast_custom_fwd, input_guard
|
||||
|
||||
|
||||
@triton.jit
|
||||
def fused_beta_sigmoid_fwd_kernel(
|
||||
x,
|
||||
y,
|
||||
scale,
|
||||
n_elements,
|
||||
BLOCK_SIZE: tl.constexpr,
|
||||
):
|
||||
pid = tl.program_id(0).to(tl.int64)
|
||||
offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE).to(tl.int64)
|
||||
mask = offs < n_elements
|
||||
b_x = tl.load(x + offs, mask=mask, other=0).to(tl.float32)
|
||||
b_y = scale * tl.sigmoid(b_x)
|
||||
tl.store(y + offs, b_y.to(y.dtype.element_ty), mask=mask)
|
||||
|
||||
|
||||
@triton.jit
|
||||
def fused_beta_sigmoid_bwd_kernel(
|
||||
x,
|
||||
dy,
|
||||
dx,
|
||||
scale,
|
||||
n_elements,
|
||||
BLOCK_SIZE: tl.constexpr,
|
||||
):
|
||||
pid = tl.program_id(0).to(tl.int64)
|
||||
offs = pid * BLOCK_SIZE + tl.arange(0, BLOCK_SIZE).to(tl.int64)
|
||||
mask = offs < n_elements
|
||||
b_x = tl.load(x + offs, mask=mask, other=0).to(tl.float32)
|
||||
b_dy = tl.load(dy + offs, mask=mask, other=0).to(tl.float32)
|
||||
b_y = tl.sigmoid(b_x)
|
||||
b_dx = b_dy * scale * b_y * (1.0 - b_y)
|
||||
tl.store(dx + offs, b_dx.to(dx.dtype.element_ty), mask=mask)
|
||||
|
||||
|
||||
_BETA_SIGMOID_BLOCK_SIZE = 2048
|
||||
_BETA_SIGMOID_NUM_WARPS = 8
|
||||
|
||||
|
||||
@dispatch('common')
|
||||
def fused_beta_sigmoid_fwd(x: torch.Tensor, scale: float = 1.0) -> torch.Tensor:
|
||||
y = torch.empty_like(x, dtype=torch.float32)
|
||||
n_elements = x.numel()
|
||||
grid = (triton.cdiv(n_elements, _BETA_SIGMOID_BLOCK_SIZE),)
|
||||
fused_beta_sigmoid_fwd_kernel[grid](
|
||||
x,
|
||||
y,
|
||||
scale,
|
||||
n_elements,
|
||||
BLOCK_SIZE=_BETA_SIGMOID_BLOCK_SIZE,
|
||||
num_warps=_BETA_SIGMOID_NUM_WARPS,
|
||||
)
|
||||
return y
|
||||
|
||||
|
||||
@dispatch('common')
|
||||
def fused_beta_sigmoid_bwd(x: torch.Tensor, dy: torch.Tensor, scale: float = 1.0) -> torch.Tensor:
|
||||
dx = torch.empty_like(x)
|
||||
n_elements = x.numel()
|
||||
grid = (triton.cdiv(n_elements, _BETA_SIGMOID_BLOCK_SIZE),)
|
||||
fused_beta_sigmoid_bwd_kernel[grid](
|
||||
x,
|
||||
dy,
|
||||
dx,
|
||||
scale,
|
||||
n_elements,
|
||||
BLOCK_SIZE=_BETA_SIGMOID_BLOCK_SIZE,
|
||||
num_warps=_BETA_SIGMOID_NUM_WARPS,
|
||||
)
|
||||
return dx
|
||||
|
||||
|
||||
class BetaSigmoidFunction(torch.autograd.Function):
|
||||
@staticmethod
|
||||
@input_guard
|
||||
@autocast_custom_fwd
|
||||
def forward(ctx, x: torch.Tensor, scale: float = 1.0) -> torch.Tensor:
|
||||
y = fused_beta_sigmoid_fwd(x, scale)
|
||||
ctx.save_for_backward(x)
|
||||
ctx.scale = scale
|
||||
return y
|
||||
|
||||
@staticmethod
|
||||
@input_guard
|
||||
@autocast_custom_bwd
|
||||
def backward(ctx, dy: torch.Tensor):
|
||||
(x,) = ctx.saved_tensors
|
||||
dx = fused_beta_sigmoid_bwd(x, dy, ctx.scale)
|
||||
return dx.type_as(x), None
|
||||
|
||||
|
||||
def fused_beta_sigmoid(x: torch.Tensor, scale: float = 1.0) -> torch.Tensor:
|
||||
return BetaSigmoidFunction.apply(x, scale)
|
||||
Reference in New Issue
Block a user