Standalone tree split from LLMRL/projects/kda. Includes Triton dt_bias backward fix, train_k3 --preset 0.5b, SFT, Docker runtime, and tests.
24 lines
580 B
Python
24 lines
580 B
Python
"""Composable mixing layers: attn and ffn both map [B,T,D] -> [B,T,D].
|
|
|
|
Depth mixing (AttnRes) is not a layer_specs kind. CausalLM reads
|
|
``config.attnres`` (off | full | block) and wraps DecoderBlock sublayers.
|
|
"""
|
|
|
|
from .block import DecoderBlock, build_attn, build_ffn
|
|
from .kda_attn import KDAAttention
|
|
from .latent_moe import LatentMoE
|
|
from .mla import GatedMLA
|
|
from .rmsnorm import RMSNorm
|
|
from .swiglu import SwiGLUMLP
|
|
|
|
__all__ = [
|
|
"DecoderBlock",
|
|
"GatedMLA",
|
|
"KDAAttention",
|
|
"LatentMoE",
|
|
"RMSNorm",
|
|
"SwiGLUMLP",
|
|
"build_attn",
|
|
"build_ffn",
|
|
]
|