Initial K3 snapshot: 0.5B KDA/MLA/MoE train path
Standalone tree split from LLMRL/projects/kda. Includes Triton dt_bias backward fix, train_k3 --preset 0.5b, SFT, Docker runtime, and tests.
This commit is contained in:
@@ -0,0 +1,23 @@
|
||||
"""Composable mixing layers: attn and ffn both map [B,T,D] -> [B,T,D].
|
||||
|
||||
Depth mixing (AttnRes) is not a layer_specs kind. CausalLM reads
|
||||
``config.attnres`` (off | full | block) and wraps DecoderBlock sublayers.
|
||||
"""
|
||||
|
||||
from .block import DecoderBlock, build_attn, build_ffn
|
||||
from .kda_attn import KDAAttention
|
||||
from .latent_moe import LatentMoE
|
||||
from .mla import GatedMLA
|
||||
from .rmsnorm import RMSNorm
|
||||
from .swiglu import SwiGLUMLP
|
||||
|
||||
__all__ = [
|
||||
"DecoderBlock",
|
||||
"GatedMLA",
|
||||
"KDAAttention",
|
||||
"LatentMoE",
|
||||
"RMSNorm",
|
||||
"SwiGLUMLP",
|
||||
"build_attn",
|
||||
"build_ffn",
|
||||
]
|
||||
Reference in New Issue
Block a user