Initial K3 snapshot: 0.5B KDA/MLA/MoE train path

Standalone tree split from LLMRL/projects/kda. Includes Triton dt_bias
backward fix, train_k3 --preset 0.5b, SFT, Docker runtime, and tests.
This commit is contained in:
dela
2026-08-25 14:43:17 +08:00
commit 584f7e9e73
140 changed files with 21592 additions and 0 deletions
+23
View File
@@ -0,0 +1,23 @@
"""Composable mixing layers: attn and ffn both map [B,T,D] -> [B,T,D].
Depth mixing (AttnRes) is not a layer_specs kind. CausalLM reads
``config.attnres`` (off | full | block) and wraps DecoderBlock sublayers.
"""
from .block import DecoderBlock, build_attn, build_ffn
from .kda_attn import KDAAttention
from .latent_moe import LatentMoE
from .mla import GatedMLA
from .rmsnorm import RMSNorm
from .swiglu import SwiGLUMLP
__all__ = [
"DecoderBlock",
"GatedMLA",
"KDAAttention",
"LatentMoE",
"RMSNorm",
"SwiGLUMLP",
"build_attn",
"build_ffn",
]