Initial K3 snapshot: 0.5B KDA/MLA/MoE train path

Standalone tree split from LLMRL/projects/kda. Includes Triton dt_bias
backward fix, train_k3 --preset 0.5b, SFT, Docker runtime, and tests.
This commit is contained in:
dela
2026-08-25 14:43:17 +08:00
commit 584f7e9e73
140 changed files with 21592 additions and 0 deletions
+20
View File
@@ -0,0 +1,20 @@
"""SwiGLU FFN: x [B,T,D] -> y [B,T,D]."""
from __future__ import annotations
import torch.nn.functional as F
from torch import nn
class SwiGLUMLP(nn.Module):
def __init__(self, hidden_size: int, intermediate_size: int):
super().__init__()
self.w1 = nn.Linear(hidden_size, intermediate_size, bias=False)
self.w3 = nn.Linear(hidden_size, intermediate_size, bias=False)
self.w2 = nn.Linear(intermediate_size, hidden_size, bias=False)
@classmethod
def from_config(cls, config) -> SwiGLUMLP:
return cls(config.hidden_size, config.intermediate_size)
def forward(self, x):
return self.w2(F.silu(self.w1(x)) * self.w3(x))