Files
K3/notes/sections/sec-11.tex
T
dela a2c4217dae Document LatentMoE sigmoid routing, sparse dispatch, and K3 block figures
Ledger C10–C12 match the permute-pad-bmm path and Switch aux/z-loss.
Section 8 adds overview and component TikZ; MoE capacity is C_moe so it
does not collide with KDA chunk size.
2026-08-26 14:43:58 +08:00

201 lines
7.7 KiB
TeX
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
% teach:
% gap: none — this is a reference appendix
% takeaway: 一表查所有符号
% jump: none
% omit: none
\section{符号表}
\subsection{形状参数}
\begin{center}
\begin{tabular}{lll}
\toprule
符号 & 含义 & 典型值 (toy) \\
\midrule
$B$ & batch size & 2--4 \\
$T$ & 序列长度 & 128--2048 \\
$D$ & hidden\_size & 64 / 256 \\
$H$ & query/key 头数 & 4 / 8 \\
$H_V$ & value 头数(GVA) & $G \cdot H$ \\
$G$ & GVA 组数 & $H_V / H$ \\
$K$ & key/query 头维度 & 16 \\
$V$ & value 头维度($= K$) & 16 \\
$C$ & chunk\_size & 16 / 64 \\
$r$ & KV latent rank (MLA) & 32 \\
$d_q$ & MLA query head dim & 16 \\
$d_v$ & MLA value head dim & 16 \\
$\ell$ & MoE latent width ($= D/2$) & 128 \\
$n_r$ & routed 专家数 & 16 \\
$k$ & Top-$k$ & 2 \\
$n_s$ & shared 专家数 & 2 \\
$d_{\mathrm{ff}}$ & 专家中间维度 & 96 \\
$C_{\mathrm{moe}}$ & MoE 专家容量(pad 宽度) & 动态 \\
$\alpha_{\mathrm{aux}}$ & Switch/GShard aux 系数 & $10^{-2}$ \\
$\alpha_z$ & router z-loss 系数 & $10^{-3}$ \\
$N$ & AttnRes 原子层数 ($= 2L$) & 8 \\
$S$ & AttnRes 块大小(原子层) & 2--24 \\
\bottomrule
\end{tabular}
\end{center}
\subsection{KDA 变量}
\begin{center}
\begin{tabular}{llp{7cm}}
\toprule
符号 & 形状 & 含义 \\
\midrule
$q_t$ & \shape{B, HV, K} & query(已 GVA 展开 + scale) \\
$k_t$ & \shape{B, HV, K} & key(已 GVA 展开) \\
$v_t$ & \shape{B, HV, V} & value \\
$g_t$ & \shape{B, HV, K} & gate(log-space 衰减,逐维逐头) \\
$\beta_t$ & \shape{B, HV} & 写入强度 \\
$S_t$ & \shape{B, HV, K, V} & KV 状态矩阵 \\
$S_{\mathrm{dec}}$ & \shape{B, HV, K, V} & 衰减后的状态 \\
$p_t$ & \shape{B, HV, V} & 旧状态对 $k_t$ 的预测 \\
$r_t$ & \shape{B, HV, V} & delta rule 残差 = $v_t - p_t$ \\
$a_t$ & \shape{B, HV, K} & 写入向量 = $\beta_t \cdot k_t$ \\
$o_t$ & \shape{B, HV, V} & 读出 = $q_t \cdot S_t$ \\
\bottomrule
\end{tabular}
\end{center}
\subsection{Gate 变量}
\begin{center}
\begin{tabular}{llp{6cm}}
\toprule
符号 & 形状 & 含义 \\
\midrule
$g_{\mathrm{raw}}$ & \shape{B, T, HV, K} & gate 投影原始输出 \\
$A_{\log}$ & \shape{HV} & head-wise 衰减参数 (log-space) \\
$\Delta_b$ & \shape{HV, K} & per-dim gate bias \\
$\mathrm{rate}$ & \shape{HV, 1} & $\exp(A_{\log})$ \\
$\mathrm{input}$ & \shape{B, T, HV, K} & $g_{\mathrm{raw}} + \Delta_b$ \\
$L$ & 标量 & lower\_bound ($-5.0$) \\
\bottomrule
\end{tabular}
\end{center}
\subsection{MLA 变量}
\begin{center}
\begin{tabular}{llp{6cm}}
\toprule
符号 & 形状 & 含义 \\
\midrule
$c$ & \shape{B, T, r} & KV latent(推理时缓存这个) \\
$q$ & \shape{B, T, H, d_q} & query(低秩路径输出) \\
$W_{UK}$ & \shape{H, d_q, r} & key 解压矩阵(吸收进 $q$) \\
$W_{UV}$ & \shape{H, d_v, r} & value 解压矩阵 \\
$q_{\mathrm{abs}}$ & \shape{B, T, H, r} & 吸收后的 query \\
score & \shape{B, H, T, T} & $q_{\mathrm{abs}} \cdot c^T$ \\
attn & \shape{B, H, T, T} & causal softmax \\
$\tilde{o}_{\mathrm{lat}}$ & \shape{B, H, T, r} & latent 加权输出 \\
$\tilde{o}$ & \shape{B, H, T, d_v} & 解压后的输出 \\
gate & \shape{B, T, H \cdot d_v} & $\sigma(W_g x)$ \\
\bottomrule
\end{tabular}
\end{center}
\subsection{LatentMoE 变量}
\begin{center}
\begin{tabular}{llp{6cm}}
\toprule
符号 & 形状 & 含义 \\
\midrule
$x$ & \shape{B, T, D} & 输入 \\
$z$ & \shape{B, T, \ell} & latent ($\ell = D/2$) \\
logits & \shape{B, T, n_r} & router logits $W_r x$ \\
$s$ & \shape{B, T, n_r} & sigmoid 分数 $\sigma(\mathrm{logits})$ \\
$b$ & \shape{n_r} & expert bias(非持久,只进 TopK) \\
ids & \shape{B, T, k} & Top-$k$ 专家索引 \\
$p_i$ & \shape{B, T, k} & sigmoid-L1 权重 $s_i/\sum_{j\in T}s_j$ \\
padded & \shape{n_r, C_{\mathrm{moe}}, \ell} & dispatch 后 pad 到容量 $C_{\mathrm{moe}}$ \\
$C_{\mathrm{moe}}$ & 标量 & 最大专家负载(pad 宽度) \\
$u$ & \shape{B, T, \ell} & routed 加权输出 \\
$s_{\mathrm{sh}}$ & \shape{B, T, D} & shared 专家求和 \\
$y$ & \shape{B, T, D} & $s_{\mathrm{sh}} + W_\uparrow \mathrm{RMSNorm}(u)$ \\
$f_e, P_e$ & 标量 & aux loss 负载占比 / 平均分数 \\
$\mathcal{L}_{\mathrm{aux}}, \mathcal{L}_z$ & 标量 & 负载均衡 / z-loss \\
$\alpha_{\mathrm{aux}}, \alpha_z$ & 标量 & 对应系数($10^{-2}$ / $10^{-3}$) \\
\bottomrule
\end{tabular}
\end{center}
\subsection{AttnRes 变量}
\begin{center}
\begin{tabular}{llp{6.4cm}}
\toprule
符号 & 形状 & 含义 \\
\midrule
$v_i$ & \shape{B, T, D} & 第 $i$ 个源($v_0 =$ embedding 输出) \\
$w_l$ & \shape{D} & 第 $l$ 层的 depth query(零初始化) \\
$\gamma_l$ & \shape{D} & DepthResidual 的 RMSNorm gain \\
$\tilde{w}_l$ & \shape{D} & 折叠后的 query $= w_l \odot \gamma_l$ \\
$s_{l,i}$ & \shape{n, B, T} & 深度打分 $= \tilde{w}_l^{\top}\mathrm{RMS}(v_i)$ \\
$\alpha_{l,i}$ & \shape{n, B, T} & 深度维 softmax 权重 \\
$h_l$ & \shape{B, T, D} & 第 $l$ 层的输入 $= \sum_i \alpha_{l,i} v_i$ \\
$b_j$ & \shape{B, T, D} & 第 $j$ 个块的输出(Block 版的源) \\
$p$ & \shape{B, T, D} & 块内 running partial \\
$m, n, d$ & \shape{B, T} / \shape{B,T,D} / \shape{B,T} & online softmax 三元组 \\
\bottomrule
\end{tabular}
\end{center}
\subsection{Einsum 速查}
\begin{center}
\small
\begin{tabular}{p{6cm}lp{3.5cm}}
\toprule
操作 & einsum & 结果形状 \\
\midrule
key 查状态 & \texttt{'bhk,bhkv->bhv'} & $p_t$ \shape{B,HV,V} \\
外积写入 & \texttt{'bhk,bhv->bhkv'} & $a_t \otimes r_t$ \shape{B,HV,K,V} \\
读出 & \texttt{'bhk,bhkv->bhv'} & $o_t$ \shape{B,HV,V} \\
MLA 吸收 $W_{UK}$ & \texttt{'bthd,hdj->bthj'} & $q_{\mathrm{abs}}$ \shape{B,T,H,r} \\
MLA 打分 & \texttt{'bthj,bsj->bhts'} & score \shape{B,H,T,T} \\
MLA latent 加权 & \texttt{'bhts,bsj->bhtj'} & $\tilde{o}_{\mathrm{lat}}$ \shape{B,H,T,r} \\
MLA 解压 & \texttt{'bhtj,hvj->bhtv'} & $\tilde{o}$ \shape{B,H,T,d_v} \\
AttnRes 深度打分 & \texttt{'d,nbtd->nbt'} & $s_{l,i}$ \shape{n,B,T} \\
AttnRes 深度加权和 & \texttt{'nbt,nbtd->btd'} & $h_l$ \shape{B,T,D} \\
AttnRes 批量打分(inter) & \texttt{'qd,nbtd->qnbt'} & logits \shape{S,n,B,T} \\
MoE dispatch pad & \texttt{index\_put} & padded \shape{R,C_{\mathrm{moe}},\ell} \\
MoE gate 投影(grouped) & \texttt{bmm(padded, w\_g.T)} & $wg$ \shape{R,C_{\mathrm{moe}},ff} \\
MoE up 投影(grouped) & \texttt{bmm(padded, w\_u.T)} & $wu$ \shape{R,C_{\mathrm{moe}},ff} \\
MoE 输出投影(grouped) & \texttt{bmm(g$\odot$h, w\_o.T)} & out \shape{R,C_{\mathrm{moe}},\ell} \\
MoE scatter-add & \texttt{index\_add(0, tok, ...)} & $u$ \shape{N,\ell} \\
\bottomrule
\end{tabular}
\end{center}
\subsection{总结与延伸}
\subsubsection*{核心要点回顾}
\begin{enumerate}[nosep]
\item \textbf{KDA} = delta rule 状态更新 + gate 衰减,线性复杂度
\item \textbf{分块} = chunk 内下三角解 + chunk 间状态递推,等价于 naive recurrent
\item \textbf{GVA} = $H_V = G \cdot H$,forward repeat\_interleave / backward view+sum
\item \textbf{MLA} = 低秩 latent + 矩阵吸收,KV cache 从 $2Hd$ 降到 $r$
\item \textbf{LatentMoE} = shared 全宽 + routed 半宽 latent + SiTU-GLU 防溢出;
K3 sigmoid-TopK 路由 + 稀疏 permute-dispatch(每 token 只算 $k$ 个专家)+
Switch/GShard aux \& z-loss 防塌缩
\item \textbf{K3 Hybrid} = 3 KDA + 1 MLA,KDA 提供位置感知
\item \textbf{AttnRes} = 深度维 softmax 残差,Block 版把源数压到 $O(N/S)$,
两阶段 = inter 批量 + intra online-softmax 合并
\end{enumerate}
\subsubsection*{未完成项}
\begin{itemize}[nosep]
\item L5 — 项目内自研 fused gate Triton kernel
\item L6 — recurrent decode cache(推理加速)
\item AttnRes 与 recurrent decode 的组合(增量解码时的深度源缓存)
\item AttnRes 开 / 关的收敛质量对比实验(目前只验证了等价性与可训练性)
\end{itemize}