% teach: % gap: none — this is a reference appendix % takeaway: 一表查所有符号 % jump: none % omit: none \section{符号表} \subsection{形状参数} \begin{center} \begin{tabular}{lll} \toprule 符号 & 含义 & 典型值 (toy) \\ \midrule $B$ & batch size & 2--4 \\ $T$ & 序列长度 & 128--2048 \\ $D$ & hidden\_size & 64 / 256 \\ $H$ & query/key 头数 & 4 / 8 \\ $H_V$ & value 头数(GVA) & $G \cdot H$ \\ $G$ & GVA 组数 & $H_V / H$ \\ $K$ & key/query 头维度 & 16 \\ $V$ & value 头维度($= K$) & 16 \\ $C$ & chunk\_size & 16 / 64 \\ $r$ & KV latent rank (MLA) & 32 \\ $d_q$ & MLA query head dim & 16 \\ $d_v$ & MLA value head dim & 16 \\ $\ell$ & MoE latent width ($= D/2$) & 128 \\ $n_r$ & routed 专家数 & 16 \\ $k$ & Top-$k$ & 2 \\ $n_s$ & shared 专家数 & 2 \\ $d_{\mathrm{ff}}$ & 专家中间维度 & 96 \\ $N$ & AttnRes 原子层数 ($= 2L$) & 8 \\ $S$ & AttnRes 块大小(原子层) & 2--24 \\ \bottomrule \end{tabular} \end{center} \subsection{KDA 变量} \begin{center} \begin{tabular}{llp{7cm}} \toprule 符号 & 形状 & 含义 \\ \midrule $q_t$ & \shape{B, HV, K} & query(已 GVA 展开 + scale) \\ $k_t$ & \shape{B, HV, K} & key(已 GVA 展开) \\ $v_t$ & \shape{B, HV, V} & value \\ $g_t$ & \shape{B, HV, K} & gate(log-space 衰减,逐维逐头) \\ $\beta_t$ & \shape{B, HV} & 写入强度 \\ $S_t$ & \shape{B, HV, K, V} & KV 状态矩阵 \\ $S_{\mathrm{dec}}$ & \shape{B, HV, K, V} & 衰减后的状态 \\ $p_t$ & \shape{B, HV, V} & 旧状态对 $k_t$ 的预测 \\ $r_t$ & \shape{B, HV, V} & delta rule 残差 = $v_t - p_t$ \\ $a_t$ & \shape{B, HV, K} & 写入向量 = $\beta_t \cdot k_t$ \\ $o_t$ & \shape{B, HV, V} & 读出 = $q_t \cdot S_t$ \\ \bottomrule \end{tabular} \end{center} \subsection{Gate 变量} \begin{center} \begin{tabular}{llp{6cm}} \toprule 符号 & 形状 & 含义 \\ \midrule $g_{\mathrm{raw}}$ & \shape{B, T, HV, K} & gate 投影原始输出 \\ $A_{\log}$ & \shape{HV} & head-wise 衰减参数 (log-space) \\ $\Delta_b$ & \shape{HV, K} & per-dim gate bias \\ $\mathrm{rate}$ & \shape{HV, 1} & $\exp(A_{\log})$ \\ $\mathrm{input}$ & \shape{B, T, HV, K} & $g_{\mathrm{raw}} + \Delta_b$ \\ $L$ & 标量 & lower\_bound ($-5.0$) \\ \bottomrule \end{tabular} \end{center} \subsection{MLA 变量} \begin{center} \begin{tabular}{llp{6cm}} \toprule 符号 & 形状 & 含义 \\ \midrule $c$ & \shape{B, T, r} & KV latent(推理时缓存这个) \\ $q$ & \shape{B, T, H, d_q} & query(低秩路径输出) \\ $W_{UK}$ & \shape{H, d_q, r} & key 解压矩阵(吸收进 $q$) \\ $W_{UV}$ & \shape{H, d_v, r} & value 解压矩阵 \\ $q_{\mathrm{abs}}$ & \shape{B, T, H, r} & 吸收后的 query \\ score & \shape{B, H, T, T} & $q_{\mathrm{abs}} \cdot c^T$ \\ attn & \shape{B, H, T, T} & causal softmax \\ $\tilde{o}_{\mathrm{lat}}$ & \shape{B, H, T, r} & latent 加权输出 \\ $\tilde{o}$ & \shape{B, H, T, d_v} & 解压后的输出 \\ gate & \shape{B, T, H \cdot d_v} & $\sigma(W_g x)$ \\ \bottomrule \end{tabular} \end{center} \subsection{LatentMoE 变量} \begin{center} \begin{tabular}{llp{6cm}} \toprule 符号 & 形状 & 含义 \\ \midrule $x$ & \shape{B, T, D} & 输入 \\ $z$ & \shape{B, T, \ell} & latent ($\ell = D/2$) \\ logits & \shape{B, T, n_r} & router logits \\ ids & \shape{B, T, k} & Top-$k$ 专家索引 \\ probs & \shape{B, T, k} & softmax 权重 \\ $u$ & \shape{B, T, \ell} & routed 加权输出 \\ $s$ & \shape{B, T, D} & shared 专家求和 \\ $y$ & \shape{B, T, D} & $s + W_\uparrow \mathrm{RMSNorm}(u)$ \\ \bottomrule \end{tabular} \end{center} \subsection{AttnRes 变量} \begin{center} \begin{tabular}{llp{6.4cm}} \toprule 符号 & 形状 & 含义 \\ \midrule $v_i$ & \shape{B, T, D} & 第 $i$ 个源($v_0 =$ embedding 输出) \\ $w_l$ & \shape{D} & 第 $l$ 层的 depth query(零初始化) \\ $\gamma_l$ & \shape{D} & DepthResidual 的 RMSNorm gain \\ $\tilde{w}_l$ & \shape{D} & 折叠后的 query $= w_l \odot \gamma_l$ \\ $s_{l,i}$ & \shape{n, B, T} & 深度打分 $= \tilde{w}_l^{\top}\mathrm{RMS}(v_i)$ \\ $\alpha_{l,i}$ & \shape{n, B, T} & 深度维 softmax 权重 \\ $h_l$ & \shape{B, T, D} & 第 $l$ 层的输入 $= \sum_i \alpha_{l,i} v_i$ \\ $b_j$ & \shape{B, T, D} & 第 $j$ 个块的输出(Block 版的源) \\ $p$ & \shape{B, T, D} & 块内 running partial \\ $m, n, d$ & \shape{B, T} / \shape{B,T,D} / \shape{B,T} & online softmax 三元组 \\ \bottomrule \end{tabular} \end{center} \subsection{Einsum 速查} \begin{center} \small \begin{tabular}{p{6cm}lp{3.5cm}} \toprule 操作 & einsum & 结果形状 \\ \midrule key 查状态 & \texttt{'bhk,bhkv->bhv'} & $p_t$ \shape{B,HV,V} \\ 外积写入 & \texttt{'bhk,bhv->bhkv'} & $a_t \otimes r_t$ \shape{B,HV,K,V} \\ 读出 & \texttt{'bhk,bhkv->bhv'} & $o_t$ \shape{B,HV,V} \\ MLA 吸收 $W_{UK}$ & \texttt{'bthd,hdj->bthj'} & $q_{\mathrm{abs}}$ \shape{B,T,H,r} \\ MLA 打分 & \texttt{'bthj,bsj->bhts'} & score \shape{B,H,T,T} \\ MLA latent 加权 & \texttt{'bhts,bsj->bhtj'} & $\tilde{o}_{\mathrm{lat}}$ \shape{B,H,T,r} \\ MLA 解压 & \texttt{'bhtj,hvj->bhtv'} & $\tilde{o}$ \shape{B,H,T,d_v} \\ AttnRes 深度打分 & \texttt{'d,nbtd->nbt'} & $s_{l,i}$ \shape{n,B,T} \\ AttnRes 深度加权和 & \texttt{'nbt,nbtd->btd'} & $h_l$ \shape{B,T,D} \\ AttnRes 批量打分(inter) & \texttt{'qd,nbtd->qnbt'} & logits \shape{S,n,B,T} \\ \bottomrule \end{tabular} \end{center} \subsection{总结与延伸} \subsubsection*{核心要点回顾} \begin{enumerate}[nosep] \item \textbf{KDA} = delta rule 状态更新 + gate 衰减,线性复杂度 \item \textbf{分块} = chunk 内下三角解 + chunk 间状态递推,等价于 naive recurrent \item \textbf{GVA} = $H_V = G \cdot H$,forward repeat\_interleave / backward view+sum \item \textbf{MLA} = 低秩 latent + 矩阵吸收,KV cache 从 $2Hd$ 降到 $r$ \item \textbf{LatentMoE} = shared 全宽 + routed 半宽 latent + SiTU-GLU 防溢出 \item \textbf{K3 Hybrid} = 3 KDA + 1 MLA,KDA 提供位置感知 \item \textbf{AttnRes} = 深度维 softmax 残差,Block 版把源数压到 $O(N/S)$, 两阶段 = inter 批量 + intra online-softmax 合并 \end{enumerate} \subsubsection*{未完成项} \begin{itemize}[nosep] \item L5 — 项目内自研 fused gate Triton kernel \item L6 — recurrent decode cache(推理加速) \item AttnRes 与 recurrent decode 的组合(增量解码时的深度源缓存) \item AttnRes 开 / 关的收敛质量对比实验(目前只验证了等价性与可训练性) \end{itemize}