\documentclass[a4paper,11pt]{article} \input{notes-macros} \begin{document} % ---------- 封面 ---------- \begin{titlepage} \centering \vspace{2cm} {\huge\bfseries \notetitle\par} \vspace{0.6cm} {\Large \notesubtitle\par} \vspace{0.8cm} {\large \notedate\par} \vspace{1.5cm} \begin{tcolorbox}[width=0.88\textwidth, colback=black!2!white, colframe=black!60, sharp corners] \textbf{项目}:\texttt{projects/kda/} — KDA 手写实现(naive recurrent → chunked → Triton)\\ \textbf{架构}:KDA + Gated MLA + Stable LatentMoE + AttnRes 深度残差 (K3-like)\\ \textbf{参考}:KDA arXiv:2510.26692; AttnRes arXiv:2603.15031; Kimi K3 architecture notes\\ \textbf{代码}:\texttt{kda/ops/}, \texttt{kda/layers/}, \texttt{kda/models/} \end{tcolorbox} \end{titlepage} \tableofcontents \newpage \input{sections/sec-01} % KDA 递归核心 \input{sections/sec-02} % Gate 激活 \input{sections/sec-03} % 分块并行计算 \input{sections/sec-04} % GVA 分组值注意力 \input{sections/sec-05} % KDAAttention 层 \input{sections/sec-06} % Gated MLA 矩阵吸收版 \input{sections/sec-07} % SiTU-GLU 与 Stable LatentMoE \input{sections/sec-08} % K3 混合架构 \input{sections/sec-09} % Attention Residual 深度残差 \input{sections/sec-10} % 反向传播推导 \input{sections/sec-11} % 符号表 \end{document}