% Golden example 1 -- tensor parallel FFN, column-then-row sharding + AllReduce. % Shows: partition geometry (shards tile the parent exactly), one hue per TP % rank held across every stage, a collective node as a real operation. % ../scripts/build.sh tp-ffn-allreduce.tex \documentclass[border=10pt]{standalone} \usepackage[cjk]{supertensor} % --- role ledger: one hue per TP rank, held from W through H to P ---------- \stsetrole{act}{stViolet} % activations that every rank sees \stsetrole{r1}{stTeal} % rank 1 \stsetrole{r2}{stOrange} % rank 2 % --- geometry ledger: one physical edge per symbolic axis ------------------ \stdim{bt}{6} % B*T rows \stdim{d}{4} % model width \stdim{dffl}{4} % d_ff / p (per-rank hidden width) \begin{document} \begin{tikzpicture} % ================================================================= formula == \node (F) at (0,0) {\stformula{$\mathbf{XW}_1=[\,\mathbf{XW}_1^{(1)}\mid \mathbf{XW}_1^{(2)}\,]=[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]$}}; \node[below=1.2mm of F] (F2) {\stformula{$\displaystyle [\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,] \begin{bmatrix}\mathbf W_2^{(1)}\\[-1pt]\mathbf W_2^{(2)}\end{bmatrix} =\sum_{r}\mathbf H^{(r)}\mathbf W_2^{(r)}=\sum_r\mathbf P^{(r)}$}}; % ============================================================ stage A row === \node[st stage, below=7mm of F2] (SA) {列切 $\mathbf W_1$:无通信}; \coordinate (a) at ($(SA)+(-5.6,-1.8)$); \stface[role=act, bracket=true]{X}{(a)}{bt}{d} \node[st op, right=5mm of X] (mA) {$\times$}; % Two shards, tiled exactly: adjacent faces, no stretching, no gap. \stface[role=r1]{W1a}{($(mA)+(1.5,0)$)}{d}{dffl} \stface[role=r2]{W1b}{($(W1a.east)+(2*\stunit,0)$)}{d}{dffl} \node[st op, right=5mm of W1b] (eA) {$=$}; \stface[role=r1]{Ha}{($(eA)+(1.5,0)$)}{bt}{dffl} \stface[role=r2]{Hb}{($(Ha.east)+(2*\stunit,0)$)}{bt}{dffl} \node[inner sep=0pt, fit=(X)(W1a)(Ha)(Hb)] (rowA) {}; \stlane{rowA} \stcaption{X}{$\mathbf X$}{$BT\times d$} \stcaption{W1a}{$\mathbf W_1^{(1)}$}{$d\times d_{\mathrm{ff}}/p$} \stcaption{W1b}{$\mathbf W_1^{(2)}$}{$d\times d_{\mathrm{ff}}/p$} \stcaption{Ha}{$\mathbf H^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$} \stcaption{Hb}{$\mathbf H^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$} \stnolane % ============================================================ stage B row === \node[st stage, below=9mm of X-shape.south west, anchor=north west] (SB) {行切 $\mathbf W_2$:一次 All-Reduce}; \coordinate (b) at ($(SB)+(-0.4,-1.9)$); \stface[role=r1]{Ga}{(b)}{bt}{dffl} \stface[role=r2]{Gb}{($(Ga.east)+(2*\stunit,0)$)}{bt}{dffl} \node[st op, right=5mm of Gb] (mB) {$\times$}; % W_2 is split along the CONTRACTED axis: the two shards stack vertically and % together have exactly the height of H's width. Splitting reverses concat. \stface[role=r1]{W2a}{($(mB)+(1.35,0.46)$)}{dffl}{d} \stface[role=r2]{W2b}{($(W2a.south)+(0,-2*\stunit)$)}{dffl}{d} \node[st op, right=5mm of W2a.east |- W2a.south] (eB) {$=$}; \stface[role=r1]{Pa}{($(eB)+(1.3,0)$)}{bt}{d} \node[st op, right=4mm of Pa] (plus) {$+$}; \stface[role=r2]{Pb}{($(plus)+(1.3,0)$)}{bt}{d} \node[st comm, right=9mm of Pb] (ar) {All-Reduce}; \stface[role=act, bracket=true]{Y}{($(ar)+(1.9,0)$)}{bt}{d} \starrow{Pb.east}{ar.west} \starrow{ar.east}{Y.west} \node[inner sep=0pt, fit=(Ga)(W2a)(W2b)(Pa)(Pb)(Y)] (rowB) {}; \stlane{rowB} \stcaption{Ga}{$\mathbf G^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$} \stcaption{Gb}{$\mathbf G^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$} \stcaption{W2b}{$\mathbf W_2^{(r)}$}{$d_{\mathrm{ff}}/p\times d$} \stcaption{Pa}{$\mathbf P^{(1)}$}{$BT\times d$} \stcaption{Pb}{$\mathbf P^{(2)}$}{$BT\times d$} \stcaption{Y}{$\mathbf Y$}{$BT\times d$} \stnolane % ============================================================== meaning box == \node[inner sep=0pt, fit=(F)(rowA)(rowB)(Y-shape)(Ga-shape)] (all) {}; \stmeaningbox{mb}{16.4cm}{all} {$BT$ 展平后的 token 数,$d$ 模型宽度,$d_{\mathrm{ff}}$ 前馈中间宽度, $p$ TP 并行度(图中 $p=2$)} {$\mathbf X,\mathbf Y$ 每个 rank 完整持有;$\mathbf W_1^{(r)},\mathbf W_2^{(r)}, \mathbf H^{(r)},\mathbf P^{(r)}$ 仅本 rank 持有,$\mathbf P^{(r)}$ 是部分和而非最终输出} {$\mathbf G^{(r)}=\mathrm{GeLU}(\mathbf H^{(r)})$ 逐元素、无跨 rank 依赖;列切 $\mathbf W_1$ 使 $\mathbf H$ 沿 $d_{\mathrm{ff}}$ 切分;行切 $\mathbf W_2$ 沿收缩维切分, 故 $\mathbf Y=\sum_r\mathbf P^{(r)}$ 需一次 All-Reduce,前向每层仅此一次通信} \stsignature{TP-FFN(GeLU + All-Reduce)}{mb} \end{tikzpicture} \end{document}