% Golden example 1 -- tensor parallel FFN, column-then-row sharding + AllReduce. % Shows: partition geometry (shards tile the parent exactly), one hue per TP % rank held across every stage, a collective node as a real operation, and a % split along the contracted axis drawn as a column sub-flow. % Layout is entirely by cursor: no absolute coordinate appears below. % ../scripts/build.sh tp-ffn-allreduce.tex \documentclass[border=10pt]{standalone} \usepackage[cjk]{supertensor} % --- role ledger: one hue per TP rank, held from W through H to P ---------- \stsetrole{act}{stViolet} % activations that every rank sees \stsetrole{r1}{stTeal} % rank 1 \stsetrole{r2}{stOrange} % rank 2 % --- geometry ledger: one physical edge per symbolic axis ------------------ \stdim{bt}{6} % B*T rows \stdim{d}{4} % model width \stdim{dffl}{4} % d_ff / p (per-rank hidden width) \stdim{dff}{8} % d_ff = p * (d_ff/p); the height of the stacked W_2 column \begin{document} \begin{tikzpicture} % ============================================================ stage A row === \ststage{SA}{列切 $\mathbf W_1$:无通信} \strow{rowA}{bt} \stface[role=act, bracket=true]{X}{}{bt}{d} \stglyph{mA}{$\times$} % Two shards, tiled exactly: gap=0pt makes them adjacent by construction, % so neither stretching nor an eyeballed offset can creep in. \stface[role=r1]{W1a}{}{d}{dffl} \stface[role=r2, gap=0pt]{W1b}{}{d}{dffl} \stglyph{eA}{$=$} \stface[role=r1]{Ha}{}{bt}{dffl} \stface[role=r2, gap=0pt]{Hb}{}{bt}{dffl} \strowend \stcaption{X}{$\mathbf X$}{$BT\times d$} \stcaption{W1a}{$\mathbf W_1^{(1)}$}{$d\times d_{\mathrm{ff}}/p$} \stcaption{W1b}{$\mathbf W_1^{(2)}$}{$d\times d_{\mathrm{ff}}/p$} \stcaption{Ha}{$\mathbf H^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$} \stcaption{Hb}{$\mathbf H^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$} % ============================================================ stage B row === \ststage{SB}{行切 $\mathbf W_2$:一次 All-Reduce} \strow{rowB}{dff} % the stacked W_2 column is the tallest object here \stface[role=r1]{Ga}{}{bt}{dffl} \stface[role=r2, gap=0pt]{Gb}{}{bt}{dffl} \stglyph{mB}{$\times$} % W_2 is split along the CONTRACTED axis: the shards stack vertically and % together have exactly the height of H's width. Splitting reverses concat, % which is what gap=0pt inside the column states. \stcol{W2}{dff} \stface[role=r1]{W2a}{}{dffl}{d} \stface[role=r2, gap=0pt]{W2b}{}{dffl}{d} \stcolend \stglyph{eB}{$=$} \stface[role=r1]{Pa}{}{bt}{d} \stglyph{plus}{$+$} \stface[role=r2]{Pb}{}{bt}{d} \stlink{lc}{} \stcomm{ar}{All-Reduce} \stlink{ly}{} \stface[role=act, bracket=true]{Y}{}{bt}{d} \strowend \stcaption{Ga}{$\mathbf G^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$} \stcaption{Gb}{$\mathbf G^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$} \stcaption{W2}{$\mathbf W_2^{(r)}$}{$d_{\mathrm{ff}}/p\times d$} \stcaption{Pa}{$\mathbf P^{(1)}$}{$BT\times d$} \stcaption{Pb}{$\mathbf P^{(2)}$}{$BT\times d$} \stcaption{Y}{$\mathbf Y$}{$BT\times d$} % ================================================================= formula == % Placed last so it is centered on the figure that was actually drawn. \sttopformula{F}{$\begin{gathered} \mathbf{XW}_1=[\,\mathbf{XW}_1^{(1)}\mid \mathbf{XW}_1^{(2)}\,]=[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]\\[1.2mm] [\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,] \begin{bmatrix}\mathbf W_2^{(1)}\\[-1pt]\mathbf W_2^{(2)}\end{bmatrix} =\sum_{r}\mathbf H^{(r)}\mathbf W_2^{(r)}=\sum_r\mathbf P^{(r)} \end{gathered}$} % ============================================================== meaning box == \stbbox{all} \stmeaningbox{mb}{16.4cm}{all} {$BT$ 展平后的 token 数,$d$ 模型宽度,$d_{\mathrm{ff}}$ 前馈中间宽度, $p$ TP 并行度(图中 $p=2$)} {$\mathbf X,\mathbf Y$ 每个 rank 完整持有;$\mathbf W_1^{(r)},\mathbf W_2^{(r)}, \mathbf H^{(r)},\mathbf P^{(r)}$ 仅本 rank 持有,$\mathbf P^{(r)}$ 是部分和而非最终输出} {$\mathbf G^{(r)}=\mathrm{GeLU}(\mathbf H^{(r)})$ 逐元素、无跨 rank 依赖;列切 $\mathbf W_1$ 使 $\mathbf H$ 沿 $d_{\mathrm{ff}}$ 切分;行切 $\mathbf W_2$ 沿收缩维切分, 故 $\mathbf Y=\sum_r\mathbf P^{(r)}$ 需一次 All-Reduce,前向每层仅此一次通信} \stsignature{TP-FFN(GeLU + All-Reduce)}{mb} \end{tikzpicture} \end{document}