Figures were positioned by hand-written offsets. Every gap was a magic number tuned against the content that happened to be there, so a label that grew two characters landed on the next tensor, and two stages started from two different x shared no rail. Both failures compile cleanly. Replace it with a cursor. Objects placed with an empty coordinate argument reserve their own width -- including a stack's offset sheets and a bracket's overhang -- and gaps are declared once (\stgutter, \strowgap, \stblockgap). The gap belongs to the object that follows it and the first object in a band gets none, so every band starts flush on a shared rail and gap=0pt states that two shards tile exactly. \stlink makes a connector's label a flow object, which is what removes the label-wider-than-its-arrow failure entirely. \stcol is a vertical sub-flow for a split along the contracted axis. \strow declares its height, so an object that does not fit -- or a column that does not add up to what it declared, and is therefore drawn off-center -- becomes a package warning, which build.sh fails on. Absolute placement is unchanged: passing a coordinate takes the original code path, and \sttrack folds a hand-placed node back into the cursor. All three golden examples and the new tests/flow.tex are converted and build clean.
95 lines
4.3 KiB
TeX
95 lines
4.3 KiB
TeX
% Golden example 1 -- tensor parallel FFN, column-then-row sharding + AllReduce.
|
||
% Shows: partition geometry (shards tile the parent exactly), one hue per TP
|
||
% rank held across every stage, a collective node as a real operation, and a
|
||
% split along the contracted axis drawn as a column sub-flow.
|
||
% Layout is entirely by cursor: no absolute coordinate appears below.
|
||
% ../scripts/build.sh tp-ffn-allreduce.tex
|
||
\documentclass[border=10pt]{standalone}
|
||
\usepackage[cjk]{supertensor}
|
||
|
||
% --- role ledger: one hue per TP rank, held from W through H to P ----------
|
||
\stsetrole{act}{stViolet} % activations that every rank sees
|
||
\stsetrole{r1}{stTeal} % rank 1
|
||
\stsetrole{r2}{stOrange} % rank 2
|
||
|
||
% --- geometry ledger: one physical edge per symbolic axis ------------------
|
||
\stdim{bt}{6} % B*T rows
|
||
\stdim{d}{4} % model width
|
||
\stdim{dffl}{4} % d_ff / p (per-rank hidden width)
|
||
\stdim{dff}{8} % d_ff = p * (d_ff/p); the height of the stacked W_2 column
|
||
|
||
\begin{document}
|
||
\begin{tikzpicture}
|
||
|
||
% ============================================================ stage A row ===
|
||
\ststage{SA}{列切 $\mathbf W_1$:无通信}
|
||
\strow{rowA}{bt}
|
||
\stface[role=act, bracket=true]{X}{}{bt}{d}
|
||
\stglyph{mA}{$\times$}
|
||
% Two shards, tiled exactly: gap=0pt makes them adjacent by construction,
|
||
% so neither stretching nor an eyeballed offset can creep in.
|
||
\stface[role=r1]{W1a}{}{d}{dffl}
|
||
\stface[role=r2, gap=0pt]{W1b}{}{d}{dffl}
|
||
\stglyph{eA}{$=$}
|
||
\stface[role=r1]{Ha}{}{bt}{dffl}
|
||
\stface[role=r2, gap=0pt]{Hb}{}{bt}{dffl}
|
||
\strowend
|
||
\stcaption{X}{$\mathbf X$}{$BT\times d$}
|
||
\stcaption{W1a}{$\mathbf W_1^{(1)}$}{$d\times d_{\mathrm{ff}}/p$}
|
||
\stcaption{W1b}{$\mathbf W_1^{(2)}$}{$d\times d_{\mathrm{ff}}/p$}
|
||
\stcaption{Ha}{$\mathbf H^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$}
|
||
\stcaption{Hb}{$\mathbf H^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$}
|
||
|
||
% ============================================================ stage B row ===
|
||
\ststage{SB}{行切 $\mathbf W_2$:一次 All-Reduce}
|
||
\strow{rowB}{dff} % the stacked W_2 column is the tallest object here
|
||
\stface[role=r1]{Ga}{}{bt}{dffl}
|
||
\stface[role=r2, gap=0pt]{Gb}{}{bt}{dffl}
|
||
\stglyph{mB}{$\times$}
|
||
% W_2 is split along the CONTRACTED axis: the shards stack vertically and
|
||
% together have exactly the height of H's width. Splitting reverses concat,
|
||
% which is what gap=0pt inside the column states.
|
||
\stcol{W2}{dff}
|
||
\stface[role=r1]{W2a}{}{dffl}{d}
|
||
\stface[role=r2, gap=0pt]{W2b}{}{dffl}{d}
|
||
\stcolend
|
||
\stglyph{eB}{$=$}
|
||
\stface[role=r1]{Pa}{}{bt}{d}
|
||
\stglyph{plus}{$+$}
|
||
\stface[role=r2]{Pb}{}{bt}{d}
|
||
\stlink{lc}{}
|
||
\stcomm{ar}{All-Reduce}
|
||
\stlink{ly}{}
|
||
\stface[role=act, bracket=true]{Y}{}{bt}{d}
|
||
\strowend
|
||
\stcaption{Ga}{$\mathbf G^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$}
|
||
\stcaption{Gb}{$\mathbf G^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$}
|
||
\stcaption{W2}{$\mathbf W_2^{(r)}$}{$d_{\mathrm{ff}}/p\times d$}
|
||
\stcaption{Pa}{$\mathbf P^{(1)}$}{$BT\times d$}
|
||
\stcaption{Pb}{$\mathbf P^{(2)}$}{$BT\times d$}
|
||
\stcaption{Y}{$\mathbf Y$}{$BT\times d$}
|
||
|
||
% ================================================================= formula ==
|
||
% Placed last so it is centered on the figure that was actually drawn.
|
||
\sttopformula{F}{$\begin{gathered}
|
||
\mathbf{XW}_1=[\,\mathbf{XW}_1^{(1)}\mid
|
||
\mathbf{XW}_1^{(2)}\,]=[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]\\[1.2mm]
|
||
[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]
|
||
\begin{bmatrix}\mathbf W_2^{(1)}\\[-1pt]\mathbf W_2^{(2)}\end{bmatrix}
|
||
=\sum_{r}\mathbf H^{(r)}\mathbf W_2^{(r)}=\sum_r\mathbf P^{(r)}
|
||
\end{gathered}$}
|
||
|
||
% ============================================================== meaning box ==
|
||
\stbbox{all}
|
||
\stmeaningbox{mb}{16.4cm}{all}
|
||
{$BT$ 展平后的 token 数,$d$ 模型宽度,$d_{\mathrm{ff}}$ 前馈中间宽度,
|
||
$p$ TP 并行度(图中 $p=2$)}
|
||
{$\mathbf X,\mathbf Y$ 每个 rank 完整持有;$\mathbf W_1^{(r)},\mathbf W_2^{(r)},
|
||
\mathbf H^{(r)},\mathbf P^{(r)}$ 仅本 rank 持有,$\mathbf P^{(r)}$ 是部分和而非最终输出}
|
||
{$\mathbf G^{(r)}=\mathrm{GeLU}(\mathbf H^{(r)})$ 逐元素、无跨 rank 依赖;列切 $\mathbf W_1$ 使 $\mathbf H$ 沿 $d_{\mathrm{ff}}$ 切分;行切 $\mathbf W_2$ 沿收缩维切分,
|
||
故 $\mathbf Y=\sum_r\mathbf P^{(r)}$ 需一次 All-Reduce,前向每层仅此一次通信}
|
||
\stsignature{TP-FFN(GeLU + All-Reduce)}{mb}
|
||
|
||
\end{tikzpicture}
|
||
\end{document}
|