Files
SuperTensor/examples/tp-ffn-allreduce.tex
T
dela 7b59c81d02 Add flow layout: cursor placement, left rail, declared band heights
Figures were positioned by hand-written offsets. Every gap was a magic
number tuned against the content that happened to be there, so a label
that grew two characters landed on the next tensor, and two stages
started from two different x shared no rail. Both failures compile
cleanly.

Replace it with a cursor. Objects placed with an empty coordinate
argument reserve their own width -- including a stack's offset sheets
and a bracket's overhang -- and gaps are declared once (\stgutter,
\strowgap, \stblockgap). The gap belongs to the object that follows it
and the first object in a band gets none, so every band starts flush on
a shared rail and gap=0pt states that two shards tile exactly.

\stlink makes a connector's label a flow object, which is what removes
the label-wider-than-its-arrow failure entirely. \stcol is a vertical
sub-flow for a split along the contracted axis. \strow declares its
height, so an object that does not fit -- or a column that does not add
up to what it declared, and is therefore drawn off-center -- becomes a
package warning, which build.sh fails on.

Absolute placement is unchanged: passing a coordinate takes the original
code path, and \sttrack folds a hand-placed node back into the cursor.

All three golden examples and the new tests/flow.tex are converted and
build clean.
2026-08-05 15:55:21 +08:00

95 lines
4.3 KiB
TeX
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
% Golden example 1 -- tensor parallel FFN, column-then-row sharding + AllReduce.
% Shows: partition geometry (shards tile the parent exactly), one hue per TP
% rank held across every stage, a collective node as a real operation, and a
% split along the contracted axis drawn as a column sub-flow.
% Layout is entirely by cursor: no absolute coordinate appears below.
% ../scripts/build.sh tp-ffn-allreduce.tex
\documentclass[border=10pt]{standalone}
\usepackage[cjk]{supertensor}
% --- role ledger: one hue per TP rank, held from W through H to P ----------
\stsetrole{act}{stViolet} % activations that every rank sees
\stsetrole{r1}{stTeal} % rank 1
\stsetrole{r2}{stOrange} % rank 2
% --- geometry ledger: one physical edge per symbolic axis ------------------
\stdim{bt}{6} % B*T rows
\stdim{d}{4} % model width
\stdim{dffl}{4} % d_ff / p (per-rank hidden width)
\stdim{dff}{8} % d_ff = p * (d_ff/p); the height of the stacked W_2 column
\begin{document}
\begin{tikzpicture}
% ============================================================ stage A row ===
\ststage{SA}{列切 $\mathbf W_1$:无通信}
\strow{rowA}{bt}
\stface[role=act, bracket=true]{X}{}{bt}{d}
\stglyph{mA}{$\times$}
% Two shards, tiled exactly: gap=0pt makes them adjacent by construction,
% so neither stretching nor an eyeballed offset can creep in.
\stface[role=r1]{W1a}{}{d}{dffl}
\stface[role=r2, gap=0pt]{W1b}{}{d}{dffl}
\stglyph{eA}{$=$}
\stface[role=r1]{Ha}{}{bt}{dffl}
\stface[role=r2, gap=0pt]{Hb}{}{bt}{dffl}
\strowend
\stcaption{X}{$\mathbf X$}{$BT\times d$}
\stcaption{W1a}{$\mathbf W_1^{(1)}$}{$d\times d_{\mathrm{ff}}/p$}
\stcaption{W1b}{$\mathbf W_1^{(2)}$}{$d\times d_{\mathrm{ff}}/p$}
\stcaption{Ha}{$\mathbf H^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stcaption{Hb}{$\mathbf H^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$}
% ============================================================ stage B row ===
\ststage{SB}{行切 $\mathbf W_2$:一次 All-Reduce}
\strow{rowB}{dff} % the stacked W_2 column is the tallest object here
\stface[role=r1]{Ga}{}{bt}{dffl}
\stface[role=r2, gap=0pt]{Gb}{}{bt}{dffl}
\stglyph{mB}{$\times$}
% W_2 is split along the CONTRACTED axis: the shards stack vertically and
% together have exactly the height of H's width. Splitting reverses concat,
% which is what gap=0pt inside the column states.
\stcol{W2}{dff}
\stface[role=r1]{W2a}{}{dffl}{d}
\stface[role=r2, gap=0pt]{W2b}{}{dffl}{d}
\stcolend
\stglyph{eB}{$=$}
\stface[role=r1]{Pa}{}{bt}{d}
\stglyph{plus}{$+$}
\stface[role=r2]{Pb}{}{bt}{d}
\stlink{lc}{}
\stcomm{ar}{All-Reduce}
\stlink{ly}{}
\stface[role=act, bracket=true]{Y}{}{bt}{d}
\strowend
\stcaption{Ga}{$\mathbf G^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stcaption{Gb}{$\mathbf G^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stcaption{W2}{$\mathbf W_2^{(r)}$}{$d_{\mathrm{ff}}/p\times d$}
\stcaption{Pa}{$\mathbf P^{(1)}$}{$BT\times d$}
\stcaption{Pb}{$\mathbf P^{(2)}$}{$BT\times d$}
\stcaption{Y}{$\mathbf Y$}{$BT\times d$}
% ================================================================= formula ==
% Placed last so it is centered on the figure that was actually drawn.
\sttopformula{F}{$\begin{gathered}
\mathbf{XW}_1=[\,\mathbf{XW}_1^{(1)}\mid
\mathbf{XW}_1^{(2)}\,]=[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]\\[1.2mm]
[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]
\begin{bmatrix}\mathbf W_2^{(1)}\\[-1pt]\mathbf W_2^{(2)}\end{bmatrix}
=\sum_{r}\mathbf H^{(r)}\mathbf W_2^{(r)}=\sum_r\mathbf P^{(r)}
\end{gathered}$}
% ============================================================== meaning box ==
\stbbox{all}
\stmeaningbox{mb}{16.4cm}{all}
{$BT$ 展平后的 token 数,$d$ 模型宽度,$d_{\mathrm{ff}}$ 前馈中间宽度,
$p$ TP 并行度(图中 $p=2$)}
{$\mathbf X,\mathbf Y$ 每个 rank 完整持有;$\mathbf W_1^{(r)},\mathbf W_2^{(r)},
\mathbf H^{(r)},\mathbf P^{(r)}$ 仅本 rank 持有,$\mathbf P^{(r)}$ 是部分和而非最终输出}
{$\mathbf G^{(r)}=\mathrm{GeLU}(\mathbf H^{(r)})$ 逐元素、无跨 rank 依赖;列切 $\mathbf W_1$ 使 $\mathbf H$ 沿 $d_{\mathrm{ff}}$ 切分;行切 $\mathbf W_2$ 沿收缩维切分,
故 $\mathbf Y=\sum_r\mathbf P^{(r)}$ 需一次 All-Reduce,前向每层仅此一次通信}
\stsignature{TP-FFN(GeLU + All-Reduce)}{mb}
\end{tikzpicture}
\end{document}