Add flow layout: cursor placement, left rail, declared band heights

Figures were positioned by hand-written offsets. Every gap was a magic
number tuned against the content that happened to be there, so a label
that grew two characters landed on the next tensor, and two stages
started from two different x shared no rail. Both failures compile
cleanly.

Replace it with a cursor. Objects placed with an empty coordinate
argument reserve their own width -- including a stack's offset sheets
and a bracket's overhang -- and gaps are declared once (\stgutter,
\strowgap, \stblockgap). The gap belongs to the object that follows it
and the first object in a band gets none, so every band starts flush on
a shared rail and gap=0pt states that two shards tile exactly.

\stlink makes a connector's label a flow object, which is what removes
the label-wider-than-its-arrow failure entirely. \stcol is a vertical
sub-flow for a split along the contracted axis. \strow declares its
height, so an object that does not fit -- or a column that does not add
up to what it declared, and is therefore drawn off-center -- becomes a
package warning, which build.sh fails on.

Absolute placement is unchanged: passing a coordinate takes the original
code path, and \sttrack folds a hand-placed node back into the cursor.

All three golden examples and the new tests/flow.tex are converted and
build clean.
This commit is contained in:
dela
2026-08-05 15:55:21 +08:00
parent 7a22bef9e3
commit 7b59c81d02
11 changed files with 672 additions and 223 deletions
+44 -56
View File
@@ -2,6 +2,7 @@
% Shows: leading axes as stack depth, a transpose that physically swaps the
% face, equal edge length on the contracted axis, and a Boolean mask drawn in
% a different grammar from the scores it gates.
% Layout is entirely by cursor: no absolute coordinate appears below.
% ../scripts/build.sh mha-causal.tex
\documentclass[border=10pt]{standalone}
\usepackage[cjk]{supertensor}
@@ -19,79 +20,66 @@
\begin{document}
\begin{tikzpicture}
% ================================================================= formula ==
\node (F) at (0,0) {\stformula{$\displaystyle
\mathbf A^{(i)}=\mathrm{softmax}\!\left(
\frac{\mathbf Q^{(i)}\mathbf K^{(i)\top}}{\sqrt{d_h}}+\mathbf M\right),\qquad
\mathbf O^{(i)}=\mathbf A^{(i)}\mathbf V^{(i)}$}};
% ============================================================ stage A row ===
\node[st stage, below=7mm of F] (SA) {每头打分:沿 $d_h$ 收缩};
\coordinate (a) at ($(SA)+(-3.9,-1.9)$);
\ststack[role=q, bracket=true]{Q}{(a)}{T}{dh}{3}
\node[st op, right=6mm of Q] (mA) {$\times$};
% K^T: the face is physically swapped, not relabelled. Its height equals Q's
% width -- that is the contracted axis d_h, drawn at one edge length.
\ststack[role=k, bracket=true]{KT}{($(mA)+(1.9,0)$)}{dh}{T}{3}
\node[st op, right=6mm of KT] (eA) {$=$};
\ststack[role=s]{S}{($(eA)+(2.0,0)$)}{T}{T}{3}
\node[inner sep=0pt, fit=(Q)(KT)(S)] (rowA) {};
\stlane{rowA}
\ststage{SA}{每头打分:沿 $d_h$ 收缩}
\strow{rowA}{T}
\ststack[role=q, bracket=true]{Q}{}{T}{dh}{3}
\stglyph{mA}{$\times$}
% K^T: the face is physically swapped, not relabelled. Its height equals Q's
% width -- that is the contracted axis d_h, drawn at one edge length.
\ststack[role=k, bracket=true]{KT}{}{dh}{T}{3}
\stglyph{eA}{$=$}
\ststack[role=s]{S}{}{T}{T}{3}
\strowend
\stcaption{Q}{$\mathbf Q^{(i)}$}{$h\times T\times d_h$}
\stcaption{KT}{$\mathbf K^{(i)\top}$}{$h\times d_h\times T$}
\stcaption{S}{$\mathbf S^{(i)}$}{$h\times T\times T$}
\stnolane
% ============================================================ stage B row ===
\node[st stage, below=9mm of Q-shape.south west, anchor=north west] (SB)
{因果掩码与加权求和};
\coordinate (b) at ($(SB)+(0.6,-2.0)$);
% The mask is a Boolean support, not a magnitude: one flat level, exact
% triangle, no stack -- it is shared by every head.
\stface[role=w, pattern=data,
data={300000,330000,333000,333300,333330,333333}]{M}{(b)}{T}{T}
\ststack[role=s, pattern=causal]{A}{($(M.east)+(3.75,0)$)}{T}{T}{3}
\starrowlabel{M.east}{A.west}{softmax}
\node[st op, right=6mm of A] (mB) {$\times$};
\ststack[role=v, bracket=true]{V}{($(mB)+(1.4,0)$)}{T}{dh}{3}
\node[st op, right=6mm of V] (eB) {$=$};
\ststack[role=v]{O}{($(eB)+(1.4,0)$)}{T}{dh}{3}
\node[inner sep=0pt, fit=(M)(A)(V)(O)] (rowB) {};
\stlane{rowB}
\ststage{SB}{因果掩码与加权求和}
\strow{rowB}{T}
% The mask is a Boolean support, not a magnitude: one flat level, exact
% triangle, no stack -- it is shared by every head.
\stface[role=w, pattern=data,
data={300000,330000,333000,333300,333330,333333}]{M}{}{T}{T}
\stlink{lB}{softmax}
\ststack[role=s, pattern=causal]{A}{}{T}{T}{3}
\stglyph{mB}{$\times$}
\ststack[role=v, bracket=true]{V}{}{T}{dh}{3}
\stglyph{eB}{$=$}
\ststack[role=v]{O}{}{T}{dh}{3}
\strowend
\stcaption{M}{$\mathbf M$}{$T\times T$}
\stcaption{A}{$\mathbf A^{(i)}$}{$h\times T\times T$}
\stcaption{V}{$\mathbf V^{(i)}$}{$h\times T\times d_h$}
\stcaption{O}{$\mathbf O^{(i)}$}{$h\times T\times d_h$}
\stnolane
% ============================================================ stage C row ===
\node[st stage, below=9mm of M-shape.south west, anchor=north west] (SC)
{沿 $d_h$ 拼接后投影};
\coordinate (c) at ($(SC)+(1.2,-2.0)$);
% Concatenation reverses the split: three h-shards of width d_h tile a face of
% width d exactly.
\stface[role=v]{C1}{(c)}{T}{dh}
\stface[role=v]{C2}{($(C1.east)+(1.5*\stunit,0)$)}{T}{dh}
\stface[role=v]{C3}{($(C2.east)+(1.5*\stunit,0)$)}{T}{dh}
\node[st op, right=6mm of C3] (mC) {$\times$};
\stface[role=w]{WO}{($(mC)+(2.5,0)$)}{d}{d}
\node[st op, right=6mm of WO] (eC) {$=$};
\stface[role=v, bracket=true]{Y}{($(eC)+(2.5,0)$)}{T}{d}
\node[inner sep=0pt, fit=(C1)(WO)(Y)] (rowC) {};
\stlane{rowC}
\ststage{SC}{沿 $d_h$ 拼接后投影}
\strow{rowC}{d} % the d x d projection is the tallest object here
% Concatenation reverses the split: gap=0pt makes the three h-shards of width
% d_h tile a face of width d exactly, with no eyeballed offset.
\stface[role=v]{C1}{}{T}{dh}
\stface[role=v, gap=0pt]{C2}{}{T}{dh}
\stface[role=v, gap=0pt]{C3}{}{T}{dh}
\stglyph{mC}{$\times$}
\stface[role=w]{WO}{}{d}{d}
\stglyph{eC}{$=$}
\stface[role=v, bracket=true]{Y}{}{T}{d}
\strowend
\stcaption{C2}{$[\,\mathbf O^{(1)}\mid\mathbf O^{(2)}\mid\mathbf O^{(3)}\,]$}{$T\times d$}
\stcaption{WO}{$\mathbf W_O$}{$d\times d$}
\stcaption{Y}{$\mathbf Y$}{$T\times d$}
\stnolane
% ================================================================= formula ==
% Placed last so it is centered on the figure that was actually drawn.
\sttopformula{F}{$\displaystyle
\mathbf A^{(i)}=\mathrm{softmax}\!\left(
\frac{\mathbf Q^{(i)}\mathbf K^{(i)\top}}{\sqrt{d_h}}+\mathbf M\right),\qquad
\mathbf O^{(i)}=\mathbf A^{(i)}\mathbf V^{(i)}$}
% ============================================================== meaning box ==
\node[inner sep=0pt, fit=(F)(rowA)(rowB)(rowC)(Y-shape)(C2-shape)] (all) {};
\stbbox{all}
\stmeaningbox{mb}{16.8cm}{all}
{$T$ 序列长度,$d_h$ 单头宽度,$h$ 头数(图中 $h=3$,即堆叠的三张面),
$d=h\,d_h$;批轴 $B$ 省略}
+47 -61
View File
@@ -3,6 +3,7 @@
% (graded lightness), an INDEX face (discrete symbols, no ramp), and a BOOLEAN
% support face (one flat level) -- plus a gather whose output heights are data
% dependent and must sum back to T*k.
% Layout is entirely by cursor: no absolute coordinate appears below.
% ../scripts/build.sh moe-topk-gather.tex
\documentclass[border=10pt]{standalone}
\usepackage[cjk]{supertensor}
@@ -22,84 +23,69 @@
\begin{document}
\begin{tikzpicture}
% ================================================================= formula ==
\node (F) at (0,0) {\stformula{$\displaystyle
\mathbf G=\mathrm{softmax}(\mathbf{XW}_g),\quad
\mathcal I_t=\operatorname*{top-}k_{e}\,\mathbf G_{t,e},\quad
\mathbf D_{t,e}=\mathbf 1[e\in\mathcal I_t],\quad
\mathbf X^{(e)}=\mathrm{gather}(\mathbf X,\mathbf D_{:,e})$}};
% ============================================================ stage A row ===
\node[st stage, below=7mm of F] (SA) {打分:token 对专家};
\coordinate (a) at ($(SA)+(-3.6,-1.8)$);
\stface[role=act, bracket=true]{X}{(a)}{T}{d}
\node[st op, right=6mm of X] (mA) {$\times$};
\stface[role=wg]{Wg}{($(mA)+(1.6,0)$)}{d}{E}
\node[st op, right=6mm of Wg] (eA) {$=$};
\stface[role=s]{G}{($(eA)+(1.6,0)$)}{T}{E}
\node[inner sep=0pt, fit=(X)(Wg)(G)] (rowA) {};
\stlane{rowA}
\ststage{SA}{打分:token 对专家}
\strow{rowA}{T}
\stface[role=act, bracket=true]{X}{}{T}{d}
\stglyph{mA}{$\times$}
\stface[role=wg]{Wg}{}{d}{E}
\stglyph{eA}{$=$}
\stface[role=s]{G}{}{T}{E}
\strowend
\stcaption{X}{$\mathbf X$}{$T\times d$}
\stcaption{Wg}{$\mathbf W_g$}{$d\times E$}
\stcaption{G}{$\mathbf G$}{$T\times E$}
\stnolane
% ============================================================ stage B row ===
\node[st stage, below=9mm of X-shape.south west, anchor=north west] (SB)
{取前 $k$:连续分数 $\rightarrow$ 离散选择};
\coordinate (b) at ($(SB.west)+(0.6,-1.7)$);
% Indices are drawn as symbols, not as magnitudes: expert 3 is not "bigger"
% than expert 0, so the index face gets no lightness ramp.
\stindexface[role=idx]{I}{(b)}{T}{k}{0,1, 1,2, 2,3, 3,0, 0,2, 1,3}
% The same routing decision as a boolean support: one flat level, exactly k
% cells per row, and every unselected cell left unfilled.
\stface[role=m, pattern=data, level=3,
data={3300,0330,0033,3003,3030,0303}]{D}{($(I.east)+(3.1,0)$)}{T}{E}
\starrowlabel{I.east}{D.west}{one-hot}
\node[st comm, right=9mm of D] (gz) {Gather};
\starrow{D.east}{gz.west}
\node[inner sep=0pt, fit=(I)(D)(gz)] (rowB) {};
\stlane{rowB}
\ststage{SB}{取前 $k$:连续分数 $\rightarrow$ 离散选择}
\strow{rowB}{T}
% Indices are drawn as symbols, not as magnitudes: expert 3 is not "bigger"
% than expert 0, so the index face gets no lightness ramp.
\stindexface[role=idx]{I}{}{T}{k}{0,1, 1,2, 2,3, 3,0, 0,2, 1,3}
\stlink{lb}{one-hot}
% The same routing decision as a boolean support: one flat level, exactly k
% cells per row, and every unselected cell left unfilled.
\stface[role=m, pattern=data, level=3,
data={3300,0330,0033,3003,3030,0303}]{D}{}{T}{E}
\stlink{lg}{}
\stcomm{gz}{Gather}
\strowend
\stcaption{I}{$\mathcal I$}{$T\times k$}
\stcaption{D}{$\mathbf D$}{$T\times E$}
\stnolane
% ============================================================ stage C row ===
% Every stage heading starts on the same left rail; only the vertical position
% follows the previous row.
\coordinate (cy) at ($(I-shape.south)+(0,-9mm)$);
\node[st stage, anchor=north west] (SC) at (SB.west |- cy)
{按专家聚合:每个缓冲区的高度是数据决定的};
\coordinate (c) at ($(SC.west)+(0.5,-1.9)$);
% Each buffer keeps X's width d -- gather regroups rows, it never reshapes the
% feature axis. The heights are n_e, and they must sum to T*k.
\stindexface[role=idx, border=false]{t0}{(c)}{ne}{1}{1,4,5}
\stface[role=act]{B0}{($(t0.east)+(2*\stunit,0)$)}{ne}{d}
\stindexface[role=idx, border=false]{t1}{($(B0.east)+(1.1,0)$)}{ne}{1}{1,2,6}
\stface[role=act]{B1}{($(t1.east)+(2*\stunit,0)$)}{ne}{d}
\stindexface[role=idx, border=false]{t2}{($(B1.east)+(1.1,0)$)}{ne}{1}{2,3,5}
\stface[role=act]{B2}{($(t2.east)+(2*\stunit,0)$)}{ne}{d}
\stindexface[role=idx, border=false]{t3}{($(B2.east)+(1.1,0)$)}{ne}{1}{3,4,6}
\stface[role=act]{B3}{($(t3.east)+(2*\stunit,0)$)}{ne}{d}
\ststage{SC}{按专家聚合:每个缓冲区的高度是数据决定的}
\strow{rowC}{ne}
% Each buffer keeps X's width d -- gather regroups rows, it never reshapes
% the feature axis. The heights are n_e, and they must sum to T*k.
% A token column and its buffer are one object: tight gap inside the pair,
% the standing gutter (widened) between pairs.
\stindexface[role=idx, border=false]{t0}{}{ne}{1}{1,4,5}
\stface[role=act, gap=2.5mm]{B0}{}{ne}{d}
\stindexface[role=idx, border=false, gap=12mm]{t1}{}{ne}{1}{1,2,6}
\stface[role=act, gap=2.5mm]{B1}{}{ne}{d}
\stindexface[role=idx, border=false, gap=12mm]{t2}{}{ne}{1}{2,3,5}
\stface[role=act, gap=2.5mm]{B2}{}{ne}{d}
\stindexface[role=idx, border=false, gap=12mm]{t3}{}{ne}{1}{3,4,6}
\stface[role=act, gap=2.5mm]{B3}{}{ne}{d}
\strowend
\stcaptiontop{t0}{\stshapefont{token}}
\node[inner sep=0pt, fit=(t0)(B3)] (rowC) {};
\stlane{rowC}
\sttrack{t0-top}
\stcaption{B0}{$\mathbf X^{(1)}$}{$n_1\times d$}
\stcaption{B1}{$\mathbf X^{(2)}$}{$n_2\times d$}
\stcaption{B2}{$\mathbf X^{(3)}$}{$n_3\times d$}
\stcaption{B3}{$\mathbf X^{(4)}$}{$n_4\times d$}
\stnolane
% ================================================================= formula ==
% Placed last so it is centered on the figure that was actually drawn.
\sttopformula{F}{$\displaystyle
\mathbf G=\mathrm{softmax}(\mathbf{XW}_g),\quad
\mathcal I_t=\operatorname*{top-}k_{e}\,\mathbf G_{t,e},\quad
\mathbf D_{t,e}=\mathbf 1[e\in\mathcal I_t],\quad
\mathbf X^{(e)}=\mathrm{gather}(\mathbf X,\mathbf D_{:,e})$}
% ============================================================== meaning box ==
\node[inner sep=0pt, fit=(F)(rowA)(rowB)(rowC)(B3-shape)(t0-top)] (all) {};
\stbbox{all}
\stmeaningbox{mb}{16.6cm}{all}
{$T$ token 数,$d$ 模型宽度,$E$ 专家数(图中 $E=4$),$k$ 每 token 选中的专家数
(图中 $k=2$),$n_e$ 落到第 $e$ 个专家的 token 数}
+49 -50
View File
@@ -1,6 +1,8 @@
% Golden example 1 -- tensor parallel FFN, column-then-row sharding + AllReduce.
% Shows: partition geometry (shards tile the parent exactly), one hue per TP
% rank held across every stage, a collective node as a real operation.
% rank held across every stage, a collective node as a real operation, and a
% split along the contracted axis drawn as a column sub-flow.
% Layout is entirely by cursor: no absolute coordinate appears below.
% ../scripts/build.sh tp-ffn-allreduce.tex
\documentclass[border=10pt]{standalone}
\usepackage[cjk]{supertensor}
@@ -14,74 +16,71 @@
\stdim{bt}{6} % B*T rows
\stdim{d}{4} % model width
\stdim{dffl}{4} % d_ff / p (per-rank hidden width)
\stdim{dff}{8} % d_ff = p * (d_ff/p); the height of the stacked W_2 column
\begin{document}
\begin{tikzpicture}
% ================================================================= formula ==
\node (F) at (0,0) {\stformula{$\mathbf{XW}_1=[\,\mathbf{XW}_1^{(1)}\mid
\mathbf{XW}_1^{(2)}\,]=[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]$}};
\node[below=1.2mm of F] (F2) {\stformula{$\displaystyle
[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]
\begin{bmatrix}\mathbf W_2^{(1)}\\[-1pt]\mathbf W_2^{(2)}\end{bmatrix}
=\sum_{r}\mathbf H^{(r)}\mathbf W_2^{(r)}=\sum_r\mathbf P^{(r)}$}};
% ============================================================ stage A row ===
\node[st stage, below=7mm of F2] (SA) {列切 $\mathbf W_1$:无通信};
\coordinate (a) at ($(SA)+(-5.6,-1.8)$);
\stface[role=act, bracket=true]{X}{(a)}{bt}{d}
\node[st op, right=5mm of X] (mA) {$\times$};
% Two shards, tiled exactly: adjacent faces, no stretching, no gap.
\stface[role=r1]{W1a}{($(mA)+(1.5,0)$)}{d}{dffl}
\stface[role=r2]{W1b}{($(W1a.east)+(2*\stunit,0)$)}{d}{dffl}
\node[st op, right=5mm of W1b] (eA) {$=$};
\stface[role=r1]{Ha}{($(eA)+(1.5,0)$)}{bt}{dffl}
\stface[role=r2]{Hb}{($(Ha.east)+(2*\stunit,0)$)}{bt}{dffl}
\node[inner sep=0pt, fit=(X)(W1a)(Ha)(Hb)] (rowA) {};
\stlane{rowA}
\ststage{SA}{列切 $\mathbf W_1$:无通信}
\strow{rowA}{bt}
\stface[role=act, bracket=true]{X}{}{bt}{d}
\stglyph{mA}{$\times$}
% Two shards, tiled exactly: gap=0pt makes them adjacent by construction,
% so neither stretching nor an eyeballed offset can creep in.
\stface[role=r1]{W1a}{}{d}{dffl}
\stface[role=r2, gap=0pt]{W1b}{}{d}{dffl}
\stglyph{eA}{$=$}
\stface[role=r1]{Ha}{}{bt}{dffl}
\stface[role=r2, gap=0pt]{Hb}{}{bt}{dffl}
\strowend
\stcaption{X}{$\mathbf X$}{$BT\times d$}
\stcaption{W1a}{$\mathbf W_1^{(1)}$}{$d\times d_{\mathrm{ff}}/p$}
\stcaption{W1b}{$\mathbf W_1^{(2)}$}{$d\times d_{\mathrm{ff}}/p$}
\stcaption{Ha}{$\mathbf H^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stcaption{Hb}{$\mathbf H^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stnolane
% ============================================================ stage B row ===
\node[st stage, below=9mm of X-shape.south west, anchor=north west] (SB)
{行切 $\mathbf W_2$:一次 All-Reduce};
\coordinate (b) at ($(SB)+(-0.4,-1.9)$);
\stface[role=r1]{Ga}{(b)}{bt}{dffl}
\stface[role=r2]{Gb}{($(Ga.east)+(2*\stunit,0)$)}{bt}{dffl}
\node[st op, right=5mm of Gb] (mB) {$\times$};
% W_2 is split along the CONTRACTED axis: the two shards stack vertically and
% together have exactly the height of H's width. Splitting reverses concat.
\stface[role=r1]{W2a}{($(mB)+(1.35,0.46)$)}{dffl}{d}
\stface[role=r2]{W2b}{($(W2a.south)+(0,-2*\stunit)$)}{dffl}{d}
\node[st op, right=5mm of W2a.east |- W2a.south] (eB) {$=$};
\stface[role=r1]{Pa}{($(eB)+(1.3,0)$)}{bt}{d}
\node[st op, right=4mm of Pa] (plus) {$+$};
\stface[role=r2]{Pb}{($(plus)+(1.3,0)$)}{bt}{d}
\node[st comm, right=9mm of Pb] (ar) {All-Reduce};
\stface[role=act, bracket=true]{Y}{($(ar)+(1.9,0)$)}{bt}{d}
\starrow{Pb.east}{ar.west}
\starrow{ar.east}{Y.west}
\node[inner sep=0pt, fit=(Ga)(W2a)(W2b)(Pa)(Pb)(Y)] (rowB) {};
\stlane{rowB}
\ststage{SB}{行切 $\mathbf W_2$:一次 All-Reduce}
\strow{rowB}{dff} % the stacked W_2 column is the tallest object here
\stface[role=r1]{Ga}{}{bt}{dffl}
\stface[role=r2, gap=0pt]{Gb}{}{bt}{dffl}
\stglyph{mB}{$\times$}
% W_2 is split along the CONTRACTED axis: the shards stack vertically and
% together have exactly the height of H's width. Splitting reverses concat,
% which is what gap=0pt inside the column states.
\stcol{W2}{dff}
\stface[role=r1]{W2a}{}{dffl}{d}
\stface[role=r2, gap=0pt]{W2b}{}{dffl}{d}
\stcolend
\stglyph{eB}{$=$}
\stface[role=r1]{Pa}{}{bt}{d}
\stglyph{plus}{$+$}
\stface[role=r2]{Pb}{}{bt}{d}
\stlink{lc}{}
\stcomm{ar}{All-Reduce}
\stlink{ly}{}
\stface[role=act, bracket=true]{Y}{}{bt}{d}
\strowend
\stcaption{Ga}{$\mathbf G^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stcaption{Gb}{$\mathbf G^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stcaption{W2b}{$\mathbf W_2^{(r)}$}{$d_{\mathrm{ff}}/p\times d$}
\stcaption{W2}{$\mathbf W_2^{(r)}$}{$d_{\mathrm{ff}}/p\times d$}
\stcaption{Pa}{$\mathbf P^{(1)}$}{$BT\times d$}
\stcaption{Pb}{$\mathbf P^{(2)}$}{$BT\times d$}
\stcaption{Y}{$\mathbf Y$}{$BT\times d$}
\stnolane
% ================================================================= formula ==
% Placed last so it is centered on the figure that was actually drawn.
\sttopformula{F}{$\begin{gathered}
\mathbf{XW}_1=[\,\mathbf{XW}_1^{(1)}\mid
\mathbf{XW}_1^{(2)}\,]=[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]\\[1.2mm]
[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]
\begin{bmatrix}\mathbf W_2^{(1)}\\[-1pt]\mathbf W_2^{(2)}\end{bmatrix}
=\sum_{r}\mathbf H^{(r)}\mathbf W_2^{(r)}=\sum_r\mathbf P^{(r)}
\end{gathered}$}
% ============================================================== meaning box ==
\node[inner sep=0pt, fit=(F)(rowA)(rowB)(Y-shape)(Ga-shape)] (all) {};
\stbbox{all}
\stmeaningbox{mb}{16.4cm}{all}
{$BT$ 展平后的 token 数,$d$ 模型宽度,$d_{\mathrm{ff}}$ 前馈中间宽度,
$p$ TP 并行度(图中 $p=2$)}