Files
dela 7b59c81d02 Add flow layout: cursor placement, left rail, declared band heights
Figures were positioned by hand-written offsets. Every gap was a magic
number tuned against the content that happened to be there, so a label
that grew two characters landed on the next tensor, and two stages
started from two different x shared no rail. Both failures compile
cleanly.

Replace it with a cursor. Objects placed with an empty coordinate
argument reserve their own width -- including a stack's offset sheets
and a bracket's overhang -- and gaps are declared once (\stgutter,
\strowgap, \stblockgap). The gap belongs to the object that follows it
and the first object in a band gets none, so every band starts flush on
a shared rail and gap=0pt states that two shards tile exactly.

\stlink makes a connector's label a flow object, which is what removes
the label-wider-than-its-arrow failure entirely. \stcol is a vertical
sub-flow for a split along the contracted axis. \strow declares its
height, so an object that does not fit -- or a column that does not add
up to what it declared, and is therefore drawn off-center -- becomes a
package warning, which build.sh fails on.

Absolute placement is unchanged: passing a coordinate takes the original
code path, and \sttrack folds a hand-placed node back into the cursor.

All three golden examples and the new tests/flow.tex are converted and
build clean.
2026-08-05 15:55:21 +08:00

101 lines
4.4 KiB
TeX

% Golden example 3 -- MoE top-k routing and gather.
% Shows the three cell semantics side by side in one figure: a SCORE face
% (graded lightness), an INDEX face (discrete symbols, no ramp), and a BOOLEAN
% support face (one flat level) -- plus a gather whose output heights are data
% dependent and must sum back to T*k.
% Layout is entirely by cursor: no absolute coordinate appears below.
% ../scripts/build.sh moe-topk-gather.tex
\documentclass[border=10pt]{standalone}
\usepackage[cjk]{supertensor}
\stsetrole{act}{stTeal} % X and the per-expert buffers: same data, regrouped
\stsetrole{wg}{stViolet} % learned router weight
\stsetrole{s}{stCoral} % scores
\stsetrole{idx}{stOrange} % indices / token ids
\stsetrole{m}{stGray} % boolean support
\stdim{T}{6}
\stdim{d}{4}
\stdim{E}{4}
\stdim{k}{2}
\stdim{ne}{3} % capacity per expert in this instance: n_e = 3
\begin{document}
\begin{tikzpicture}
% ============================================================ stage A row ===
\ststage{SA}{打分:token 对专家}
\strow{rowA}{T}
\stface[role=act, bracket=true]{X}{}{T}{d}
\stglyph{mA}{$\times$}
\stface[role=wg]{Wg}{}{d}{E}
\stglyph{eA}{$=$}
\stface[role=s]{G}{}{T}{E}
\strowend
\stcaption{X}{$\mathbf X$}{$T\times d$}
\stcaption{Wg}{$\mathbf W_g$}{$d\times E$}
\stcaption{G}{$\mathbf G$}{$T\times E$}
% ============================================================ stage B row ===
\ststage{SB}{取前 $k$:连续分数 $\rightarrow$ 离散选择}
\strow{rowB}{T}
% Indices are drawn as symbols, not as magnitudes: expert 3 is not "bigger"
% than expert 0, so the index face gets no lightness ramp.
\stindexface[role=idx]{I}{}{T}{k}{0,1, 1,2, 2,3, 3,0, 0,2, 1,3}
\stlink{lb}{one-hot}
% The same routing decision as a boolean support: one flat level, exactly k
% cells per row, and every unselected cell left unfilled.
\stface[role=m, pattern=data, level=3,
data={3300,0330,0033,3003,3030,0303}]{D}{}{T}{E}
\stlink{lg}{}
\stcomm{gz}{Gather}
\strowend
\stcaption{I}{$\mathcal I$}{$T\times k$}
\stcaption{D}{$\mathbf D$}{$T\times E$}
% ============================================================ stage C row ===
\ststage{SC}{按专家聚合:每个缓冲区的高度是数据决定的}
\strow{rowC}{ne}
% Each buffer keeps X's width d -- gather regroups rows, it never reshapes
% the feature axis. The heights are n_e, and they must sum to T*k.
% A token column and its buffer are one object: tight gap inside the pair,
% the standing gutter (widened) between pairs.
\stindexface[role=idx, border=false]{t0}{}{ne}{1}{1,4,5}
\stface[role=act, gap=2.5mm]{B0}{}{ne}{d}
\stindexface[role=idx, border=false, gap=12mm]{t1}{}{ne}{1}{1,2,6}
\stface[role=act, gap=2.5mm]{B1}{}{ne}{d}
\stindexface[role=idx, border=false, gap=12mm]{t2}{}{ne}{1}{2,3,5}
\stface[role=act, gap=2.5mm]{B2}{}{ne}{d}
\stindexface[role=idx, border=false, gap=12mm]{t3}{}{ne}{1}{3,4,6}
\stface[role=act, gap=2.5mm]{B3}{}{ne}{d}
\strowend
\stcaptiontop{t0}{\stshapefont{token}}
\sttrack{t0-top}
\stcaption{B0}{$\mathbf X^{(1)}$}{$n_1\times d$}
\stcaption{B1}{$\mathbf X^{(2)}$}{$n_2\times d$}
\stcaption{B2}{$\mathbf X^{(3)}$}{$n_3\times d$}
\stcaption{B3}{$\mathbf X^{(4)}$}{$n_4\times d$}
% ================================================================= formula ==
% Placed last so it is centered on the figure that was actually drawn.
\sttopformula{F}{$\displaystyle
\mathbf G=\mathrm{softmax}(\mathbf{XW}_g),\quad
\mathcal I_t=\operatorname*{top-}k_{e}\,\mathbf G_{t,e},\quad
\mathbf D_{t,e}=\mathbf 1[e\in\mathcal I_t],\quad
\mathbf X^{(e)}=\mathrm{gather}(\mathbf X,\mathbf D_{:,e})$}
% ============================================================== meaning box ==
\stbbox{all}
\stmeaningbox{mb}{16.6cm}{all}
{$T$ token 数,$d$ 模型宽度,$E$ 专家数(图中 $E=4$),$k$ 每 token 选中的专家数
(图中 $k=2$),$n_e$ 落到第 $e$ 个专家的 token 数}
{$\mathbf G$ 是分数,深浅可比大小;$\mathcal I$ 是索引,格内是符号不是数值,
故不用深浅;$\mathbf D$ 是布尔支撑,只有一档灰、每行恰好 $k$ 格;
$\mathbf X^{(e)}$ 与 $\mathbf X$ 同色,因为它是同一批数据换了分组}
{top-$k$ 把连续分数截成离散选择,这一步不可微;gather 只重排行、不动特征轴,
故每个缓冲区仍是 $d$ 宽;$\sum_e n_e=Tk$,图中 $4\times3=6\times2$}
\stsignature{MoE 路由:top-$k$ 选择与按专家 gather}{mb}
\end{tikzpicture}
\end{document}