% Golden example 3 -- MoE top-k routing and gather. % Shows the three cell semantics side by side in one figure: a SCORE face % (graded lightness), an INDEX face (discrete symbols, no ramp), and a BOOLEAN % support face (one flat level) -- plus a gather whose output heights are data % dependent and must sum back to T*k. % Layout is entirely by cursor: no absolute coordinate appears below. % ../scripts/build.sh moe-topk-gather.tex \documentclass[border=10pt]{standalone} \usepackage[cjk]{supertensor} \stsetrole{act}{stTeal} % X and the per-expert buffers: same data, regrouped \stsetrole{wg}{stViolet} % learned router weight \stsetrole{s}{stCoral} % scores \stsetrole{idx}{stOrange} % indices / token ids \stsetrole{m}{stGray} % boolean support \stdim{T}{6} \stdim{d}{4} \stdim{E}{4} \stdim{k}{2} \stdim{ne}{3} % capacity per expert in this instance: n_e = 3 \begin{document} \begin{tikzpicture} % ============================================================ stage A row === \ststage{SA}{打分:token 对专家} \strow{rowA}{T} \stface[role=act, bracket=true]{X}{}{T}{d} \stglyph{mA}{$\times$} \stface[role=wg]{Wg}{}{d}{E} \stglyph{eA}{$=$} \stface[role=s]{G}{}{T}{E} \strowend \stcaption{X}{$\mathbf X$}{$T\times d$} \stcaption{Wg}{$\mathbf W_g$}{$d\times E$} \stcaption{G}{$\mathbf G$}{$T\times E$} % ============================================================ stage B row === \ststage{SB}{取前 $k$:连续分数 $\rightarrow$ 离散选择} \strow{rowB}{T} % Indices are drawn as symbols, not as magnitudes: expert 3 is not "bigger" % than expert 0, so the index face gets no lightness ramp. \stindexface[role=idx]{I}{}{T}{k}{0,1, 1,2, 2,3, 3,0, 0,2, 1,3} \stlink{lb}{one-hot} % The same routing decision as a boolean support: one flat level, exactly k % cells per row, and every unselected cell left unfilled. \stface[role=m, pattern=data, level=3, data={3300,0330,0033,3003,3030,0303}]{D}{}{T}{E} \stlink{lg}{} \stcomm{gz}{Gather} \strowend \stcaption{I}{$\mathcal I$}{$T\times k$} \stcaption{D}{$\mathbf D$}{$T\times E$} % ============================================================ stage C row === \ststage{SC}{按专家聚合:每个缓冲区的高度是数据决定的} \strow{rowC}{ne} % Each buffer keeps X's width d -- gather regroups rows, it never reshapes % the feature axis. The heights are n_e, and they must sum to T*k. % A token column and its buffer are one object: tight gap inside the pair, % the standing gutter (widened) between pairs. \stindexface[role=idx, border=false]{t0}{}{ne}{1}{1,4,5} \stface[role=act, gap=2.5mm]{B0}{}{ne}{d} \stindexface[role=idx, border=false, gap=12mm]{t1}{}{ne}{1}{1,2,6} \stface[role=act, gap=2.5mm]{B1}{}{ne}{d} \stindexface[role=idx, border=false, gap=12mm]{t2}{}{ne}{1}{2,3,5} \stface[role=act, gap=2.5mm]{B2}{}{ne}{d} \stindexface[role=idx, border=false, gap=12mm]{t3}{}{ne}{1}{3,4,6} \stface[role=act, gap=2.5mm]{B3}{}{ne}{d} \strowend \stcaptiontop{t0}{\stshapefont{token}} \sttrack{t0-top} \stcaption{B0}{$\mathbf X^{(1)}$}{$n_1\times d$} \stcaption{B1}{$\mathbf X^{(2)}$}{$n_2\times d$} \stcaption{B2}{$\mathbf X^{(3)}$}{$n_3\times d$} \stcaption{B3}{$\mathbf X^{(4)}$}{$n_4\times d$} % ================================================================= formula == % Placed last so it is centered on the figure that was actually drawn. \sttopformula{F}{$\displaystyle \mathbf G=\mathrm{softmax}(\mathbf{XW}_g),\quad \mathcal I_t=\operatorname*{top-}k_{e}\,\mathbf G_{t,e},\quad \mathbf D_{t,e}=\mathbf 1[e\in\mathcal I_t],\quad \mathbf X^{(e)}=\mathrm{gather}(\mathbf X,\mathbf D_{:,e})$} % ============================================================== meaning box == \stbbox{all} \stmeaningbox{mb}{16.6cm}{all} {$T$ token 数,$d$ 模型宽度,$E$ 专家数(图中 $E=4$),$k$ 每 token 选中的专家数 (图中 $k=2$),$n_e$ 落到第 $e$ 个专家的 token 数} {$\mathbf G$ 是分数,深浅可比大小;$\mathcal I$ 是索引,格内是符号不是数值, 故不用深浅;$\mathbf D$ 是布尔支撑,只有一档灰、每行恰好 $k$ 格; $\mathbf X^{(e)}$ 与 $\mathbf X$ 同色,因为它是同一批数据换了分组} {top-$k$ 把连续分数截成离散选择,这一步不可微;gather 只重排行、不动特征轴, 故每个缓冲区仍是 $d$ 宽;$\sum_e n_e=Tk$,图中 $4\times3=6\times2$} \stsignature{MoE 路由:top-$k$ 选择与按专家 gather}{mb} \end{tikzpicture} \end{document}