Add flow layout: cursor placement, left rail, declared band heights
Figures were positioned by hand-written offsets. Every gap was a magic number tuned against the content that happened to be there, so a label that grew two characters landed on the next tensor, and two stages started from two different x shared no rail. Both failures compile cleanly. Replace it with a cursor. Objects placed with an empty coordinate argument reserve their own width -- including a stack's offset sheets and a bracket's overhang -- and gaps are declared once (\stgutter, \strowgap, \stblockgap). The gap belongs to the object that follows it and the first object in a band gets none, so every band starts flush on a shared rail and gap=0pt states that two shards tile exactly. \stlink makes a connector's label a flow object, which is what removes the label-wider-than-its-arrow failure entirely. \stcol is a vertical sub-flow for a split along the contracted axis. \strow declares its height, so an object that does not fit -- or a column that does not add up to what it declared, and is therefore drawn off-center -- becomes a package warning, which build.sh fails on. Absolute placement is unchanged: passing a coordinate takes the original code path, and \sttrack folds a hand-placed node back into the cursor. All three golden examples and the new tests/flow.tex are converted and build clean.
This commit is contained in:
@@ -3,6 +3,7 @@
|
||||
% (graded lightness), an INDEX face (discrete symbols, no ramp), and a BOOLEAN
|
||||
% support face (one flat level) -- plus a gather whose output heights are data
|
||||
% dependent and must sum back to T*k.
|
||||
% Layout is entirely by cursor: no absolute coordinate appears below.
|
||||
% ../scripts/build.sh moe-topk-gather.tex
|
||||
\documentclass[border=10pt]{standalone}
|
||||
\usepackage[cjk]{supertensor}
|
||||
@@ -22,84 +23,69 @@
|
||||
\begin{document}
|
||||
\begin{tikzpicture}
|
||||
|
||||
% ================================================================= formula ==
|
||||
\node (F) at (0,0) {\stformula{$\displaystyle
|
||||
\mathbf G=\mathrm{softmax}(\mathbf{XW}_g),\quad
|
||||
\mathcal I_t=\operatorname*{top-}k_{e}\,\mathbf G_{t,e},\quad
|
||||
\mathbf D_{t,e}=\mathbf 1[e\in\mathcal I_t],\quad
|
||||
\mathbf X^{(e)}=\mathrm{gather}(\mathbf X,\mathbf D_{:,e})$}};
|
||||
|
||||
% ============================================================ stage A row ===
|
||||
\node[st stage, below=7mm of F] (SA) {打分:token 对专家};
|
||||
\coordinate (a) at ($(SA)+(-3.6,-1.8)$);
|
||||
|
||||
\stface[role=act, bracket=true]{X}{(a)}{T}{d}
|
||||
\node[st op, right=6mm of X] (mA) {$\times$};
|
||||
\stface[role=wg]{Wg}{($(mA)+(1.6,0)$)}{d}{E}
|
||||
\node[st op, right=6mm of Wg] (eA) {$=$};
|
||||
\stface[role=s]{G}{($(eA)+(1.6,0)$)}{T}{E}
|
||||
|
||||
\node[inner sep=0pt, fit=(X)(Wg)(G)] (rowA) {};
|
||||
\stlane{rowA}
|
||||
\ststage{SA}{打分:token 对专家}
|
||||
\strow{rowA}{T}
|
||||
\stface[role=act, bracket=true]{X}{}{T}{d}
|
||||
\stglyph{mA}{$\times$}
|
||||
\stface[role=wg]{Wg}{}{d}{E}
|
||||
\stglyph{eA}{$=$}
|
||||
\stface[role=s]{G}{}{T}{E}
|
||||
\strowend
|
||||
\stcaption{X}{$\mathbf X$}{$T\times d$}
|
||||
\stcaption{Wg}{$\mathbf W_g$}{$d\times E$}
|
||||
\stcaption{G}{$\mathbf G$}{$T\times E$}
|
||||
\stnolane
|
||||
|
||||
% ============================================================ stage B row ===
|
||||
\node[st stage, below=9mm of X-shape.south west, anchor=north west] (SB)
|
||||
{取前 $k$:连续分数 $\rightarrow$ 离散选择};
|
||||
\coordinate (b) at ($(SB.west)+(0.6,-1.7)$);
|
||||
|
||||
% Indices are drawn as symbols, not as magnitudes: expert 3 is not "bigger"
|
||||
% than expert 0, so the index face gets no lightness ramp.
|
||||
\stindexface[role=idx]{I}{(b)}{T}{k}{0,1, 1,2, 2,3, 3,0, 0,2, 1,3}
|
||||
% The same routing decision as a boolean support: one flat level, exactly k
|
||||
% cells per row, and every unselected cell left unfilled.
|
||||
\stface[role=m, pattern=data, level=3,
|
||||
data={3300,0330,0033,3003,3030,0303}]{D}{($(I.east)+(3.1,0)$)}{T}{E}
|
||||
\starrowlabel{I.east}{D.west}{one-hot}
|
||||
|
||||
\node[st comm, right=9mm of D] (gz) {Gather};
|
||||
\starrow{D.east}{gz.west}
|
||||
|
||||
\node[inner sep=0pt, fit=(I)(D)(gz)] (rowB) {};
|
||||
\stlane{rowB}
|
||||
\ststage{SB}{取前 $k$:连续分数 $\rightarrow$ 离散选择}
|
||||
\strow{rowB}{T}
|
||||
% Indices are drawn as symbols, not as magnitudes: expert 3 is not "bigger"
|
||||
% than expert 0, so the index face gets no lightness ramp.
|
||||
\stindexface[role=idx]{I}{}{T}{k}{0,1, 1,2, 2,3, 3,0, 0,2, 1,3}
|
||||
\stlink{lb}{one-hot}
|
||||
% The same routing decision as a boolean support: one flat level, exactly k
|
||||
% cells per row, and every unselected cell left unfilled.
|
||||
\stface[role=m, pattern=data, level=3,
|
||||
data={3300,0330,0033,3003,3030,0303}]{D}{}{T}{E}
|
||||
\stlink{lg}{}
|
||||
\stcomm{gz}{Gather}
|
||||
\strowend
|
||||
\stcaption{I}{$\mathcal I$}{$T\times k$}
|
||||
\stcaption{D}{$\mathbf D$}{$T\times E$}
|
||||
\stnolane
|
||||
|
||||
% ============================================================ stage C row ===
|
||||
% Every stage heading starts on the same left rail; only the vertical position
|
||||
% follows the previous row.
|
||||
\coordinate (cy) at ($(I-shape.south)+(0,-9mm)$);
|
||||
\node[st stage, anchor=north west] (SC) at (SB.west |- cy)
|
||||
{按专家聚合:每个缓冲区的高度是数据决定的};
|
||||
\coordinate (c) at ($(SC.west)+(0.5,-1.9)$);
|
||||
|
||||
% Each buffer keeps X's width d -- gather regroups rows, it never reshapes the
|
||||
% feature axis. The heights are n_e, and they must sum to T*k.
|
||||
\stindexface[role=idx, border=false]{t0}{(c)}{ne}{1}{1,4,5}
|
||||
\stface[role=act]{B0}{($(t0.east)+(2*\stunit,0)$)}{ne}{d}
|
||||
\stindexface[role=idx, border=false]{t1}{($(B0.east)+(1.1,0)$)}{ne}{1}{1,2,6}
|
||||
\stface[role=act]{B1}{($(t1.east)+(2*\stunit,0)$)}{ne}{d}
|
||||
\stindexface[role=idx, border=false]{t2}{($(B1.east)+(1.1,0)$)}{ne}{1}{2,3,5}
|
||||
\stface[role=act]{B2}{($(t2.east)+(2*\stunit,0)$)}{ne}{d}
|
||||
\stindexface[role=idx, border=false]{t3}{($(B2.east)+(1.1,0)$)}{ne}{1}{3,4,6}
|
||||
\stface[role=act]{B3}{($(t3.east)+(2*\stunit,0)$)}{ne}{d}
|
||||
|
||||
\ststage{SC}{按专家聚合:每个缓冲区的高度是数据决定的}
|
||||
\strow{rowC}{ne}
|
||||
% Each buffer keeps X's width d -- gather regroups rows, it never reshapes
|
||||
% the feature axis. The heights are n_e, and they must sum to T*k.
|
||||
% A token column and its buffer are one object: tight gap inside the pair,
|
||||
% the standing gutter (widened) between pairs.
|
||||
\stindexface[role=idx, border=false]{t0}{}{ne}{1}{1,4,5}
|
||||
\stface[role=act, gap=2.5mm]{B0}{}{ne}{d}
|
||||
\stindexface[role=idx, border=false, gap=12mm]{t1}{}{ne}{1}{1,2,6}
|
||||
\stface[role=act, gap=2.5mm]{B1}{}{ne}{d}
|
||||
\stindexface[role=idx, border=false, gap=12mm]{t2}{}{ne}{1}{2,3,5}
|
||||
\stface[role=act, gap=2.5mm]{B2}{}{ne}{d}
|
||||
\stindexface[role=idx, border=false, gap=12mm]{t3}{}{ne}{1}{3,4,6}
|
||||
\stface[role=act, gap=2.5mm]{B3}{}{ne}{d}
|
||||
\strowend
|
||||
\stcaptiontop{t0}{\stshapefont{token}}
|
||||
|
||||
\node[inner sep=0pt, fit=(t0)(B3)] (rowC) {};
|
||||
\stlane{rowC}
|
||||
\sttrack{t0-top}
|
||||
\stcaption{B0}{$\mathbf X^{(1)}$}{$n_1\times d$}
|
||||
\stcaption{B1}{$\mathbf X^{(2)}$}{$n_2\times d$}
|
||||
\stcaption{B2}{$\mathbf X^{(3)}$}{$n_3\times d$}
|
||||
\stcaption{B3}{$\mathbf X^{(4)}$}{$n_4\times d$}
|
||||
\stnolane
|
||||
|
||||
% ================================================================= formula ==
|
||||
% Placed last so it is centered on the figure that was actually drawn.
|
||||
\sttopformula{F}{$\displaystyle
|
||||
\mathbf G=\mathrm{softmax}(\mathbf{XW}_g),\quad
|
||||
\mathcal I_t=\operatorname*{top-}k_{e}\,\mathbf G_{t,e},\quad
|
||||
\mathbf D_{t,e}=\mathbf 1[e\in\mathcal I_t],\quad
|
||||
\mathbf X^{(e)}=\mathrm{gather}(\mathbf X,\mathbf D_{:,e})$}
|
||||
|
||||
% ============================================================== meaning box ==
|
||||
\node[inner sep=0pt, fit=(F)(rowA)(rowB)(rowC)(B3-shape)(t0-top)] (all) {};
|
||||
\stbbox{all}
|
||||
\stmeaningbox{mb}{16.6cm}{all}
|
||||
{$T$ token 数,$d$ 模型宽度,$E$ 专家数(图中 $E=4$),$k$ 每 token 选中的专家数
|
||||
(图中 $k=2$),$n_e$ 落到第 $e$ 个专家的 token 数}
|
||||
|
||||
Reference in New Issue
Block a user