supertensor: shape-aware tensor figure toolkit

Extracted from the tensor-formula-viz skill and rebuilt around the idea that
the geometry rules should be enforced by construction rather than restated as
prose an agent has to remember.

- assets/supertensor.sty: faces, stacks, index faces, shared caption lanes,
  meaning box, signature. Macros take a declared axis and a declared role, so
  equal shapes get equal edges, a x a is square, a transpose swaps the face,
  and contracted axes share an edge length -- without any manual alignment.
- scripts/preflight.sh: decide the TikZ/CJK path before drawing.
- scripts/build.sh: compile and fail on silent corruption (missing CJK glyphs,
  overfull boxes, undeclared roles), then export pdf/svg/png/thumb.
- scripts/test.sh: build every figure as a regression test for the package.
- examples/: three golden figures (TP-FFN, causal MHA, MoE top-k gather) plus
  an anti-pattern gallery of figures that compile cleanly and still lie.
- SKILL.md + references/: lean entry point, details loaded on demand.
This commit is contained in:
dela
2026-08-05 12:17:33 +08:00
commit 7a22bef9e3
20 changed files with 1661 additions and 0 deletions
+95
View File
@@ -0,0 +1,95 @@
% Golden example 1 -- tensor parallel FFN, column-then-row sharding + AllReduce.
% Shows: partition geometry (shards tile the parent exactly), one hue per TP
% rank held across every stage, a collective node as a real operation.
% ../scripts/build.sh tp-ffn-allreduce.tex
\documentclass[border=10pt]{standalone}
\usepackage[cjk]{supertensor}
% --- role ledger: one hue per TP rank, held from W through H to P ----------
\stsetrole{act}{stViolet} % activations that every rank sees
\stsetrole{r1}{stTeal} % rank 1
\stsetrole{r2}{stOrange} % rank 2
% --- geometry ledger: one physical edge per symbolic axis ------------------
\stdim{bt}{6} % B*T rows
\stdim{d}{4} % model width
\stdim{dffl}{4} % d_ff / p (per-rank hidden width)
\begin{document}
\begin{tikzpicture}
% ================================================================= formula ==
\node (F) at (0,0) {\stformula{$\mathbf{XW}_1=[\,\mathbf{XW}_1^{(1)}\mid
\mathbf{XW}_1^{(2)}\,]=[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]$}};
\node[below=1.2mm of F] (F2) {\stformula{$\displaystyle
[\,\mathbf H^{(1)}\mid\mathbf H^{(2)}\,]
\begin{bmatrix}\mathbf W_2^{(1)}\\[-1pt]\mathbf W_2^{(2)}\end{bmatrix}
=\sum_{r}\mathbf H^{(r)}\mathbf W_2^{(r)}=\sum_r\mathbf P^{(r)}$}};
% ============================================================ stage A row ===
\node[st stage, below=7mm of F2] (SA) {列切 $\mathbf W_1$:无通信};
\coordinate (a) at ($(SA)+(-5.6,-1.8)$);
\stface[role=act, bracket=true]{X}{(a)}{bt}{d}
\node[st op, right=5mm of X] (mA) {$\times$};
% Two shards, tiled exactly: adjacent faces, no stretching, no gap.
\stface[role=r1]{W1a}{($(mA)+(1.5,0)$)}{d}{dffl}
\stface[role=r2]{W1b}{($(W1a.east)+(2*\stunit,0)$)}{d}{dffl}
\node[st op, right=5mm of W1b] (eA) {$=$};
\stface[role=r1]{Ha}{($(eA)+(1.5,0)$)}{bt}{dffl}
\stface[role=r2]{Hb}{($(Ha.east)+(2*\stunit,0)$)}{bt}{dffl}
\node[inner sep=0pt, fit=(X)(W1a)(Ha)(Hb)] (rowA) {};
\stlane{rowA}
\stcaption{X}{$\mathbf X$}{$BT\times d$}
\stcaption{W1a}{$\mathbf W_1^{(1)}$}{$d\times d_{\mathrm{ff}}/p$}
\stcaption{W1b}{$\mathbf W_1^{(2)}$}{$d\times d_{\mathrm{ff}}/p$}
\stcaption{Ha}{$\mathbf H^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stcaption{Hb}{$\mathbf H^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stnolane
% ============================================================ stage B row ===
\node[st stage, below=9mm of X-shape.south west, anchor=north west] (SB)
{行切 $\mathbf W_2$:一次 All-Reduce};
\coordinate (b) at ($(SB)+(-0.4,-1.9)$);
\stface[role=r1]{Ga}{(b)}{bt}{dffl}
\stface[role=r2]{Gb}{($(Ga.east)+(2*\stunit,0)$)}{bt}{dffl}
\node[st op, right=5mm of Gb] (mB) {$\times$};
% W_2 is split along the CONTRACTED axis: the two shards stack vertically and
% together have exactly the height of H's width. Splitting reverses concat.
\stface[role=r1]{W2a}{($(mB)+(1.35,0.46)$)}{dffl}{d}
\stface[role=r2]{W2b}{($(W2a.south)+(0,-2*\stunit)$)}{dffl}{d}
\node[st op, right=5mm of W2a.east |- W2a.south] (eB) {$=$};
\stface[role=r1]{Pa}{($(eB)+(1.3,0)$)}{bt}{d}
\node[st op, right=4mm of Pa] (plus) {$+$};
\stface[role=r2]{Pb}{($(plus)+(1.3,0)$)}{bt}{d}
\node[st comm, right=9mm of Pb] (ar) {All-Reduce};
\stface[role=act, bracket=true]{Y}{($(ar)+(1.9,0)$)}{bt}{d}
\starrow{Pb.east}{ar.west}
\starrow{ar.east}{Y.west}
\node[inner sep=0pt, fit=(Ga)(W2a)(W2b)(Pa)(Pb)(Y)] (rowB) {};
\stlane{rowB}
\stcaption{Ga}{$\mathbf G^{(1)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stcaption{Gb}{$\mathbf G^{(2)}$}{$BT\times d_{\mathrm{ff}}/p$}
\stcaption{W2b}{$\mathbf W_2^{(r)}$}{$d_{\mathrm{ff}}/p\times d$}
\stcaption{Pa}{$\mathbf P^{(1)}$}{$BT\times d$}
\stcaption{Pb}{$\mathbf P^{(2)}$}{$BT\times d$}
\stcaption{Y}{$\mathbf Y$}{$BT\times d$}
\stnolane
% ============================================================== meaning box ==
\node[inner sep=0pt, fit=(F)(rowA)(rowB)(Y-shape)(Ga-shape)] (all) {};
\stmeaningbox{mb}{16.4cm}{all}
{$BT$ 展平后的 token 数,$d$ 模型宽度,$d_{\mathrm{ff}}$ 前馈中间宽度,
$p$ TP 并行度(图中 $p=2$)}
{$\mathbf X,\mathbf Y$ 每个 rank 完整持有;$\mathbf W_1^{(r)},\mathbf W_2^{(r)},
\mathbf H^{(r)},\mathbf P^{(r)}$ 仅本 rank 持有,$\mathbf P^{(r)}$ 是部分和而非最终输出}
{$\mathbf G^{(r)}=\mathrm{GeLU}(\mathbf H^{(r)})$ 逐元素、无跨 rank 依赖;列切 $\mathbf W_1$ 使 $\mathbf H$ 沿 $d_{\mathrm{ff}}$ 切分;行切 $\mathbf W_2$ 沿收缩维切分,
故 $\mathbf Y=\sum_r\mathbf P^{(r)}$ 需一次 All-Reduce,前向每层仅此一次通信}
\stsignature{TP-FFN(GeLU + All-Reduce)}{mb}
\end{tikzpicture}
\end{document}