feat: land superpaper v1 notes scaffold
Add the ledger schema, class router, lint codes, ingest/build pipeline, and three work-tree examples: align derivation, superfig delegation, and supertensor delegation.
@@ -0,0 +1,46 @@
|
||||
schema: superpaper.ledger/v1
|
||||
retired_ids: []
|
||||
paper:
|
||||
id: excerpt-toy
|
||||
title: A One-Step Predictor
|
||||
authors: ["Fixture"]
|
||||
year: 2026
|
||||
venue: Superpaper examples
|
||||
notes_language: zh
|
||||
source: {kind: excerpt}
|
||||
coverage:
|
||||
mode: excerpt
|
||||
sections_in: ["1"]
|
||||
questions:
|
||||
- {id: Q1, text: "一次前向如何得到预测,损失该放在哪?", source: "excerpt"}
|
||||
claims:
|
||||
- id: C1
|
||||
text: "一次前向是 x → f_θ → ŷ;损失在预测之后单独计算"
|
||||
kind: contribution
|
||||
status: core
|
||||
supports: [Q1]
|
||||
source: "excerpt"
|
||||
definitions:
|
||||
- {id: D1, name: "one-step predictor", text: "ŷ = f_θ(x)", source: "excerpt"}
|
||||
assumptions: []
|
||||
lemmas: []
|
||||
symbols:
|
||||
- {name: x, latex: "x", meaning: "输入", kind: value, introduced: "excerpt"}
|
||||
- {name: yhat, latex: "\\hat y", meaning: "预测", kind: value}
|
||||
- {name: L, latex: "L", meaning: "损失", kind: scalar}
|
||||
- {name: d, latex: "d", meaning: "特征维", kind: "shape parameter"}
|
||||
derivations:
|
||||
- id: DER1
|
||||
claim: C1
|
||||
title: "缩放来自方差"
|
||||
source: "excerpt"
|
||||
expand: true
|
||||
figure: null
|
||||
steps:
|
||||
- {id: S1, from: "u^\\top v", to: "u^\\top v / \\sqrt{d}", rule: scale,
|
||||
justify: "点积方差随 d 增长"}
|
||||
figures: []
|
||||
evidence: []
|
||||
terms:
|
||||
- {canonical: "one-step predictor", aliases: ["一次前向"]}
|
||||
source_assets: []
|
||||
@@ -0,0 +1,29 @@
|
||||
\documentclass[a4paper]{article}
|
||||
\input{notes-macros}
|
||||
\renewcommand{\notetitle}{一次前向预测器}
|
||||
\renewcommand{\noteauthors}{Superpaper fixture}
|
||||
\renewcommand{\notepaper}{A One-Step Predictor}
|
||||
\renewcommand{\notevenue}{examples/excerpt-toy}
|
||||
\begin{document}
|
||||
\begin{titlepage}
|
||||
\centering
|
||||
\vspace{2cm}
|
||||
{\huge\bfseries \notetitle\par}
|
||||
\vspace{1cm}
|
||||
{\large \notepaper\par}
|
||||
\vspace{0.5cm}
|
||||
{\large \noteauthors\par}
|
||||
\end{titlepage}
|
||||
\tableofcontents
|
||||
\newpage
|
||||
\input{sections/sec-01.tex}
|
||||
\input{sections/sec-02.tex}
|
||||
\input{sections/sec-03.tex}
|
||||
\input{sections/sec-04.tex}
|
||||
\input{sections/sec-05.tex}
|
||||
\appendix
|
||||
\section{符号表}
|
||||
\input{sections/symbols.tex}
|
||||
\section{推导链一览}
|
||||
DER1:缩放来自方差,见 \spref{C1}。
|
||||
\end{document}
|
||||
@@ -0,0 +1,3 @@
|
||||
\section{这篇论文在问什么}
|
||||
要把输入变成预测,最简单的机制是什么?损失要不要走在前向主路上?
|
||||
这是摘录 fixture,不假装读完全文。
|
||||
@@ -0,0 +1,3 @@
|
||||
\section{主张与贡献}
|
||||
\splabel{C1}
|
||||
一次前向是 $x \to f_\theta \to \hat y$。损失 $L(\hat y,y)$ 在预测之后单独计算,不是主路上的一站。
|
||||
@@ -0,0 +1,2 @@
|
||||
\section{预备:定义、假设、符号}
|
||||
预测器定义为 $\hat y = f_\theta(x)$。符号见附录。
|
||||
@@ -0,0 +1,18 @@
|
||||
\section{一次前向与损失}
|
||||
内积的方差会随维数 $d$ 涨。为了不让后续非线性饱和,要把点积除掉 $\sqrt{d}$。
|
||||
|
||||
\begin{align}
|
||||
u^\top v &\longrightarrow \frac{u^\top v}{\sqrt{d}}.
|
||||
\end{align}
|
||||
|
||||
\begin{itemize}
|
||||
\item $u,v$ — 两个 $d$ 维向量
|
||||
\item $d$ — 特征维(shape parameter)
|
||||
\end{itemize}
|
||||
|
||||
\begin{importantbox}{主路与损失}
|
||||
损失比较 $\hat y$ 与 $y$,梯度再回到 $\theta$。不要把 $L$ 画成前向的一站。
|
||||
\end{importantbox}
|
||||
|
||||
\subsection{本章小结}
|
||||
前向只负责预测;缩放是改写,不是新算子。
|
||||
@@ -0,0 +1,2 @@
|
||||
\section{总结与延伸}
|
||||
摘录只保留一条机制:一次前向加侧路损失。更长的论文用 ledger 把 claim 钉住,再按路由出图。
|
||||
@@ -0,0 +1,17 @@
|
||||
# Outline: A One-Step Predictor
|
||||
|
||||
## Lecture map
|
||||
|
||||
| file | lecture_title | paper_sections | ledger_ids |
|
||||
|---|---|---|---|
|
||||
| sec-01.tex | 这篇论文在问什么 | 1 | Q1 |
|
||||
| sec-02.tex | 主张与贡献 | 1 | C1 |
|
||||
| sec-03.tex | 预备:定义、假设、符号 | 1 | D1 |
|
||||
| sec-04.tex | 一次前向与损失 | 1 | C1, DER1 |
|
||||
| sec-05.tex | 总结与延伸 | 1 | C1 |
|
||||
| sec-app-a.tex | 符号表 | — | |
|
||||
| sec-app-b.tex | 推导链一览 | — | DER1 |
|
||||
|
||||
## Locked
|
||||
- 首节标题必须是「这篇论文在问什么」
|
||||
- 末节(appendix 前)必须是「总结与延伸」
|
||||
@@ -0,0 +1,8 @@
|
||||
# A One-Step Predictor (fixture)
|
||||
|
||||
We predict $\hat y = f_\theta(x)$ in one forward pass. The loss
|
||||
$L(\hat y, y)$ is computed after the prediction; it is not a station
|
||||
on the forward path.
|
||||
|
||||
The scale $1/\sqrt{d}$ is introduced so that the variance of the
|
||||
inner product does not grow with $d$.
|
||||
@@ -0,0 +1,47 @@
|
||||
schema: superpaper.ledger/v1
|
||||
retired_ids: []
|
||||
paper:
|
||||
id: pipeline-delegate
|
||||
title: A One-Step Predictor
|
||||
authors: ["Fixture"]
|
||||
notes_language: zh
|
||||
source: {kind: excerpt}
|
||||
coverage:
|
||||
mode: excerpt
|
||||
sections_in: ["1"]
|
||||
questions:
|
||||
- {id: Q1, text: "前向主路和损失如何分开?", source: "excerpt"}
|
||||
claims:
|
||||
- id: C1
|
||||
text: "一次前向是 x → f_θ → ŷ;损失不在主路上"
|
||||
kind: contribution
|
||||
status: core
|
||||
supports: [Q1]
|
||||
definitions: []
|
||||
assumptions: []
|
||||
lemmas: []
|
||||
symbols:
|
||||
- {name: x, latex: "x", meaning: "输入", kind: value}
|
||||
- {name: yhat, latex: "\\hat y", meaning: "预测", kind: value}
|
||||
- {name: L, latex: "L", meaning: "损失", kind: scalar}
|
||||
derivations:
|
||||
- id: DER1
|
||||
claim: C1
|
||||
title: "前向定义"
|
||||
expand: true
|
||||
figure: null
|
||||
steps:
|
||||
- {id: S1, from: "x", to: "f_\\theta(x)=\\hat y", rule: definition}
|
||||
figures:
|
||||
- id: F1
|
||||
claim: C1
|
||||
title: "一次前向与侧路损失"
|
||||
grammar: pipeline
|
||||
toolkit: superfig
|
||||
signals: [pipeline, what-eats-what]
|
||||
request: figures/F1/F1.request.md
|
||||
include: figures/F1/build/F1.pdf
|
||||
status: included
|
||||
evidence: []
|
||||
terms: []
|
||||
source_assets: []
|
||||
@@ -0,0 +1 @@
|
||||
一次前向 $x\to f_\theta\to\hat y$。损失挂在预测下方,不是主路车站。
|
||||
@@ -0,0 +1,27 @@
|
||||
# Figure request F1
|
||||
toolkit: superfig
|
||||
language: cjk
|
||||
claim: 一次前向是 x → f_θ → ŷ;损失不在主路上。
|
||||
grammar: pipeline
|
||||
work_rel_dir: figures/F1
|
||||
|
||||
## Roles
|
||||
- {role: input, color: sfTeal}
|
||||
- {role: model, color: sfOrange}
|
||||
- {role: loss, color: sfCoral}
|
||||
- {role: output, color: sfViolet}
|
||||
|
||||
## Flow (cursor). \sfconn 没有 endpoints。
|
||||
stage: {name: SA, text: "推理流程:一次前向"}
|
||||
row: {name: R1, height: 16mm}
|
||||
in_row:
|
||||
- {macro: sfnode, name: x, role: input, label: "输入 $x$", w: 16mm, h: 12mm}
|
||||
- {macro: sfconn, name: e1, label: 预处理}
|
||||
- {macro: sfnode, name: f, role: model, label: "模型 $f_\\theta$", w: 18mm, h: 12mm}
|
||||
- {macro: sfconn, name: e2, label: logits}
|
||||
- {macro: sfnode, name: y, role: output, label: "预测 $\\hat y$", w: 16mm, h: 12mm}
|
||||
|
||||
## Fixed topology
|
||||
- {macro: sfnode, name: s, role: loss, label: "损失 $L$", w: 14mm, h: 12mm,
|
||||
at: "($(y.south)+(0,-22mm)$)"}
|
||||
- {macro: sfarrowlabel, from: "y.south", to: "s.north", label: "$L(\\hat y, y)$"}
|
||||
@@ -0,0 +1,47 @@
|
||||
% superfig golden example 1 -- one horizontal paper-figure pipeline.
|
||||
% Main path is input -> model -> prediction. Loss is a side object, not a
|
||||
% station on the forward path.
|
||||
% ../scripts/build.sh pipeline.tex
|
||||
\documentclass[border=10pt]{standalone}
|
||||
\usepackage[cjk]{superfig}
|
||||
|
||||
\sfsetrole{input}{sfTeal}
|
||||
\sfsetrole{model}{sfOrange}
|
||||
\sfsetrole{loss}{sfCoral}
|
||||
\sfsetrole{output}{sfViolet}
|
||||
|
||||
\begin{document}
|
||||
\begin{tikzpicture}
|
||||
|
||||
\sfstage{SA}{推理流程:一次前向}
|
||||
\sfrow{R1}{16mm}
|
||||
\sfnode[role=input]{x}{输入 $x$}{16mm}{12mm}
|
||||
\sfconn{e1}{预处理}
|
||||
\sfnode[role=model]{f}{模型 $f_\theta$}{18mm}{12mm}
|
||||
\sfconn{e2}{logits}
|
||||
\sfnode[role=output]{y}{预测 $\hat y$}{16mm}{12mm}
|
||||
\sfrowend
|
||||
|
||||
% Loss compares the prediction with the target; it is not on the main path.
|
||||
% Hang it below ŷ with enough shaft that no caption sits on the arrow.
|
||||
\sfnode[role=loss, at={($(y.south)+(0,-22mm)$)}]{s}{损失 $L$}{14mm}{12mm}
|
||||
\sfarrowlabel{y.south}{s.north}{$L(\hat y,y)$}
|
||||
|
||||
\sflane{R1}
|
||||
\sfcaption{x}{$x$}{原始输入}
|
||||
\sfcaption{f}{$f_\theta$}{可学习参数}
|
||||
\sfnolane
|
||||
\sfcaption{s}{$L$}{与真值比较}
|
||||
|
||||
\sfbbox{all}
|
||||
\sftopformula{F}{%
|
||||
$x \;\xrightarrow{\;f_\theta\;}\; \hat y,\qquad
|
||||
\min_\theta\; L\bigl(f_\theta(x),\,y\bigr)$}
|
||||
\sfmeaningbox{mb}{96mm}{all}
|
||||
{一次从输入到预测的前向;损失在预测之后单独计算}
|
||||
{数据、模型参数、预测、损失}
|
||||
{预处理后送入模型;模型产生 logits 得到预测;损失比较预测与真值,梯度再回到参数}
|
||||
\sfsignature{推理流程示意}{mb}
|
||||
|
||||
\end{tikzpicture}
|
||||
\end{document}
|
||||
|
After Width: | Height: | Size: 169 KiB |
|
After Width: | Height: | Size: 30 KiB |
|
After Width: | Height: | Size: 153 KiB |
|
After Width: | Height: | Size: 313 KiB |
@@ -0,0 +1,14 @@
|
||||
\documentclass[a4paper]{article}
|
||||
\input{notes-macros}
|
||||
\renewcommand{\notetitle}{一次前向与侧路损失}
|
||||
\renewcommand{\notepaper}{A One-Step Predictor}
|
||||
\begin{document}
|
||||
\tableofcontents
|
||||
\newpage
|
||||
\input{sections/sec-01.tex}
|
||||
\input{sections/sec-02.tex}
|
||||
\input{sections/sec-03.tex}
|
||||
\appendix
|
||||
\section{符号表}
|
||||
\input{sections/symbols.tex}
|
||||
\end{document}
|
||||
@@ -0,0 +1,2 @@
|
||||
\section{这篇论文在问什么}
|
||||
预测怎么从输入算出来,损失该不该站在前向主路上?
|
||||
@@ -0,0 +1,8 @@
|
||||
\section{主张与贡献}
|
||||
\splabel{C1}
|
||||
一次前向是 $x\to f_\theta\to\hat y$。损失在预测之后单独比较。
|
||||
|
||||
\spfig{F1}{一次前向与侧路损失。}{重绘自 fixture Figure~1;toolkit: \texttt{superfig};ledger id: F1。}
|
||||
|
||||
\subsection{本章小结}
|
||||
损失不是前向的一站。
|
||||
@@ -0,0 +1,2 @@
|
||||
\section{总结与延伸}
|
||||
这张图走 superfig,因为要画的是谁吃谁,不是轴长。
|
||||
@@ -0,0 +1,14 @@
|
||||
# Outline: A One-Step Predictor
|
||||
|
||||
## Lecture map
|
||||
|
||||
| file | lecture_title | paper_sections | ledger_ids |
|
||||
|---|---|---|---|
|
||||
| sec-01.tex | 这篇论文在问什么 | 1 | Q1 |
|
||||
| sec-02.tex | 主张与贡献 | 1 | C1, F1 |
|
||||
| sec-03.tex | 总结与延伸 | 1 | C1 |
|
||||
| sec-app-a.tex | 符号表 | — | |
|
||||
|
||||
## Locked
|
||||
- 首节标题必须是「这篇论文在问什么」
|
||||
- 末节(appendix 前)必须是「总结与延伸」
|
||||
@@ -0,0 +1,47 @@
|
||||
schema: superpaper.ledger/v1
|
||||
retired_ids: []
|
||||
paper:
|
||||
id: tensor-delegate
|
||||
title: Head scores
|
||||
authors: ["Fixture"]
|
||||
notes_language: zh
|
||||
source: {kind: excerpt}
|
||||
coverage:
|
||||
mode: excerpt
|
||||
questions:
|
||||
- {id: Q1, text: "每头打分沿哪一维收缩?", source: "excerpt"}
|
||||
claims:
|
||||
- id: C1
|
||||
text: "每头打分沿 d_h 收缩;K^T 必须物理换面"
|
||||
kind: theoretical
|
||||
status: core
|
||||
supports: [Q1]
|
||||
definitions: []
|
||||
assumptions: []
|
||||
lemmas: []
|
||||
symbols:
|
||||
- {name: Q, latex: "Q", meaning: "query", kind: value}
|
||||
- {name: K, latex: "K", meaning: "key", kind: value}
|
||||
- {name: S, latex: "S", meaning: "score", kind: score}
|
||||
- {name: dh, latex: "d_h", meaning: "头维", kind: "shape parameter"}
|
||||
derivations:
|
||||
- id: DER1
|
||||
claim: C1
|
||||
title: "打分"
|
||||
expand: true
|
||||
figure: null
|
||||
steps:
|
||||
- {id: S1, from: "Q K", to: "Q K^{\\top}", rule: rearrange}
|
||||
figures:
|
||||
- id: F2
|
||||
claim: C1
|
||||
title: "每头打分"
|
||||
grammar: tensor-face
|
||||
toolkit: supertensor
|
||||
signals: [axis, shape, transpose, contraction, face]
|
||||
request: figures/F2/F2.request.md
|
||||
include: figures/F2/build/F2.pdf
|
||||
status: included
|
||||
evidence: []
|
||||
terms: []
|
||||
source_assets: []
|
||||
@@ -0,0 +1 @@
|
||||
$S^{(i)}=Q^{(i)}K^{(i)\top}$。收缩维 $d_h$ 在两个操作数上同一边长;$K^\top$ 换面。
|
||||
@@ -0,0 +1,25 @@
|
||||
# Figure request F2
|
||||
toolkit: supertensor
|
||||
language: cjk
|
||||
claim: 每头打分沿 d_h 收缩;K^T 必须物理换面。
|
||||
grammar: tensor-face
|
||||
work_rel_dir: figures/F2
|
||||
|
||||
## Roles
|
||||
- {role: q, color: stTeal}
|
||||
- {role: k, color: stOrange}
|
||||
- {role: s, color: stCoral}
|
||||
|
||||
## Geometry
|
||||
- {axis: T, cells: 6}
|
||||
- {axis: dh, cells: 3}
|
||||
|
||||
## Flow
|
||||
stage: {name: SA, text: "每头打分:沿 $d_h$ 收缩"}
|
||||
row: {name: rowA, height: T}
|
||||
in_row:
|
||||
- {macro: ststack, name: Q, role: q, coord: "", rows: T, cols: dh, sheets: 3, bracket: true}
|
||||
- {macro: stglyph, name: mA, coord: "", glyph: "$\\times$"}
|
||||
- {macro: ststack, name: KT, role: k, coord: "", rows: dh, cols: T, sheets: 3, bracket: true}
|
||||
- {macro: stglyph, name: eA, coord: "", glyph: "$=$"}
|
||||
- {macro: ststack, name: S, role: s, coord: "", rows: T, cols: T, sheets: 3}
|
||||
@@ -0,0 +1,31 @@
|
||||
\documentclass[border=10pt]{standalone}
|
||||
\usepackage[cjk]{supertensor}
|
||||
|
||||
\stsetrole{q}{stTeal}
|
||||
\stsetrole{k}{stOrange}
|
||||
\stsetrole{s}{stCoral}
|
||||
|
||||
\stdim{T}{6}
|
||||
\stdim{dh}{3}
|
||||
|
||||
\begin{document}
|
||||
\begin{tikzpicture}
|
||||
\ststage{SA}{每头打分:沿 $d_h$ 收缩}
|
||||
\strow{rowA}{T}
|
||||
\ststack[role=q, bracket=true]{Q}{}{T}{dh}{3}
|
||||
\stglyph{mA}{$\times$}
|
||||
\ststack[role=k, bracket=true]{KT}{}{dh}{T}{3}
|
||||
\stglyph{eA}{$=$}
|
||||
\ststack[role=s]{S}{}{T}{T}{3}
|
||||
\strowend
|
||||
\stcaption{Q}{$\mathbf Q^{(i)}$}{$h\times T\times d_h$}
|
||||
\stcaption{KT}{$\mathbf K^{(i)\top}$}{$h\times d_h\times T$}
|
||||
\stcaption{S}{$\mathbf S^{(i)}$}{$h\times T\times T$}
|
||||
\sttopformula{F}{$S^{(i)}=Q^{(i)}K^{(i)\top}$}
|
||||
\stbbox{all}
|
||||
\stmeaningbox{mb}{120mm}{all}
|
||||
{$T$ 时间;$d_h$ 头维;$h$ 头数画成 stack 深度}
|
||||
{$Q/K$ 是 value;$S$ 是 score,不是 mask}
|
||||
{沿 $d_h$ 收缩;$K^\top$ 换面,收缩边等长}
|
||||
\end{tikzpicture}
|
||||
\end{document}
|
||||
|
After Width: | Height: | Size: 110 KiB |
|
After Width: | Height: | Size: 17 KiB |
|
After Width: | Height: | Size: 102 KiB |
|
After Width: | Height: | Size: 213 KiB |
@@ -0,0 +1,14 @@
|
||||
\documentclass[a4paper]{article}
|
||||
\input{notes-macros}
|
||||
\renewcommand{\notetitle}{每头打分沿 $d_h$ 收缩}
|
||||
\renewcommand{\notepaper}{Head scores}
|
||||
\begin{document}
|
||||
\tableofcontents
|
||||
\newpage
|
||||
\input{sections/sec-01.tex}
|
||||
\input{sections/sec-02.tex}
|
||||
\input{sections/sec-03.tex}
|
||||
\appendix
|
||||
\section{符号表}
|
||||
\input{sections/symbols.tex}
|
||||
\end{document}
|
||||
@@ -0,0 +1,2 @@
|
||||
\section{这篇论文在问什么}
|
||||
每头的 $Q$ 和 $K$ 沿哪一条边收缩,$K^\top$ 要不要真的换面?
|
||||
@@ -0,0 +1,20 @@
|
||||
\section{主张与贡献}
|
||||
\splabel{C1}
|
||||
打分是 $(T\times d_h)(d_h\times T)\to(T\times T)$。$K^\top$ 必须物理换面,收缩边等长。
|
||||
|
||||
先用中文说完,再写式子:
|
||||
|
||||
\[
|
||||
S^{(i)}=Q^{(i)}K^{(i)\top}.
|
||||
\]
|
||||
|
||||
\begin{itemize}
|
||||
\item $Q^{(i)}$ — 第 $i$ 头 query,$T\times d_h$
|
||||
\item $K^{(i)\top}$ — 转置后的 key,$d_h\times T$
|
||||
\item $S^{(i)}$ — 分数,不是 mask
|
||||
\end{itemize}
|
||||
|
||||
\spfig{F2}{每头打分沿 $d_h$ 收缩。}{重绘自 fixture;toolkit: \texttt{supertensor};ledger id: F2。}
|
||||
|
||||
\subsection{本章小结}
|
||||
轴长是这张图的主张,所以走 supertensor,不走 superfig。
|
||||
@@ -0,0 +1,2 @@
|
||||
\section{总结与延伸}
|
||||
形状对齐的公式图只交给 supertensor。
|
||||
@@ -0,0 +1,14 @@
|
||||
# Outline: Head scores
|
||||
|
||||
## Lecture map
|
||||
|
||||
| file | lecture_title | paper_sections | ledger_ids |
|
||||
|---|---|---|---|
|
||||
| sec-01.tex | 这篇论文在问什么 | 1 | Q1 |
|
||||
| sec-02.tex | 主张与贡献 | 1 | C1, F2 |
|
||||
| sec-03.tex | 总结与延伸 | 1 | C1 |
|
||||
| sec-app-a.tex | 符号表 | — | |
|
||||
|
||||
## Locked
|
||||
- 首节标题必须是「这篇论文在问什么」
|
||||
- 末节(appendix 前)必须是「总结与延伸」
|
||||