From 5170111823b3c4f718137f9f9350b719e8c279ce Mon Sep 17 00:00:00 2001 From: dela Date: Thu, 17 Sep 2026 10:06:54 +0800 Subject: [PATCH] feat: gate writer sections on a % teach: block (SP025) Rewrite references/pedagogy.md around recite-vs-teach: reader model, coverage density (sections_in is cite permission, not a to-do list), intuition-before-formula three-beat, and paper-jump filling. Every notes/sections/sec-*.tex must now answer gap / takeaway / jump / omit above the first \section. lint.py checks presence only (it cannot judge honesty); files opening with "% generated by" are exempt. - scripts/lint.py: SP025 + is_writer_section / teach_gaps helpers - tests/no-teach-block/: fixture with takeaway only, wired into test.sh - examples/*: all 14 writer sections get real teach blocks - SKILL.md, references/{agents,antipatterns,checklist}.md, DESIGN.md, assets/notes-template.tex: route writers and consistency agent through pedagogy.md --- DESIGN.md | 2 + SKILL.md | 6 +- assets/notes-template.tex | 4 +- .../derive-delegate/notes/sections/sec-01.tex | 5 + .../derive-delegate/notes/sections/sec-02.tex | 5 + .../derive-delegate/notes/sections/sec-03.tex | 5 + .../excerpt-toy/notes/sections/sec-01.tex | 5 + .../excerpt-toy/notes/sections/sec-02.tex | 5 + .../excerpt-toy/notes/sections/sec-03.tex | 5 + .../excerpt-toy/notes/sections/sec-04.tex | 5 + .../excerpt-toy/notes/sections/sec-05.tex | 5 + .../notes/sections/sec-01.tex | 5 + .../notes/sections/sec-02.tex | 5 + .../notes/sections/sec-03.tex | 5 + .../tensor-delegate/notes/sections/sec-01.tex | 5 + .../tensor-delegate/notes/sections/sec-02.tex | 5 + .../tensor-delegate/notes/sections/sec-03.tex | 5 + references/agents.md | 6 +- references/antipatterns.md | 1 + references/checklist.md | 2 + references/pedagogy.md | 242 +++++++++++++++++- scripts/lint.py | 26 ++ scripts/test.sh | 1 + tests/no-teach-block/ledger.yaml | 20 ++ tests/no-teach-block/notes/notes.tex | 5 + .../no-teach-block/notes/sections/sec-01.tex | 4 + 26 files changed, 373 insertions(+), 16 deletions(-) create mode 100644 tests/no-teach-block/ledger.yaml create mode 100644 tests/no-teach-block/notes/notes.tex create mode 100644 tests/no-teach-block/notes/sections/sec-01.tex diff --git a/DESIGN.md b/DESIGN.md index 0f32967..8dca2c9 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -663,6 +663,7 @@ source_assets: | `tests/ledger-invalid/bad-toolkit.yaml` | `SP001`(enum) | | `tests/ledger-invalid/duplicate-id.yaml` | `SP002` | | `tests/notes-loads-sty.tex` + 最小 ledger | `SP003` | +| `tests/no-teach-block/` | `SP025` | --- @@ -1333,6 +1334,7 @@ examples/excerpt-toy/ | `SP022` | 4 | v1(`SUPERPAPER_PHASE` 缺省 2,PR 7 已落地;仅当显式设为 1 时触发)出现 `toolkit: superderive` | | `SP023` | 4 | 非 dropped 的新 id ∈ `retired_ids` | | `SP024` | 4 | `figures[].claim` / `derivations[].claim` 不是已有 `C*` | +| `SP025` | 4 | `notes/sections/sec-*.tex` 首个 `\section` 之前缺 `% gap:` / `% takeaway:` / `% jump:` / `% omit:` 任一非空行(`% generated by` 开头的文件豁免) | stderr 格式**抄** `superfig/scripts/lint.py`:先 `!! {path}`,随后缩进行 ` line N: SP00X message`(无行号则 ` SP00X message`)。 diff --git a/SKILL.md b/SKILL.md index 9413b13..4f41286 100644 --- a/SKILL.md +++ b/SKILL.md @@ -47,7 +47,7 @@ or when the source is a lecture video. 1. `./scripts/preflight.sh`。0 = 全路径,1 = 降级(说出口),2 = 无 LaTeX。 2. `./scripts/ingest.sh --work …`(`references/input.md`)。 3. 从 `assets/ledger.example.yaml` 写 `ledger.yaml`,再写 `outline.md`,再写正文。 -4. 公式三拍:中文动机 → `\[` / `align` → 扁平符号表(`references/pedagogy.md`)。 +4. 写正文:教,不要复述(`references/pedagogy.md`)。公式三拍:直觉 → `\[` / `align` → 扁平符号表。 5. 按 `references/router.md` 填 figure plan。每张重绘图一份 `F*.request.md`,figure agent 只读它和对应子仓库 `SKILL.md`。 6. `./scripts/lint.py --work ` 然后 `./scripts/build.sh --work `。 7. 改术语或改图:先改 ledger,再改那一处。 @@ -59,9 +59,9 @@ Worked trees: `examples/excerpt-toy/`(`align`)、`examples/pipeline-delegate ``` $superpaper <源> 请 spawn 多 sub agents,隔离上下文: - 1 个 outline agent:写 ledger.yaml + outline.md + 节边界 - - N 个 writer agents:各写 sections/sec-XX.tex,引用 ledger id + - N 个 writer agents:先读 references/pedagogy.md,各写 sections/sec-XX.tex,引用 ledger id - 每个计划重绘图 1 个 figure agent:只读 F*.request.md,调用对应子仓库 skill - - 1 个 consistency agent:符号、术语、claim 覆盖、路由一致性 + - 1 个 consistency agent:符号、术语、claim 覆盖、路由一致性、是否复述(pedagogy.md) ``` 必须拆 agent 当且仅当:正文 > 12 页,或顶层节 > 4,或重绘图 ≥ 2,或用户显式要求 spawn。 diff --git a/assets/notes-template.tex b/assets/notes-template.tex index 1ba4928..8dadaf9 100644 --- a/assets/notes-template.tex +++ b/assets/notes-template.tex @@ -13,7 +13,9 @@ % \section{推导链一览} % \section{图表清单} % -% Formula: Chinese motive, then \[ / align, then a flat symbol list. +% Teach, don't recap. Each writer section starts with +% % teach: gap / takeaway / jump / omit (references/pedagogy.md). +% Formula: intuition off-stage, then \[ / align, then a flat symbol list. % Keep figures outside knowledgebox / importantbox / warningbox / quotebox. % Do not \usepackage{superfig|supertensor|superderive}. diff --git a/examples/derive-delegate/notes/sections/sec-01.tex b/examples/derive-delegate/notes/sections/sec-01.tex index 1345846..886e3c7 100644 --- a/examples/derive-delegate/notes/sections/sec-01.tex +++ b/examples/derive-delegate/notes/sections/sec-01.tex @@ -1,2 +1,7 @@ +% teach: +% gap: 见过 $1/\sqrt{d_k}$,但不知道它是装饰性常数还是一次必须看见的改写 +% takeaway: 这一节要判断的是「这步改写值不值得画出来」 +% jump: none +% omit: 摘录之外的全文结构 \section{这篇论文在问什么} 注意力里的 $1/\sqrt{d_k}$ 是装饰,还是一次必须看见的改写? diff --git a/examples/derive-delegate/notes/sections/sec-02.tex b/examples/derive-delegate/notes/sections/sec-02.tex index 0b623b7..089bb63 100644 --- a/examples/derive-delegate/notes/sections/sec-02.tex +++ b/examples/derive-delegate/notes/sections/sec-02.tex @@ -1,3 +1,8 @@ +% teach: +% gap: 知道要除 $\sqrt{d_k}$,但看不见方差是从哪一步冒出来的 +% takeaway: 缩放来自点积方差随 $d_k$ 增长,是改写不是新算子 +% jump: fixture 直接给出结论,没展开方差那一步 +% omit: 其他归一化方案的对比 \section{主张与贡献} \splabel{C1} 点积方差随 $d_k$ 涨。除掉 $\sqrt{d_k}$ 是改写,不是新算子。 diff --git a/examples/derive-delegate/notes/sections/sec-03.tex b/examples/derive-delegate/notes/sections/sec-03.tex index a1f226b..cc31f94 100644 --- a/examples/derive-delegate/notes/sections/sec-03.tex +++ b/examples/derive-delegate/notes/sections/sec-03.tex @@ -1,2 +1,7 @@ +% teach: +% gap: 不知道什么时候该出 superderive 图、什么时候 align 就够 +% takeaway: $\le 4$ 步且没有消项或代入要看见时,留在 align +% jump: none +% omit: superderive 的完整语法(见子仓库 SKILL.md) \section{总结与延伸} 四步以内、没有消项或代入要画出来时,仍然用 \texttt{align}。 diff --git a/examples/excerpt-toy/notes/sections/sec-01.tex b/examples/excerpt-toy/notes/sections/sec-01.tex index 06abff0..61dd9ef 100644 --- a/examples/excerpt-toy/notes/sections/sec-01.tex +++ b/examples/excerpt-toy/notes/sections/sec-01.tex @@ -1,3 +1,8 @@ +% teach: +% gap: 知道模型要输出预测,但没想过损失算不算前向路径上的一站 +% takeaway: 这份摘录只问一件事——前向到哪里为止 +% jump: none +% omit: 摘录之外的全文结构与实验 \section{这篇论文在问什么} 要把输入变成预测,最简单的机制是什么?损失要不要走在前向主路上? 这是摘录 fixture,不假装读完全文。 diff --git a/examples/excerpt-toy/notes/sections/sec-02.tex b/examples/excerpt-toy/notes/sections/sec-02.tex index 7170234..2a35734 100644 --- a/examples/excerpt-toy/notes/sections/sec-02.tex +++ b/examples/excerpt-toy/notes/sections/sec-02.tex @@ -1,3 +1,8 @@ +% teach: +% gap: 还没有一句能离开 PDF 复述的核心主张 +% takeaway: 一次前向是 $x\to f_\theta\to\hat y$;损失在预测之后单独比较 +% jump: none +% omit: 重复摘要的贡献条目 \section{主张与贡献} \splabel{C1} 一次前向是 $x \to f_\theta \to \hat y$。损失 $L(\hat y,y)$ 在预测之后单独计算,不是主路上的一站。 diff --git a/examples/excerpt-toy/notes/sections/sec-03.tex b/examples/excerpt-toy/notes/sections/sec-03.tex index af57f2d..9a73cdd 100644 --- a/examples/excerpt-toy/notes/sections/sec-03.tex +++ b/examples/excerpt-toy/notes/sections/sec-03.tex @@ -1,2 +1,7 @@ +% teach: +% gap: 不知道 $f_\theta$、$\hat y$ 在本讲义里各指什么 +% takeaway: 预测器就是 $\hat y=f_\theta(x)$,其余符号查附录,不在正文重列 +% jump: none +% omit: 论文式的记号巡游 \section{预备:定义、假设、符号} 预测器定义为 $\hat y = f_\theta(x)$。符号见附录。 diff --git a/examples/excerpt-toy/notes/sections/sec-04.tex b/examples/excerpt-toy/notes/sections/sec-04.tex index 21127ed..7be895b 100644 --- a/examples/excerpt-toy/notes/sections/sec-04.tex +++ b/examples/excerpt-toy/notes/sections/sec-04.tex @@ -1,3 +1,8 @@ +% teach: +% gap: 会算内积,但不知道维数一涨为什么会把后面的非线性弄坏 +% takeaway: 除 $\sqrt{d}$ 是把方差按回 1 的改写,不是新算子 +% jump: 摘录直接写下 $1/\sqrt{d}$,没说维数涨会让点积方差跟着涨 +% omit: 数据集、超参、硬件 \section{一次前向与损失} 内积的方差会随维数 $d$ 涨。为了不让后续非线性饱和,要把点积除掉 $\sqrt{d}$。 diff --git a/examples/excerpt-toy/notes/sections/sec-05.tex b/examples/excerpt-toy/notes/sections/sec-05.tex index 522cc36..9198f8f 100644 --- a/examples/excerpt-toy/notes/sections/sec-05.tex +++ b/examples/excerpt-toy/notes/sections/sec-05.tex @@ -1,2 +1,7 @@ +% teach: +% gap: 学完这一条机制,不知道同样的流程怎么套到一篇完整论文 +% takeaway: 先用 ledger 钉住 claim,再按路由决定出不出图 +% jump: none +% omit: 摘录没覆盖的实验与消融 \section{总结与延伸} 摘录只保留一条机制:一次前向加侧路损失。更长的论文用 ledger 把 claim 钉住,再按路由出图。 diff --git a/examples/pipeline-delegate/notes/sections/sec-01.tex b/examples/pipeline-delegate/notes/sections/sec-01.tex index e6be679..4f21d3a 100644 --- a/examples/pipeline-delegate/notes/sections/sec-01.tex +++ b/examples/pipeline-delegate/notes/sections/sec-01.tex @@ -1,2 +1,7 @@ +% teach: +% gap: 知道有前向和损失两件事,但不知道它们在图上谁接谁 +% takeaway: 这一节问的是数据流拓扑,不是张量轴长 +% jump: none +% omit: 摘录之外的全文结构 \section{这篇论文在问什么} 预测怎么从输入算出来,损失该不该站在前向主路上? diff --git a/examples/pipeline-delegate/notes/sections/sec-02.tex b/examples/pipeline-delegate/notes/sections/sec-02.tex index 825a213..7ab3ad5 100644 --- a/examples/pipeline-delegate/notes/sections/sec-02.tex +++ b/examples/pipeline-delegate/notes/sections/sec-02.tex @@ -1,3 +1,8 @@ +% teach: +% gap: 容易把损失画成前向链条上的下一个方块 +% takeaway: 损失是侧路,不是主路上的一站 +% jump: fixture 的框图没说清 $L$ 是挂在 $\hat y$ 之后还是在链条之内 +% omit: 重复摘要的贡献条目 \section{主张与贡献} \splabel{C1} 一次前向是 $x\to f_\theta\to\hat y$。损失在预测之后单独比较。 diff --git a/examples/pipeline-delegate/notes/sections/sec-03.tex b/examples/pipeline-delegate/notes/sections/sec-03.tex index c8fd1b8..ef9cbbb 100644 --- a/examples/pipeline-delegate/notes/sections/sec-03.tex +++ b/examples/pipeline-delegate/notes/sections/sec-03.tex @@ -1,2 +1,7 @@ +% teach: +% gap: 不知道这类图为什么交给 superfig 而不是 supertensor +% takeaway: 主张是「谁吃谁」时走 superfig +% jump: none +% omit: 轴长与形状细节(那是 supertensor 的活) \section{总结与延伸} 这张图走 superfig,因为要画的是谁吃谁,不是轴长。 diff --git a/examples/tensor-delegate/notes/sections/sec-01.tex b/examples/tensor-delegate/notes/sections/sec-01.tex index 1ded60b..719c514 100644 --- a/examples/tensor-delegate/notes/sections/sec-01.tex +++ b/examples/tensor-delegate/notes/sections/sec-01.tex @@ -1,2 +1,7 @@ +% teach: +% gap: 会写 $QK^\top$,但没想过收缩沿哪条边、$K^\top$ 要不要真的换面 +% takeaway: 这一节问的是形状对齐,不是模块拓扑 +% jump: none +% omit: 摘录之外的全文结构 \section{这篇论文在问什么} 每头的 $Q$ 和 $K$ 沿哪一条边收缩,$K^\top$ 要不要真的换面? diff --git a/examples/tensor-delegate/notes/sections/sec-02.tex b/examples/tensor-delegate/notes/sections/sec-02.tex index b316723..cb5af2d 100644 --- a/examples/tensor-delegate/notes/sections/sec-02.tex +++ b/examples/tensor-delegate/notes/sections/sec-02.tex @@ -1,3 +1,8 @@ +% teach: +% gap: 把 $K^\top$ 当成记号上的装饰,不当成一次真实的换面 +% takeaway: 打分沿 $d_h$ 收缩,收缩边必须等长 +% jump: fixture 只写 $QK^\top$,没说明转置是物理换面而非书写约定 +% omit: 多头拼接与输出投影 \section{主张与贡献} \splabel{C1} 打分是 $(T\times d_h)(d_h\times T)\to(T\times T)$。$K^\top$ 必须物理换面,收缩边等长。 diff --git a/examples/tensor-delegate/notes/sections/sec-03.tex b/examples/tensor-delegate/notes/sections/sec-03.tex index 4390f1d..ed3330c 100644 --- a/examples/tensor-delegate/notes/sections/sec-03.tex +++ b/examples/tensor-delegate/notes/sections/sec-03.tex @@ -1,2 +1,7 @@ +% teach: +% gap: 不知道形状类的图该交给哪个子仓库 +% takeaway: 主张是轴长与收缩边时走 supertensor +% jump: none +% omit: 模块拓扑(那是 superfig 的活) \section{总结与延伸} 形状对齐的公式图只交给 supertensor。 diff --git a/references/agents.md b/references/agents.md index a049465..ae54a01 100644 --- a/references/agents.md +++ b/references/agents.md @@ -4,11 +4,13 @@ Must split when pages > 12 **or** top sections > 4 **or** redraws ≥ 2 **or** t Writer jobs are lecture rows in `outline.md`, not `coverage.sections_in`. Outline writes `ledger.yaml`, `outline.md`, `notes.tex` inputs, and `F*.request.md`. Writers only touch their `sec-XX.tex`. Figure agents only see the request. +Outline, writers, and consistency read `references/pedagogy.md`. Consistency flags missing `% teach:` blocks and section-order recaps. + ``` $superpaper <源> 请 spawn 多 sub agents,隔离上下文: - 1 个 outline agent:ledger.yaml + outline.md - - N 个 writer agents:sections/sec-XX.tex + - N 个 writer agents:先读 references/pedagogy.md;sections/sec-XX.tex - 每个计划重绘图 1 个 figure agent:F*.request.md + sibling skill - - 1 个 consistency agent:符号、术语、claim 覆盖 + - 1 个 consistency agent:符号、术语、claim 覆盖、是否复述(pedagogy.md) 完成后用 scripts/lint.py 与 build.sh 收口。 ``` diff --git a/references/antipatterns.md b/references/antipatterns.md index 0fd758f..72fc277 100644 --- a/references/antipatterns.md +++ b/references/antipatterns.md @@ -1,5 +1,6 @@ # Antipatterns +- Recapping the paper's sections in lecture clothing (see `pedagogy.md`). - Drawing a rewrite as `\sfnode` boxes (use `align`). - Sending a tensor-shape claim to superfig, or an architecture to supertensor. - Loading a figure `.sty` in the notes so sibling warnings leak into the article log. diff --git a/references/checklist.md b/references/checklist.md index 0858f23..459d4f3 100644 --- a/references/checklist.md +++ b/references/checklist.md @@ -1,6 +1,8 @@ # Delivery checklist - Ledger written first; every core claim has `\splabel`. +- Notes teach from a reader gap, not the paper's section order (`references/pedagogy.md`). +- Every `sec-*.tex` opens with the four `% teach:` lines (`SP025`). - Formula three-beat present wherever display math appears. - Each figure toolkit matches `scripts/router.py`; mixed-class rows were split. - Notes do not `\usepackage{superfig|supertensor|superderive}`. diff --git a/references/pedagogy.md b/references/pedagogy.md index fd0d744..8d5e937 100644 --- a/references/pedagogy.md +++ b/references/pedagogy.md @@ -1,15 +1,237 @@ # Pedagogy -Reuse `youtube-render-pdf` teaching order: motive → idea → mechanism → evidence → takeaway. +Job: teach the argument thread. The paper is the source of claims, not +the outline of the notes. -Paper-side changes: +Section order is still motive → idea → mechanism → evidence → takeaway. +That order is not a license to recap the PDF in a new sequence. -- Cite `§` / `Eq.(n)` / Figure / Table / page, not timestamps. -- Front page is a bibliography card, not a PDF cover screenshot. -- Math is `\[` or `align`, never `$$`. -- Formula three-beat: Chinese motive, display math, flat symbol list. -- `quotebox` for a short quotation with a source; no long PDF paste. -- End major sections with `\subsection{本章小结}`; end the notes with `\section{总结与延伸}`. -- Figures stay outside boxes. +## Recite vs teach -Locked first/last titles: 「这篇论文在问什么」…「总结与延伸」, then appendix 符号表 / 推导链一览 / 图表清单. +The default failure is a **closed-book failure**: a colleague who has +not read this paper still cannot use the core claim, because the notes +assume everything the paper assumes. + +| | Recite | Teach | +|---|---|---| +| Starts from | the paper's next subsection | what the reader still lacks | +| Formula | "本节给出 Eq.(n)" then symbols | a world without the formula, then the formula | +| Gap | whatever the paper already wrote | the jump the paper did not write | +| Stops when | every cited paragraph is covered | the takeaway is usable | + +Reader model: a competent colleague in the ambient field, textbook +level. They have **not** read this paper, do not know its notation, and +will not reconstruct skipped algebra. Do not reteach the ambient +textbook (softmax, SGD, …) unless a core claim hangs on a twist. + +If deleting the PDF would make the section unreadable as a lecture, it +was a recap. + +## Before each section + +At the top of every writer `sec-XX.tex` (not `symbols.tex` / generated +appendix), answer these four lines **before** the `\section`: + +```tex +% teach: +% gap: <读者此刻还缺什么,不是论文小节标题> +% takeaway: <离开本节必须带走的一句> +% jump: <论文跳过、读者会卡住的那一步;没有则 none> +% omit: <论文这里有、本讲义故意不写的> +``` + +- `gap` — entering deficit, in the reader's words. +- `takeaway` — one sentence. This is also the `本章小结`. +- `jump` — the missing why / scale / naive alternative / extreme case. +- `omit` — ceremony you will not import "for completeness". + +Do not start the body until all four are filled. `lint.py` enforces the +four keys on every `notes/sections/sec-*.tex` as **`SP025`** (presence +only — it cannot judge whether the answers are honest). + +## Coverage density + +`coverage.sections_in` is **cite permission**, not a to-do list. +`ledger_ids` on the outline row are the teaching budget. + +- Outline: fewer lecture rows than paper subsections. Merge until each + row has **one** takeaway. A row per paper `\subsection` is too dense. +- Writer: teach this row's ids. A paper paragraph that is not required + for a `status: core` claim stays out. +- Supporting claims: in only if they unblock a core claim. +- Dropped claims: never. + +Default skip (paper ceremony): + +- "the rest of this paper is organized as follows" +- related-work tour (keep the 2–3 contrasts that **define the gap**) +- contribution bullets that restate the abstract +- dataset / hyperparameter / hardware laundry lists, unless a core + empirical claim hangs on a specific number +- every ablation, every appendix proof, every lemma not on an + `expand: true` derivation +- notation paragraphs that only duplicate the symbol appendix + +## Intuition before formulas + +Display math is a three-beat that must not be split: + +1. **Intuition in Chinese, with the formula still off-stage.** Use at + least one of: analogy, contrast with the naive move, extreme case + (0 / 1 / ∞), or tiny numbers. 「下面给出公式」/「本节引入 Eq.(n)」 + is not a motive. +2. `\[` or `align` / `aligned`. Never `$$`. +3. Flat symbol list: one item per symbol — 符号 — 含义 — 定义处. + +The section that owns the core mechanism then puts **one** +`importantbox{如果你只记一件事}` with the takeaway sentence. At most +one such box per major section. `本章小结` restates that sentence; it +is not a new inventory. + +Pick one intuition tool, not all four. A running numeric example is +allowed when scale or cancellation is the point; it is not mandatory. + +## Paper jumps + +Papers skip the why, the scale, and the obvious alternative. Before +mechanism, name the stall and fill it: + +- Why this form, not the naive one? +- What happens at 0 / 1 / ∞, or if the term is dropped? +- Which quantity is a count vs a rate vs a scale? +- Where does the gradient / information / mass actually flow? + +Fill in prose or `warningbox`. If the paper is silent and you would +have to invent a result, say it is silent — do not fabricate. + +## Boxes and citations + +| box | payload | +|---|---| +| `importantbox` | walk-away: core claim, mechanism, 「如果你只记一件事」 | +| `knowledgebox` | not the main thread: prerequisite, analogy, running example | +| `warningbox` | naive-vs-correct, hidden assumption, paper-is-silent | +| `quotebox` | short quotation + `§` / `Eq.(n)` in the title; ≲ 8 lines; no PDF paste | + +Routine exposition stays in prose. Figures stay outside every box. + +Cite `§` / `Eq.(n)` / Figure / Table / page, not timestamps. No +`[cite]` placeholders. `\spsource{...}` at the point of use. + +Front page is a bibliography card (title, authors, year, venue, arXiv), +not a PDF-cover screenshot. Default language is Chinese. + +End every major section with `\subsection{本章小结}`. End the notes +with `\section{总结与延伸}` (limitations + compressed takeaways + +open questions). + +Locked first/last titles: 「这篇论文在问什么」…「总结与延伸」, then +appendix 符号表 / 推导链一览 / 图表清单. Outline may add or drop +middle sections; it may not rename or drop the lock. + +Suggested skeleton (same lock as `assets/notes-template.tex`): + +```tex +\section{这篇论文在问什么} % questions[] + motive +\section{主张与贡献} % claims[status=core],\splabel{C*} +\section{预备:定义、假设、符号} % D* / A* / 符号摘要;不是论文 §2 巡游 +% --- outline-chosen mechanism sections --- +\section{实验与证据} % only E* that isolate a core claim +\section{总结与延伸} +``` + +## Antipatterns + +Each pair is the same lecture beat. Left recites; right teaches. + +### 1. Paper order, new wording + +```tex +% recite +本文第 3 节提出 scaled dot-product attention。 +作者将 $Q$、$K$ 做点积,除以 $\sqrt{d_k}$,再 softmax 乘 $V$。 +``` + +```tex +% teach +% gap: 还以为「对齐再加权」就完了,不知道维数一涨会出什么事 +读者已经会「用相似度加权」。先看极端情况:$d_k$ 很大时, +点积方差跟着涨,softmax 进饱和,梯度没了。 +除以 $\sqrt{d_k}$ 不是新算子,是把方差按回 1。 +``` + +### 2. Formula with no off-stage intuition + +```tex +% recite +本节引入如下公式: +\[ + \mathrm{Attention}(Q,K,V)=\mathrm{softmax}(QK^{\top}/\sqrt{d_k})V. +\] +\begin{itemize} + \item $Q$ — query + \item $K$ — key + \item $d_k$ — key 维 +\end{itemize} +``` + +```tex +% teach +两个随机向量的点积有多尖,随维数走。不先压方差,后面的 softmax 是死的。 +\[ + \mathrm{Attention}(Q,K,V)=\mathrm{softmax}(QK^{\top}/\sqrt{d_k})V. +\] +\begin{itemize} + \item $Q,K,V$ — query / key / value + \item $d_k$ — key 维(shape parameter;缩放跟它走) +\end{itemize} +\begin{importantbox}{如果你只记一件事} +$\sqrt{d_k}$ 是改写,不是新注意力。 +\end{importantbox} +``` + +### 3. Paper jump left as 「如式所示」 + +```tex +% recite +如 Eq.(3) 所示,取 $A=d/H_{\mathrm{act}}$。 +``` + +```tex +% teach +% jump: 论文直接写下 A,没说为什么不是 1/N +走得越宽,越要把输出缩小,否则残差被宽隐层撑爆。 +所以是 $A=d/H_{\mathrm{act}}$:宽 $4096$ 时 $A=1/4$,宽 $1024$ 时 $A=1$。 +不是 $1/N$——专家总数还没进这把尺子。 +``` + +### 4. Ceremony imported for completeness + +```tex +% recite +文献三条线:…(半页 related work) +数据集为 ImageNet / CIFAR / …,硬件为 8$\times$A100,超参见表 7。 +``` + +```tex +% teach +% omit: related-work 巡游、数据集表、硬件 +现成方法把问题当静态参数;缺的是带延迟和互馈的那一截。 +经验数字只在它能隔离这个机制时才进正文。 +``` + +### 5. One lecture row per paper paragraph + +```tex +% recite +\subsection{Lemma 2} +\subsection{Lemma 3} +\subsection{Ablation on dropout} +\subsection{Ablation on warmup} +``` + +```tex +% teach +% omit: 不在 DER1 上的 lemma;不能隔离 C1 的 ablation +只展开 DER1 用到的那一步。Ablation 只留「去掉这项,C1 是否还成立」。 +``` diff --git a/scripts/lint.py b/scripts/lint.py index c0784a9..a0fee91 100755 --- a/scripts/lint.py +++ b/scripts/lint.py @@ -43,6 +43,13 @@ KIND_ALIASES = { } STY_RE = re.compile(r"\\usepackage(?:\[[^\]]*\])?\{(superfig|supertensor|superderive)\}") LABEL_RE = re.compile(r"\\splabel\{(C[1-9][0-9]*)\}") +SECTION_RE = re.compile(r"^\s*\\section\b", re.MULTILINE) +GENERATED_RE = re.compile(r"^\s*%.*generated by", re.IGNORECASE) +TEACH_KEYS = ("gap", "takeaway", "jump", "omit") +TEACH_RE = { + key: re.compile(rf"^%\s*{key}\s*:\s*\S", re.MULTILINE) + for key in TEACH_KEYS +} def emit(path: Path, messages: list[str]) -> None: @@ -70,6 +77,19 @@ def collect_ids(data: dict) -> list[str]: return ids +def is_writer_section(path: Path) -> bool: + return path.parent.name == "sections" and path.name.startswith("sec-") + + +def teach_gaps(text: str) -> list[str]: + """Keys missing from the `% teach:` block above the first \\section.""" + if GENERATED_RE.match(text): + return [] + match = SECTION_RE.search(text) + head = text[: match.start()] if match else text + return [key for key in TEACH_KEYS if not TEACH_RE[key].search(head)] + + def notes_text(notes_paths: list[Path]) -> str: chunks: list[str] = [] for path in notes_paths: @@ -155,6 +175,12 @@ def lint( text = path.read_text(encoding="utf-8") if STY_RE.search(text): errors.append(f"SP003 {path.name} loads a figure package") + if is_writer_section(path): + missing = teach_gaps(text) + if missing: + errors.append( + f"SP025 {path.name} % teach: block missing {', '.join(missing)}" + ) labels = set(LABEL_RE.findall(body)) if notes_scan else set() diff --git a/scripts/test.sh b/scripts/test.sh index a5b1b03..530871b 100755 --- a/scripts/test.sh +++ b/scripts/test.sh @@ -76,6 +76,7 @@ expect_work "$ROOT/tests/loads-sty" SP003 expect_work "$ROOT/tests/missing-request" SP012 expect_work "$ROOT/tests/unlabeled-core" SP020 expect_work "$ROOT/tests/missing-include" SP021 +expect_work "$ROOT/tests/no-teach-block" SP025 echo "==> compile" for dir in \ diff --git a/tests/no-teach-block/ledger.yaml b/tests/no-teach-block/ledger.yaml new file mode 100644 index 0000000..1c25654 --- /dev/null +++ b/tests/no-teach-block/ledger.yaml @@ -0,0 +1,20 @@ +schema: superpaper.ledger/v1 +retired_ids: [] +paper: + id: toy + title: Toy + source: {kind: excerpt} +coverage: + mode: excerpt +questions: [] +claims: + - {id: C1, text: core claim, kind: contribution, status: core} +definitions: [] +assumptions: [] +lemmas: [] +symbols: [] +derivations: [] +figures: [] +evidence: [] +terms: [] +source_assets: [] diff --git a/tests/no-teach-block/notes/notes.tex b/tests/no-teach-block/notes/notes.tex new file mode 100644 index 0000000..2b526dc --- /dev/null +++ b/tests/no-teach-block/notes/notes.tex @@ -0,0 +1,5 @@ +\documentclass{article} +\usepackage{amsmath} +\begin{document} +\input{sections/sec-01} +\end{document} diff --git a/tests/no-teach-block/notes/sections/sec-01.tex b/tests/no-teach-block/notes/sections/sec-01.tex new file mode 100644 index 0000000..ad1d381 --- /dev/null +++ b/tests/no-teach-block/notes/sections/sec-01.tex @@ -0,0 +1,4 @@ +% takeaway: 有 takeaway,但缺 gap / jump / omit —— 闸门没走完 +\section{一次前向与损失} +\splabel{C1} +前向只负责预测;损失在预测之后单独比较。