Files
dela fc3c69bc23 fix: harden ingest.sh against partial downloads and tar size limits
arxiv pdf: download to temp file then move, so a failed download no
longer poisons re-entry with a corrupt partial file. e-print tar:
--max-size is not a GNU tar option and silently did nothing; extract
first then audit with du, failing if >50M.
2026-08-21 14:57:27 +08:00

121 lines
3.6 KiB
Bash
Executable File

#!/usr/bin/env bash
# Ingest a paper source into a --work tree. Never compiles e-print TeX.
set -euo pipefail
WORK=""
KIND=""
SRC=""
while [[ $# -gt 0 ]]; do
case "$1" in
--work|--out) WORK="$2"; shift 2 ;;
--arxiv) KIND=arxiv; SRC="$2"; shift 2 ;;
--pdf) KIND=pdf; SRC="$2"; shift 2 ;;
--tex) KIND=tex; SRC="$2"; shift 2 ;;
--excerpt|--markdown) KIND=excerpt; SRC="$2"; shift 2 ;;
*) echo "usage: ingest.sh --work DIR (--arxiv ID|--pdf FILE|--tex FILE|--excerpt FILE)" >&2; exit 2 ;;
esac
done
[[ -n "$WORK" && -n "$KIND" ]] || { echo "need --work and a source" >&2; exit 2; }
ARXIV_RE='^(ar[Xx]iv:)?([0-9]{4}\.[0-9]{4,5}(v[0-9]+)?|[a-z-]+/[0-9]{7})$'
mkdir -p "$WORK/source/pages" "$WORK/notes/sections" "$WORK/out"
meta="$WORK/source/meta.yaml"
sha_of() { sha256sum "$1" | awk '{print $1}'; }
write_meta() {
local extra="$1"
cat > "$meta" <<EOF
kind: $KIND
$extra
EOF
}
case "$KIND" in
excerpt)
cp -- "$SRC" "$WORK/source/excerpt.md"
write_meta "excerpt: source/excerpt.md"
;;
tex)
mkdir -p "$WORK/source/tex"
if [[ -d "$SRC" ]]; then
cp -R -- "$SRC/." "$WORK/source/tex/"
else
cp -- "$SRC" "$WORK/source/tex/"
fi
write_meta "tex: source/tex"
;;
pdf)
cp -- "$SRC" "$WORK/source/paper.pdf"
pdftotext -layout "$WORK/source/paper.pdf" "$WORK/source/paper.txt" || true
pdftoppm -png -r 120 "$WORK/source/paper.pdf" "$WORK/source/pages/pg"
python3 - <<'PY' "$WORK/source/pages"
from pathlib import Path
import sys, re
d = Path(sys.argv[1])
files = sorted(d.glob("pg*.png"))
for i, p in enumerate(files, 1):
dest = d / f"pg-{i:03d}.png"
if p.resolve() != dest.resolve():
p.rename(dest)
PY
pages=$(find "$WORK/source/pages" -name 'pg-*.png' | wc -l)
write_meta "local_pdf: source/paper.pdf
sha256: $(sha_of "$WORK/source/paper.pdf")
pages: $pages"
;;
arxiv)
id="$SRC"
[[ "$id" =~ $ARXIV_RE ]] || { echo "bad arxiv id: $id" >&2; exit 2; }
id="${id#arxiv:}"; id="${id#arXiv:}"
mkdir -p "$WORK/source"
pdf="$WORK/source/paper.pdf"
if [[ ! -f "$pdf" ]]; then
pdf_tmp=$(mktemp --suffix=.pdf)
if curl -fsSL "https://arxiv.org/pdf/${id}.pdf" -o "$pdf_tmp" \
|| curl -fsSL "https://export.arxiv.org/pdf/${id}.pdf" -o "$pdf_tmp"; then
mv "$pdf_tmp" "$pdf"
else
rm -f "$pdf_tmp"
echo "error: failed to download arxiv pdf $id" >&2
exit 1
fi
fi
pdftotext -layout "$pdf" "$WORK/source/paper.txt" || true
pdftoppm -png -r 120 "$pdf" "$WORK/source/pages/pg"
python3 - <<'PY' "$WORK/source/pages"
from pathlib import Path
import sys
d = Path(sys.argv[1])
files = sorted(d.glob("pg*.png"))
for i, p in enumerate(files, 1):
dest = d / f"pg-{i:03d}.png"
if p.resolve() != dest.resolve():
p.rename(dest)
PY
# best-effort e-print; never compile
if [[ ! -d "$WORK/source/eprint" ]]; then
tmp=$(mktemp)
if curl -fsSL "https://arxiv.org/e-print/${id}" -o "$tmp"; then
mkdir -p "$WORK/source/eprint"
tar -xf "$tmp" -C "$WORK/source/eprint" 2>/dev/null || true
extracted_size=$(du -sb "$WORK/source/eprint" | awk '{print $1}')
if (( extracted_size > 52428800 )); then
rm -rf "$WORK/source/eprint"
echo "error: e-print exceeds 50M limit ($extracted_size bytes)" >&2
exit 1
fi
fi
rm -f "$tmp"
fi
pages=$(find "$WORK/source/pages" -name 'pg-*.png' | wc -l)
write_meta "arxiv: $id
local_pdf: source/paper.pdf
sha256: $(sha_of "$pdf")
pages: $pages"
;;
esac
echo "ingested $KIND -> $WORK"