arxiv pdf: download to temp file then move, so a failed download no longer poisons re-entry with a corrupt partial file. e-print tar: --max-size is not a GNU tar option and silently did nothing; extract first then audit with du, failing if >50M.
121 lines
3.6 KiB
Bash
Executable File
121 lines
3.6 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Ingest a paper source into a --work tree. Never compiles e-print TeX.
|
|
set -euo pipefail
|
|
|
|
WORK=""
|
|
KIND=""
|
|
SRC=""
|
|
while [[ $# -gt 0 ]]; do
|
|
case "$1" in
|
|
--work|--out) WORK="$2"; shift 2 ;;
|
|
--arxiv) KIND=arxiv; SRC="$2"; shift 2 ;;
|
|
--pdf) KIND=pdf; SRC="$2"; shift 2 ;;
|
|
--tex) KIND=tex; SRC="$2"; shift 2 ;;
|
|
--excerpt|--markdown) KIND=excerpt; SRC="$2"; shift 2 ;;
|
|
*) echo "usage: ingest.sh --work DIR (--arxiv ID|--pdf FILE|--tex FILE|--excerpt FILE)" >&2; exit 2 ;;
|
|
esac
|
|
done
|
|
[[ -n "$WORK" && -n "$KIND" ]] || { echo "need --work and a source" >&2; exit 2; }
|
|
|
|
ARXIV_RE='^(ar[Xx]iv:)?([0-9]{4}\.[0-9]{4,5}(v[0-9]+)?|[a-z-]+/[0-9]{7})$'
|
|
|
|
mkdir -p "$WORK/source/pages" "$WORK/notes/sections" "$WORK/out"
|
|
meta="$WORK/source/meta.yaml"
|
|
|
|
sha_of() { sha256sum "$1" | awk '{print $1}'; }
|
|
|
|
write_meta() {
|
|
local extra="$1"
|
|
cat > "$meta" <<EOF
|
|
kind: $KIND
|
|
$extra
|
|
EOF
|
|
}
|
|
|
|
case "$KIND" in
|
|
excerpt)
|
|
cp -- "$SRC" "$WORK/source/excerpt.md"
|
|
write_meta "excerpt: source/excerpt.md"
|
|
;;
|
|
tex)
|
|
mkdir -p "$WORK/source/tex"
|
|
if [[ -d "$SRC" ]]; then
|
|
cp -R -- "$SRC/." "$WORK/source/tex/"
|
|
else
|
|
cp -- "$SRC" "$WORK/source/tex/"
|
|
fi
|
|
write_meta "tex: source/tex"
|
|
;;
|
|
pdf)
|
|
cp -- "$SRC" "$WORK/source/paper.pdf"
|
|
pdftotext -layout "$WORK/source/paper.pdf" "$WORK/source/paper.txt" || true
|
|
pdftoppm -png -r 120 "$WORK/source/paper.pdf" "$WORK/source/pages/pg"
|
|
python3 - <<'PY' "$WORK/source/pages"
|
|
from pathlib import Path
|
|
import sys, re
|
|
d = Path(sys.argv[1])
|
|
files = sorted(d.glob("pg*.png"))
|
|
for i, p in enumerate(files, 1):
|
|
dest = d / f"pg-{i:03d}.png"
|
|
if p.resolve() != dest.resolve():
|
|
p.rename(dest)
|
|
PY
|
|
pages=$(find "$WORK/source/pages" -name 'pg-*.png' | wc -l)
|
|
write_meta "local_pdf: source/paper.pdf
|
|
sha256: $(sha_of "$WORK/source/paper.pdf")
|
|
pages: $pages"
|
|
;;
|
|
arxiv)
|
|
id="$SRC"
|
|
[[ "$id" =~ $ARXIV_RE ]] || { echo "bad arxiv id: $id" >&2; exit 2; }
|
|
id="${id#arxiv:}"; id="${id#arXiv:}"
|
|
mkdir -p "$WORK/source"
|
|
pdf="$WORK/source/paper.pdf"
|
|
if [[ ! -f "$pdf" ]]; then
|
|
pdf_tmp=$(mktemp --suffix=.pdf)
|
|
if curl -fsSL "https://arxiv.org/pdf/${id}.pdf" -o "$pdf_tmp" \
|
|
|| curl -fsSL "https://export.arxiv.org/pdf/${id}.pdf" -o "$pdf_tmp"; then
|
|
mv "$pdf_tmp" "$pdf"
|
|
else
|
|
rm -f "$pdf_tmp"
|
|
echo "error: failed to download arxiv pdf $id" >&2
|
|
exit 1
|
|
fi
|
|
fi
|
|
pdftotext -layout "$pdf" "$WORK/source/paper.txt" || true
|
|
pdftoppm -png -r 120 "$pdf" "$WORK/source/pages/pg"
|
|
python3 - <<'PY' "$WORK/source/pages"
|
|
from pathlib import Path
|
|
import sys
|
|
d = Path(sys.argv[1])
|
|
files = sorted(d.glob("pg*.png"))
|
|
for i, p in enumerate(files, 1):
|
|
dest = d / f"pg-{i:03d}.png"
|
|
if p.resolve() != dest.resolve():
|
|
p.rename(dest)
|
|
PY
|
|
# best-effort e-print; never compile
|
|
if [[ ! -d "$WORK/source/eprint" ]]; then
|
|
tmp=$(mktemp)
|
|
if curl -fsSL "https://arxiv.org/e-print/${id}" -o "$tmp"; then
|
|
mkdir -p "$WORK/source/eprint"
|
|
tar -xf "$tmp" -C "$WORK/source/eprint" 2>/dev/null || true
|
|
extracted_size=$(du -sb "$WORK/source/eprint" | awk '{print $1}')
|
|
if (( extracted_size > 52428800 )); then
|
|
rm -rf "$WORK/source/eprint"
|
|
echo "error: e-print exceeds 50M limit ($extracted_size bytes)" >&2
|
|
exit 1
|
|
fi
|
|
fi
|
|
rm -f "$tmp"
|
|
fi
|
|
pages=$(find "$WORK/source/pages" -name 'pg-*.png' | wc -l)
|
|
write_meta "arxiv: $id
|
|
local_pdf: source/paper.pdf
|
|
sha256: $(sha_of "$pdf")
|
|
pages: $pages"
|
|
;;
|
|
esac
|
|
|
|
echo "ingested $KIND -> $WORK"
|