diff --git a/scripts/ingest.sh b/scripts/ingest.sh index f859ed2..b49d1fe 100755 --- a/scripts/ingest.sh +++ b/scripts/ingest.sh @@ -72,8 +72,15 @@ pages: $pages" mkdir -p "$WORK/source" pdf="$WORK/source/paper.pdf" if [[ ! -f "$pdf" ]]; then - curl -fsSL "https://arxiv.org/pdf/${id}.pdf" -o "$pdf" \ - || curl -fsSL "https://export.arxiv.org/pdf/${id}.pdf" -o "$pdf" + pdf_tmp=$(mktemp --suffix=.pdf) + if curl -fsSL "https://arxiv.org/pdf/${id}.pdf" -o "$pdf_tmp" \ + || curl -fsSL "https://export.arxiv.org/pdf/${id}.pdf" -o "$pdf_tmp"; then + mv "$pdf_tmp" "$pdf" + else + rm -f "$pdf_tmp" + echo "error: failed to download arxiv pdf $id" >&2 + exit 1 + fi fi pdftotext -layout "$pdf" "$WORK/source/paper.txt" || true pdftoppm -png -r 120 "$pdf" "$WORK/source/pages/pg" @@ -92,8 +99,13 @@ PY tmp=$(mktemp) if curl -fsSL "https://arxiv.org/e-print/${id}" -o "$tmp"; then mkdir -p "$WORK/source/eprint" - tar -xf "$tmp" -C "$WORK/source/eprint" --max-size=50M 2>/dev/null \ - || tar -xf "$tmp" -C "$WORK/source/eprint" || true + tar -xf "$tmp" -C "$WORK/source/eprint" 2>/dev/null || true + extracted_size=$(du -sb "$WORK/source/eprint" | awk '{print $1}') + if (( extracted_size > 52428800 )); then + rm -rf "$WORK/source/eprint" + echo "error: e-print exceeds 50M limit ($extracted_size bytes)" >&2 + exit 1 + fi fi rm -f "$tmp" fi diff --git a/superderive b/superderive index 45db61f..f2ccd6f 160000 --- a/superderive +++ b/superderive @@ -1 +1 @@ -Subproject commit 45db61f3e2de53bf0c0c990fe709b3a51aa4cdc5 +Subproject commit f2ccd6fb088f545572f513c158666e4b2a0ce4c6 diff --git a/supertensor b/supertensor index 8ada8e5..7210f89 160000 --- a/supertensor +++ b/supertensor @@ -1 +1 @@ -Subproject commit 8ada8e57c5f5989d54667842338f12487f0ab64a +Subproject commit 7210f890caefffc6e08e5d53d177da96e3c6f858