Standalone tree split from LLMRL/projects/kda. Includes Triton dt_bias backward fix, train_k3 --preset 0.5b, SFT, Docker runtime, and tests.
49 lines
1.5 KiB
Docker
49 lines
1.5 KiB
Docker
# syntax=docker/dockerfile:1
|
|
#
|
|
# Single GPU runtime for train + SwanLab client + eval.
|
|
# Build: docker build -t kda:<tag> .
|
|
#
|
|
# Does not bake data, checkpoints, or API keys. Mount them at run time.
|
|
# Host: NVIDIA driver >= 570, nvidia-container-toolkit. See README.
|
|
|
|
FROM pytorch/pytorch:2.9.0-cuda12.8-cudnn9-devel
|
|
|
|
ENV DEBIAN_FRONTEND=noninteractive \
|
|
PIP_NO_CACHE_DIR=1 \
|
|
PYTHONUNBUFFERED=1 \
|
|
PYTHONPATH=/workspace/kda \
|
|
HF_HOME=/cache/huggingface \
|
|
HUGGINGFACE_HUB_CACHE=/cache/huggingface \
|
|
HF_HUB_DISABLE_TELEMETRY=1
|
|
|
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
git \
|
|
ca-certificates \
|
|
&& rm -rf /var/lib/apt/lists/*
|
|
|
|
WORKDIR /workspace/kda
|
|
|
|
# Layer cache: install deps from pyproject before the rest of the tree.
|
|
COPY pyproject.toml ./
|
|
COPY kda ./kda
|
|
# Base image already has torch/cuda/triton; do not let pip re-resolve torch.
|
|
RUN pip install --no-cache-dir --no-deps -e . && \
|
|
pip install --no-cache-dir \
|
|
"einops>=0.7.0" \
|
|
"packaging>=23.0" \
|
|
"sentencepiece>=0.2.0" \
|
|
"datasets>=3.0.0" \
|
|
"transformers>=4.51.0" \
|
|
"swanlab>=0.6.0" \
|
|
"sacrebleu>=2.4.0" \
|
|
"langdetect>=1.0.9" \
|
|
"pytest>=7.0"
|
|
|
|
COPY . /workspace/kda
|
|
RUN pip install --no-cache-dir --no-deps -e . && \
|
|
mkdir -p /cache/huggingface /workspace/kda/ckpts /workspace/kda/swanlog \
|
|
/data/pretrain /data/eval /data/sft
|
|
|
|
# Require an explicit entry (train / eval / pytest / swanlab ping).
|
|
CMD ["python", "scripts/container_help.py"]
|