mostafa922's picture
V5 retrain: 840 examples with 20 compliance fixes (Q24/Q37 Arabic greeting fix)
8297746 verified
Raw
History Blame
1.66 kB
# Hayat Elixir AI V5 - Meditron3-70B QLoRA Training
# ERROR-PROOF Dockerfile - addresses all 16 known HF Space errors
# Last updated: Feb 2026
# ERR-07 FIX: CUDA 12.1 base image (PyTorch 2.5.1 requires CUDA 12.1+)
FROM nvidia/cuda:12.1.0-devel-ubuntu22.04
# ERR-08 FIX: Set cache paths BEFORE any model downloads
# HF containers run as non-root (UID 1000) — cannot write to /.cache
ENV DEBIAN_FRONTEND=noninteractive
ENV HF_HOME=/app/hf_cache
ENV TRANSFORMERS_CACHE=/app/hf_cache
ENV TORCH_HOME=/app/torch_cache
ENV PYTHONUNBUFFERED=1
# System dependencies
RUN apt-get update && apt-get install -y \
python3 python3-pip git wget curl \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
# ERR-08 FIX: chmod BEFORE pip install to avoid permission issues
RUN chmod -R 777 /app
# ERR-07 FIX: PyTorch >= 2.4 (we use 2.5.1 with CUDA 12.1)
RUN pip3 install --no-cache-dir \
torch==2.5.1 --index-url https://download.pytorch.org/whl/cu121
# ERR-09 FIX: Pin ALL dependency versions exactly
# ERR-10 FIX: Include 'rich' explicitly (trl requires it but doesn't declare it)
RUN pip3 install --no-cache-dir \
transformers==4.46.0 \
peft==0.13.0 \
trl==0.9.6 \
bitsandbytes==0.44.1 \
accelerate==1.0.0 \
datasets==3.0.0 \
huggingface_hub==0.26.0 \
rich \
flask \
scipy \
sentencepiece \
protobuf
# Copy training script and data
COPY train.py /app/
COPY hayat_writer_v5_MASTER.jsonl /app/
# ERR-08 FIX: Ensure all directories writable after copy
RUN chmod -R 777 /app
# ERR-16 NOTE: Port 7860 for HF health check (may return HTML from proxy — use logs instead)
EXPOSE 7860
CMD ["python3", "train.py"]