Qwen3.5-ocr-jp-2b / Dockerfile
ebinan92's picture
Fix ZeroGPU compat: torch 2.6.0 -> 2.8.0, update causal_conv1d wheel
6a66612 verified
Raw
History Blame Contribute Delete
2.44 kB
# Local-test Dockerfile for spaces/Qwen3.5-ocr-jp-2b.
#
# Mirrors the HF Space (ZeroGPU) stack so we can validate the inference path
# locally without the host's CUDA-13/Python-3.12 incompatibilities:
# - Ubuntu 22.04 ships Python 3.10 (= Space's Python)
# - cu12.8 runtime libs (pre-built torch 2.8.0 cu12 wheels target this)
# - causal_conv1d wheel for cu12torch2.8 / cp310 / cxx11abi=TRUE
# (torch >=2.8 ships with cxx11abi=TRUE by default)
#
# Build:
# docker build -t qwen3.5-ocr-jp-2b-local spaces/Qwen3.5-ocr-jp-2b/
# Run:
# docker run --gpus all --rm -p 7860:7860 \
# -v $(pwd)/checkpoint/Qwen3.5-ocr-jp-2b:/model:ro \
# -v $(pwd)/.cache/hf:/root/.cache/huggingface \
# qwen3.5-ocr-jp-2b-local
# Open http://localhost:7860
FROM nvidia/cuda:12.8.1-devel-ubuntu22.04
# `devel` (not `runtime`) is required: triton JIT-compiles its CUDA driver
# bindings (`backends/nvidia/driver.c` → `cuda_utils.so`) at first invocation,
# which needs gcc + cuda.h. Without these the kernels silently fall back to CPU.
ENV DEBIAN_FRONTEND=noninteractive \
PYTHONUNBUFFERED=1 \
PIP_NO_CACHE_DIR=1 \
PIP_DISABLE_PIP_VERSION_CHECK=1
RUN apt-get update && apt-get install -y --no-install-recommends \
python3.10 \
python3-pip \
python3.10-dev \
gcc \
g++ \
ca-certificates \
libgl1 \
libglib2.0-0 \
&& rm -rf /var/lib/apt/lists/* \
&& ln -sf /usr/bin/python3.10 /usr/bin/python
WORKDIR /app
# Pin torch+vision from the cu12.8 index so the wheel ABI matches our cuda image.
RUN pip install --no-cache-dir \
--extra-index-url https://download.pytorch.org/whl/cu128 \
torch==2.8.0 torchvision==0.23.0
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
# Pre-install causal_conv1d for fla fast path (Python 3.10 / cu12 / torch 2.8 / cxx11abi=TRUE).
# app.py also tries to install at runtime; this pre-install just makes startup faster.
RUN pip install --no-cache-dir --no-deps \
https://github.com/Dao-AILab/causal-conv1d/releases/download/v1.6.2.post1/causal_conv1d-1.6.2.post1+cu12torch2.8cxx11abiTRUE-cp310-cp310-linux_x86_64.whl
COPY app.py .
# Defaults: load model from the bind-mounted /model, expose Gradio on all interfaces.
ENV MODEL_ID=/model \
GRADIO_SERVER_NAME=0.0.0.0 \
GRADIO_SERVER_PORT=7860 \
HF_HOME=/root/.cache/huggingface
EXPOSE 7860
CMD ["python", "app.py"]