FROM ghcr.io/avesed/vllm-ampere-optimized@sha256:6744eac95c9d0ee426c5c60ebaf72437da551e8a1786e8d2270e5164fb17dbfc # Inherit the base's Torch 2.13.0+cu130 without duplicating CUDA wheels. Shadow # calibration dependencies only in this venv; do not serve vLLM from it. RUN python3 -m venv --system-site-packages /opt/mellum-quantize COPY requirements.quantize.txt /opt/mellum/requirements.quantize.txt RUN /opt/mellum-quantize/bin/python -m pip install --no-cache-dir -r /opt/mellum/requirements.quantize.txt \ && /opt/mellum-quantize/bin/python -m pip freeze > /opt/mellum/quantizer-freeze.txt COPY quantize.py audit_export.py check_quantizer.py recipe.yaml /opt/mellum/ ENV PATH="/opt/mellum-quantize/bin:${PATH}" WORKDIR /opt/mellum ENTRYPOINT ["python", "/opt/mellum/quantize.py"]