# inference (GGUF, grammar-constrained, recommended for the 4GB laptop) llama-cpp-python>=0.2.79 # inference via the merged fp16 model (optional) transformers>=4.44 accelerate>=0.30 torch>=2.1 # utilities huggingface_hub>=0.23