Spaces:
Sleeping
Sleeping
File size: 1,554 Bytes
deceb2b | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 | #!/bin/bash
set -e
echo "=== PERMANENCE Training Space ==="
python3 -c "import torch; print(f'GPU: {torch.cuda.get_device_name(0)}'); print(f'VRAM: {torch.cuda.get_device_properties(0).total_mem / 1e9:.1f}GB')" 2>/dev/null || echo "WARNING: No GPU detected"
# Start server in background so HF health checks pass
echo ""
echo "Starting server (background)..."
python3 -m uvicorn server.app:app --host 0.0.0.0 --port 7860 &
SERVER_PID=$!
sleep 5
# Run the 4-stage training pipeline.
# The pipeline writes structured artifacts and status.json after every stage.
# It exits non-zero if any stage fails — entrypoint.sh continues so we can
# still upload partial artifacts for post-mortem.
echo ""
echo "Starting 4-stage training pipeline..."
echo " stage 1: SFT (~5 min)"
echo " stage 2: format-coverage gate (~1 min)"
echo " stage 3: GRPO (~4-5 hours)"
echo " stage 4: held-out eval (~15 min)"
echo ""
python3 -m training.pipeline --config training/config.yaml 2>&1 || echo "Pipeline reported failure — continuing for artifact upload"
# Generate curves from training_log.json
echo ""
echo "Generating curves..."
python3 tools/generate_curves.py 2>&1 || echo "Curve generation skipped"
# CRITICAL: auto-upload all artifacts to HF repos so they survive container eviction.
echo ""
echo "Auto-uploading artifacts to HF Hub..."
python3 -m training.auto_upload 2>&1 || echo "Auto-upload had errors (non-fatal)"
echo ""
echo "Pipeline complete. Server still running (PID $SERVER_PID)."
# Keep container alive for artifact retrieval
wait $SERVER_PID
|