File size: 1,554 Bytes
deceb2b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
#!/bin/bash
set -e

echo "=== PERMANENCE Training Space ==="
python3 -c "import torch; print(f'GPU: {torch.cuda.get_device_name(0)}'); print(f'VRAM: {torch.cuda.get_device_properties(0).total_mem / 1e9:.1f}GB')" 2>/dev/null || echo "WARNING: No GPU detected"

# Start server in background so HF health checks pass
echo ""
echo "Starting server (background)..."
python3 -m uvicorn server.app:app --host 0.0.0.0 --port 7860 &
SERVER_PID=$!
sleep 5

# Run the 4-stage training pipeline.
# The pipeline writes structured artifacts and status.json after every stage.
# It exits non-zero if any stage fails — entrypoint.sh continues so we can
# still upload partial artifacts for post-mortem.
echo ""
echo "Starting 4-stage training pipeline..."
echo "  stage 1: SFT (~5 min)"
echo "  stage 2: format-coverage gate (~1 min)"
echo "  stage 3: GRPO (~4-5 hours)"
echo "  stage 4: held-out eval (~15 min)"
echo ""
python3 -m training.pipeline --config training/config.yaml 2>&1 || echo "Pipeline reported failure — continuing for artifact upload"

# Generate curves from training_log.json
echo ""
echo "Generating curves..."
python3 tools/generate_curves.py 2>&1 || echo "Curve generation skipped"

# CRITICAL: auto-upload all artifacts to HF repos so they survive container eviction.
echo ""
echo "Auto-uploading artifacts to HF Hub..."
python3 -m training.auto_upload 2>&1 || echo "Auto-upload had errors (non-fatal)"

echo ""
echo "Pipeline complete. Server still running (PID $SERVER_PID)."

# Keep container alive for artifact retrieval
wait $SERVER_PID