""" Quick script to generate telecom dataset and place it where the backend expects it. Run: python generate_data.py """ import sys from pathlib import Path # Add backend to path sys.path.insert(0, str(Path(__file__).parent)) import pandas as pd import numpy as np from datetime import datetime, timedelta import random np.random.seed(42) random.seed(42) CITIES = [ ("Lagos", "Nigeria", 6.5244, 3.3792), ("Abuja", "Nigeria", 9.0765, 7.3986), ("Kano", "Nigeria", 12.0022, 8.5920), ("Port Harcourt", "Nigeria", 4.8156, 7.0498), ("Ibadan", "Nigeria", 7.3775, 3.9470), ("Benin City", "Nigeria", 6.3350, 5.6037), ("Kaduna", "Nigeria", 10.5222, 7.4383), ("Enugu", "Nigeria", 6.4402, 7.4943), ("Jos", "Nigeria", 9.8965, 8.8583), ("Ilorin", "Nigeria", 8.4966, 4.5421), ("New York", "USA", 40.7128, -74.0060), ("Los Angeles", "USA", 34.0522, -118.2437), ("Chicago", "USA", 41.8781, -87.6298), ("Houston", "USA", 29.7604, -95.3698), ("London", "UK", 51.5074, -0.1278), ("Paris", "France", 48.8566, 2.3522), ("Berlin", "Germany", 52.5200, 13.4050), ("Amsterdam", "Netherlands", 52.3676, 4.9041), ("Tokyo", "Japan", 35.6762, 139.6503), ("Seoul", "South Korea", 37.5665, 126.9780), ("Mumbai", "India", 19.0760, 72.8777), ("Dubai", "UAE", 25.2048, 55.2708), ("Singapore", "Singapore", 1.3521, 103.8198), ("Sydney", "Australia", -33.8688, 151.2093), ("Toronto", "Canada", 43.6532, -79.3832), ] NUM_CELLS = 50 NUM_RECORDS = 20000 print("Generating telecom dataset...") cells = [] # Assign cities with varying "health profiles" for realistic map colors CITY_PROFILES = { # Good zones (low SLA risk → green on map) "Lagos": "good", "Abuja": "good", "Toronto": "good", "Singapore": "good", "Sydney": "good", "Amsterdam": "good", # Medium zones (moderate SLA risk → yellow on map) "London": "medium", "Paris": "medium", "Tokyo": "medium", "Seoul": "medium", "Dubai": "medium", "Mumbai": "medium", "Berlin": "medium", "Houston": "medium", # Poor zones (high SLA risk → orange/red on map) "Kano": "poor", "Port Harcourt": "poor", "Ibadan": "poor", "Benin City": "poor", "Kaduna": "poor", "Enugu": "poor", "Jos": "poor", "Ilorin": "poor", # Critical zones "New York": "critical", "Los Angeles": "critical", "Chicago": "critical", } for i in range(NUM_CELLS): city_data = CITIES[i % len(CITIES)] cells.append({ "cell_id": f"CELL_{i+1:03d}", "city": city_data[0], "country": city_data[1], "lat": city_data[2] + np.random.uniform(-0.1, 0.1), "lon": city_data[3] + np.random.uniform(-0.1, 0.1), "profile": CITY_PROFILES.get(city_data[0], "medium"), }) end_time = datetime.now() start_time = end_time - timedelta(hours=72) records = [] for i in range(NUM_RECORDS): cell = cells[i % NUM_CELLS] ts = start_time + timedelta(minutes=i * (72*60/NUM_RECORDS)) hour = ts.hour is_busy = hour in range(17, 22) profile = cell.get("profile", "medium") # Adjust scenario probability based on city profile if profile == "good": r = random.random() * 0.55 # mostly normal (0-0.55 → normal) elif profile == "poor": r = 0.40 + random.random() * 0.60 # mostly issues elif profile == "critical": r = 0.55 + random.random() * 0.45 # mostly critical issues else: # medium r = random.random() if r < 0.40: scenario = "Normal" rsrp = np.random.uniform(-90, -70) sinr = np.random.uniform(15, 25) prb_dl = np.random.uniform(30, 55) prb_ul = np.random.uniform(25, 50) thr_dl = np.random.uniform(50, 150) thr_ul = np.random.uniform(10, 30) lat = np.random.uniform(10, 35) mos = np.random.uniform(3.8, 4.5) cdr = np.random.uniform(0, 0.8) ho = np.random.uniform(95, 99.5) bler = np.random.uniform(0.5, 2) cqi = np.random.uniform(12, 15) rlc = np.random.uniform(0.5, 2) sev = "NONE" sla_risk = np.random.uniform(0.0, 0.15) # LOW risk → green on map capex = np.random.uniform(5, 25) elif r < 0.55: scenario = "Capacity Problem" rsrp = np.random.uniform(-90, -75) sinr = np.random.uniform(10, 18) prb_dl = np.random.uniform(87, 100) prb_ul = np.random.uniform(85, 100) thr_dl = np.random.uniform(10, 30) thr_ul = np.random.uniform(2, 8) lat = np.random.uniform(40, 90) mos = np.random.uniform(2.5, 3.5) cdr = np.random.uniform(1.5, 4) ho = np.random.uniform(85, 92) bler = np.random.uniform(3, 8) cqi = np.random.uniform(6, 10) rlc = np.random.uniform(3, 8) sev = "CRITICAL" if is_busy else "HIGH" sla_risk = np.random.uniform(0.6, 0.9) capex = np.random.uniform(70, 95) elif r < 0.67: scenario = "High Latency" rsrp = np.random.uniform(-95, -75) sinr = np.random.uniform(10, 18) prb_dl = np.random.uniform(55, 80) prb_ul = np.random.uniform(50, 75) thr_dl = np.random.uniform(30, 60) thr_ul = np.random.uniform(5, 15) lat = np.random.uniform(80, 200) mos = np.random.uniform(2.8, 3.5) cdr = np.random.uniform(0.5, 2) ho = np.random.uniform(90, 96) bler = np.random.uniform(2, 5) cqi = np.random.uniform(8, 12) rlc = np.random.uniform(10, 25) sev = "MEDIUM" sla_risk = np.random.uniform(0.4, 0.7) capex = np.random.uniform(40, 65) elif r < 0.77: scenario = "Call Drop" rsrp = np.random.uniform(-115, -105) sinr = np.random.uniform(2, 8) prb_dl = np.random.uniform(40, 70) prb_ul = np.random.uniform(35, 65) thr_dl = np.random.uniform(5, 20) thr_ul = np.random.uniform(1, 5) lat = np.random.uniform(50, 120) mos = np.random.uniform(1.5, 2.5) cdr = np.random.uniform(5, 12) ho = np.random.uniform(70, 84) bler = np.random.uniform(5, 12) cqi = np.random.uniform(3, 7) rlc = np.random.uniform(5, 15) sev = "HIGH" sla_risk = np.random.uniform(0.7, 0.95) capex = np.random.uniform(60, 85) elif r < 0.87: scenario = "Handover Failure" rsrp = np.random.uniform(-98, -80) sinr = np.random.uniform(8, 15) prb_dl = np.random.uniform(40, 65) prb_ul = np.random.uniform(35, 60) thr_dl = np.random.uniform(25, 55) thr_ul = np.random.uniform(5, 15) lat = np.random.uniform(30, 60) mos = np.random.uniform(2.5, 3.5) cdr = np.random.uniform(2, 6) ho = np.random.uniform(65, 80) bler = np.random.uniform(2, 6) cqi = np.random.uniform(7, 11) rlc = np.random.uniform(3, 7) sev = "HIGH" sla_risk = np.random.uniform(0.5, 0.75) capex = np.random.uniform(45, 70) else: scenario = "Low MOS" rsrp = np.random.uniform(-100, -85) sinr = np.random.uniform(5, 12) prb_dl = np.random.uniform(60, 85) prb_ul = np.random.uniform(55, 80) thr_dl = np.random.uniform(15, 40) thr_ul = np.random.uniform(3, 10) lat = np.random.uniform(40, 80) mos = np.random.uniform(1.5, 2.8) cdr = np.random.uniform(1, 4) ho = np.random.uniform(82, 92) bler = np.random.uniform(4, 10) cqi = np.random.uniform(5, 9) rlc = np.random.uniform(4, 10) sev = "MEDIUM" sla_risk = np.random.uniform(0.45, 0.70) capex = np.random.uniform(35, 60) anomaly_flag = 1 if sev in ("CRITICAL", "HIGH") else 0 # Assign critical_count based on severity for map coloring if sev == "CRITICAL": critical_count = random.randint(3, 8) elif sev == "HIGH": critical_count = random.randint(1, 3) else: critical_count = 0 records.append({ "timestamp": ts.strftime("%Y-%m-%d %H:%M:%S"), "cell_id": cell["cell_id"], "city": cell["city"], "country": cell["country"], "lat": round(cell["lat"], 6), "lon": round(cell["lon"], 6), "RSRP": round(rsrp, 2), "SINR": round(sinr, 2), "PRB_UL": round(prb_ul, 2), "PRB_DL": round(prb_dl, 2), "throughput_DL": round(thr_dl, 2), "throughput_UL": round(thr_ul, 2), "latency_ms": round(lat, 2), "MOS": round(mos, 3), "call_drop_rate": round(cdr, 3), "handover_success_rate": round(ho, 2), "BLER_UL": round(bler, 2), "BLER_DL": round(bler * 0.9, 2), "CQI": round(cqi, 1), "RLC_retransmissions": round(rlc, 2), "hour": hour, "issue_type": scenario, "severity": sev, "anomaly_flag": anomaly_flag, "sla_breach_risk": round(sla_risk, 4), "capex_score": round(capex, 2), "critical_count": critical_count, "source_bo": "BO1" if anomaly_flag else ("BO2" if sla_risk > 0.4 else "BO3"), "risk_tier": sev if sev != "NONE" else "LOW", }) df = pd.DataFrame(records) # Replace any NaN/inf values with 0 df = df.fillna(0) df = df.replace([float('inf'), float('-inf')], 0) # Save to all expected locations import os paths = [ Path(__file__).parent / "data" / "processed" / "network_kpis_full.csv", Path(__file__).parent / "unified_telecom_dataset_FILLED.csv", Path(__file__).parent.parent / "data" / "unified_telecom_dataset_FILLED.csv", ] for p in paths: p.parent.mkdir(parents=True, exist_ok=True) df.to_csv(p, index=False) print(f"Saved {len(df)} records to: {p}") print(f"\nDataset summary:") print(f" Records: {len(df):,}") print(f" Cells: {df['cell_id'].nunique()}") print(f" Cities: {df['city'].nunique()}") print(f" Date range: {df['timestamp'].min()} → {df['timestamp'].max()}") print(f" Issues: {df['issue_type'].value_counts().to_dict()}") print(f" Severities: {df['severity'].value_counts().to_dict()}") print("\nDone!")