"""UK Crime Tracker. A Streamlit dashboard for police-recorded crime across England and Wales, broken down by local area (Community Safety Partnership, e.g. Solihull), with rates calculated per 100,000 residents. Data is pulled live from the Home Office "Police recorded crime and outcomes open data tables" (Community Safety Partnership tables) and joined to ONS mid-year population estimates. """ import io import re import time import zipfile from collections import defaultdict from pathlib import Path import pandas as pd import plotly.graph_objects as go import requests import streamlit as st import xml.etree.ElementTree as ET GOVUK_CSP_PAGE = ( "https://www.gov.uk/government/statistical-data-sets/" "police-recorded-crime-and-outcomes-open-data-tables" ) ONS_POP_CSV = ( "https://download.ons.gov.uk/downloads/datasets/mid-year-pop-est/" "editions/mid-2021-april-2022-geography/versions/1.csv" ) CSP_ODS_RE = re.compile( r"https://assets\.publishing\.service\.gov\.uk/media/[0-9a-f]+/" r"prc-csp-mar(\d{4})-mar(\d{4})-tables-\d+\.ods" ) USER_AGENT = {"User-Agent": "uk-crime-tracker/1.0 (+huggingface space)"} CACHE_DIR = Path(__file__).resolve().parent / "cache" CACHE_DIR.mkdir(exist_ok=True) TBL = "{urn:oasis:names:tc:opendocument:xmlns:table:1.0}" TXT = "{urn:oasis:names:tc:opendocument:xmlns:text:1.0}" ROW_TAG = TBL + "table-row" CELL_TAG = TBL + "table-cell" TABLE_TAG = TBL + "table" NAME_ATTR = TBL + "name" REPEAT_COLS = TBL + "number-columns-repeated" P_TAG = TXT + "p" LAD_CODE_RE = re.compile(r"^(E0[6789]|W06)") def normalize(name): """Normalise an area name so CSP and ONS labels can be matched.""" value = str(name).lower().replace("&", " and ") value = re.sub(r"[^a-z0-9]+", " ", value) value = re.sub(r"\b(city of|royal borough of|county of|the)\b", " ", value) return re.sub(r"\s+", " ", value).strip() def _discover_csp_dataset(): """Find the most recent Community Safety Partnership ODS file.""" response = requests.get(GOVUK_CSP_PAGE, headers=USER_AGENT, timeout=60) response.raise_for_status() urls = {m.group(0): int(m.group(2)) for m in CSP_ODS_RE.finditer(response.text)} if not urls: raise RuntimeError("Could not find a Community Safety Partnership dataset on GOV.UK.") latest = max(urls, key=urls.get) end_year = urls[latest] start_year = end_year - 1 return latest, f"{start_year}/{str(end_year)[-2:]}", end_year discover_csp_dataset = st.cache_data(ttl=6 * 3600, show_spinner=False)(_discover_csp_dataset) def parse_csp_ods(ods_bytes): """Stream-parse the ODS workbook and aggregate counts by area/offence.""" counts = defaultdict(int) table_name = None with zipfile.ZipFile(io.BytesIO(ods_bytes)) as archive: with archive.open("content.xml") as stream: for event, element in ET.iterparse(stream, events=("start", "end")): if event == "start" and element.tag == TABLE_TAG: table_name = element.get(NAME_ATTR) elif event == "end" and element.tag == ROW_TAG: if table_name and re.match(r"^\d{4}_\d{2}$", table_name): values = [] for cell in element: if cell.tag != CELL_TAG: continue text = "".join(node.text or "" for node in cell.iter(P_TAG)) if text: values.append(text) elif int(cell.get(REPEAT_COLS) or 1) > 1: break if len(values) >= 9 and values[0] != "Financial Year": try: count = int(float(values[8])) except (TypeError, ValueError): count = 0 year = values[0] quarter = values[1] force = values[2] area = values[3] group = values[5] counts[(year, quarter, force, area, group)] += count element.clear() elif event == "end" and element.tag == TABLE_TAG: table_name = None element.clear() rows = [ { "period": period, "year_end": 2000 + int(period.split("/")[1]), "quarter": int(quarter) if str(quarter).isdigit() else 0, "force": force, "area": area, "offence_group": group, "count": count, } for (period, quarter, force, area, group), count in counts.items() ] return pd.DataFrame(rows) def load_crime_data(url, label, end_year): """Download and cache the area-level crime table.""" cache_file = CACHE_DIR / f"crime_csp_{end_year}.csv" stamp = CACHE_DIR / f"crime_csp_{end_year}.url" if cache_file.exists() and stamp.exists() and stamp.read_text().strip() == url: return pd.read_csv(cache_file) response = requests.get(url, headers=USER_AGENT, timeout=300) response.raise_for_status() frame = parse_csp_ods(response.content) frame.to_csv(cache_file, index=False) stamp.write_text(url) return frame def load_population(): """Download and cache ONS mid-year population estimates by local authority.""" cache_file = CACHE_DIR / "population.csv" if cache_file.exists() and time.time() - cache_file.stat().st_mtime < 120 * 86400: return pd.read_csv(cache_file) response = requests.get(ONS_POP_CSV, headers=USER_AGENT, timeout=120) response.raise_for_status() raw = pd.read_csv(io.BytesIO(response.content)) mask = (raw["Age"] == "Total") & (raw["Sex"] == "All") raw = raw.loc[mask] raw = raw[raw["administrative-geography"].astype(str).str.match(LAD_CODE_RE)] frame = ( raw[["administrative-geography", "Geography", "v4_0"]] .rename(columns={"administrative-geography": "area_code", "Geography": "ons_name", "v4_0": "population"}) .drop_duplicates("area_code") ) frame["area_key"] = frame["ons_name"].map(normalize) frame.to_csv(cache_file, index=False) return frame def warm_cache(): """Pre-compute both caches (used at image build time).""" url, label, end_year = _discover_csp_dataset() load_crime_data(url, label, end_year) load_population() return url, label def build_area_table(crime, population, group, year_end): """Area-level totals for one offence group and year, with rates.""" subset = crime[crime["year_end"] == year_end] if group != "All offences": subset = subset[subset["offence_group"] == group] totals = ( subset.groupby(["area", "force"], as_index=False)["count"].sum() ) totals["area_key"] = totals["area"].map(normalize) pop = population[["area_key", "population", "ons_name"]].drop_duplicates("area_key") totals = totals.merge(pop, on="area_key", how="left") totals["rate_per_100k"] = totals["count"] / totals["population"] * 100_000 return totals def format_int(value): if pd.isna(value): return "n/a" return f"{int(value):,}" def main(): st.set_page_config( page_title="UK Crime Tracker", page_icon="map", layout="wide", ) st.title("UK Crime Tracker") st.caption( "Police-recorded crime by local area across England and Wales, shown in " "proportion to population. Data: Home Office recorded crime open data; " "population: ONS mid-year estimates." ) try: with st.spinner("Discovering the latest Home Office release..."): url, label, end_year = discover_csp_dataset() with st.spinner("Loading crime data (first run can take ~1 minute)..."): crime = load_crime_data(url, label, end_year) with st.spinner("Loading ONS population estimates..."): population = load_population() except Exception as exc: # noqa: BLE001 st.error(f"Could not load data: {exc}") st.stop() crime = crime[~crime["area"].str.lower().str.startswith(("unassigned", "unallocated"))] groups = sorted(g for g in crime["offence_group"].dropna().unique()) years = sorted(crime["year_end"].unique()) latest_year = int(max(years)) total_area_names = crime["area"].nunique() st.markdown( f"**Data release:** year ending March {label} | " f"**Areas covered:** {total_area_names} | " f"**Offence groups:** {len(groups)}" ) st.divider() controls = st.columns([2, 1, 1]) with controls[0]: area = st.selectbox("Search for an area", sorted(crime["area"].unique())) with controls[1]: group = st.selectbox("Offence group", ["All offences"] + groups) with controls[2]: min_population = st.number_input( "Minimum population for rankings", min_value=0, value=50_000, step=10_000 ) latest = build_area_table(crime, population, group, latest_year) previous = build_area_table(crime, population, group, latest_year - 1) selected = latest[latest["area"] == area] selected_prev = previous[previous["area"] == area] ranked = latest.dropna(subset=["rate_per_100k"]) ranked = ranked[ranked["population"] >= min_population] ranked = ranked.sort_values("rate_per_100k", ascending=False).reset_index(drop=True) ranked["rank"] = ranked.index + 1 if not selected.empty: row = selected.iloc[0] prev_count = selected_prev["count"].sum() if not selected_prev.empty else None count = int(row["count"]) yoy = None if prev_count: yoy = (count - prev_count) / prev_count * 100 metric_cols = st.columns(4) metric_cols[0].metric(f"Recorded crimes ({label})", format_int(count)) if pd.isna(row["population"]): metric_cols[1].metric("Rate per 100,000", "no population match") else: metric_cols[1].metric("Rate per 100,000", f"{row['rate_per_100k']:,.0f}") metric_cols[2].metric("Population", format_int(row["population"])) if yoy is None: metric_cols[3].metric("Year-on-year", "n/a") else: metric_cols[3].metric("Year-on-year", f"{yoy:+.1f}%", f"{count - prev_count:+,} crimes") rank_row = ranked[ranked["area"] == area] national = latest.dropna(subset=["population"]) if not rank_row.empty: st.markdown( f"**{area}** ranks **{int(rank_row.iloc[0]['rank'])}** of " f"{len(ranked)} areas by {group.lower()} rate per 100,000 " f"(areas with population >= {int(min_population):,})." ) if not national.empty: national_rate = national["count"].sum() / national["population"].sum() * 100_000 st.caption( f"England and Wales benchmark for {group.lower()}: " f"{national_rate:,.0f} per 100,000." ) else: st.warning("No data found for the selected area.") st.divider() st.subheader("Top 10 areas by crime rate") top = ranked.head(10).iloc[::-1] if top.empty: st.info("No areas meet the population threshold.") else: figure = go.Figure( go.Bar( x=top["rate_per_100k"], y=top["area"], orientation="h", marker_color="#c8102e", customdata=top[["count", "population"]].to_numpy(), hovertemplate=( "%{y}
Rate: %{x:,.0f} per 100,000" "
Crimes: %{customdata[0]:,.0f}" "
Population: %{customdata[1]:,.0f}" ), ) ) figure.update_layout( xaxis_title="Crimes per 100,000 residents", yaxis_title="", height=460, margin=dict(l=10, r=10, t=10, b=10), ) st.plotly_chart(figure, width="stretch") area_history = crime[crime["area"] == area] if group != "All offences": area_history = area_history[area_history["offence_group"] == group] trend = ( area_history.groupby(["year_end", "quarter"], as_index=False)["count"].sum() ) trend = trend[trend["quarter"] > 0].sort_values(["year_end", "quarter"]) if not trend.empty: trend["label"] = ( trend["year_end"].sub(1).astype(str) + " Q" + trend["quarter"].astype(str) ) st.subheader(f"Quarterly trend: {area}") line = go.Figure( go.Scatter( x=trend["label"], y=trend["count"], mode="lines+markers", line=dict(color="#1d2b53", width=2.5), hovertemplate="%{x}
Crimes: %{y:,}", ) ) line.update_layout( xaxis_title="Quarter", yaxis_title="Recorded crimes", height=360, margin=dict(l=10, r=10, t=10, b=10), ) st.plotly_chart(line, width="stretch") breakdown = ( crime[(crime["area"] == area) & (crime["year_end"] == latest_year)] .groupby("offence_group", as_index=False)["count"] .sum() .sort_values("count", ascending=False) ) if not breakdown.empty: st.subheader(f"Offence mix for {area}, year ending March {label}") pie = go.Figure( go.Pie( labels=breakdown["offence_group"], values=breakdown["count"], hole=0.45, hovertemplate="%{label}
%{value:,} (%{percent})", ) ) pie.update_layout(height=420, margin=dict(l=10, r=10, t=10, b=10)) st.plotly_chart(pie, width="stretch") st.divider() st.subheader("All areas") table = latest[["area", "force", "count", "population", "rate_per_100k"]].copy() table = table.sort_values("rate_per_100k", ascending=False) table = table.rename( columns={ "area": "Area", "force": "Police force", "count": "Recorded crimes", "population": "Population", "rate_per_100k": "Rate per 100,000", } ) st.dataframe(table, width="stretch", hide_index=True) matched = table["Rate per 100,000"].notna().sum() st.caption( f"{matched} of {len(table)} areas matched to ONS population estimates; " "unmatched areas still show recorded crime counts. Home Office open data " "covers England and Wales." ) if __name__ == "__main__": main()