"""UK Crime Tracker.
A Streamlit dashboard for police-recorded crime across England and Wales,
broken down by local area (Community Safety Partnership, e.g. Solihull),
with rates calculated per 100,000 residents.
Data is pulled live from the Home Office "Police recorded crime and outcomes
open data tables" (Community Safety Partnership tables) and joined to ONS
mid-year population estimates.
"""
import io
import re
import time
import zipfile
from collections import defaultdict
from pathlib import Path
import pandas as pd
import plotly.graph_objects as go
import requests
import streamlit as st
import xml.etree.ElementTree as ET
GOVUK_CSP_PAGE = (
"https://www.gov.uk/government/statistical-data-sets/"
"police-recorded-crime-and-outcomes-open-data-tables"
)
ONS_POP_CSV = (
"https://download.ons.gov.uk/downloads/datasets/mid-year-pop-est/"
"editions/mid-2021-april-2022-geography/versions/1.csv"
)
CSP_ODS_RE = re.compile(
r"https://assets\.publishing\.service\.gov\.uk/media/[0-9a-f]+/"
r"prc-csp-mar(\d{4})-mar(\d{4})-tables-\d+\.ods"
)
USER_AGENT = {"User-Agent": "uk-crime-tracker/1.0 (+huggingface space)"}
CACHE_DIR = Path(__file__).resolve().parent / "cache"
CACHE_DIR.mkdir(exist_ok=True)
TBL = "{urn:oasis:names:tc:opendocument:xmlns:table:1.0}"
TXT = "{urn:oasis:names:tc:opendocument:xmlns:text:1.0}"
ROW_TAG = TBL + "table-row"
CELL_TAG = TBL + "table-cell"
TABLE_TAG = TBL + "table"
NAME_ATTR = TBL + "name"
REPEAT_COLS = TBL + "number-columns-repeated"
P_TAG = TXT + "p"
LAD_CODE_RE = re.compile(r"^(E0[6789]|W06)")
def normalize(name):
"""Normalise an area name so CSP and ONS labels can be matched."""
value = str(name).lower().replace("&", " and ")
value = re.sub(r"[^a-z0-9]+", " ", value)
value = re.sub(r"\b(city of|royal borough of|county of|the)\b", " ", value)
return re.sub(r"\s+", " ", value).strip()
def _discover_csp_dataset():
"""Find the most recent Community Safety Partnership ODS file."""
response = requests.get(GOVUK_CSP_PAGE, headers=USER_AGENT, timeout=60)
response.raise_for_status()
urls = {m.group(0): int(m.group(2)) for m in CSP_ODS_RE.finditer(response.text)}
if not urls:
raise RuntimeError("Could not find a Community Safety Partnership dataset on GOV.UK.")
latest = max(urls, key=urls.get)
end_year = urls[latest]
start_year = end_year - 1
return latest, f"{start_year}/{str(end_year)[-2:]}", end_year
discover_csp_dataset = st.cache_data(ttl=6 * 3600, show_spinner=False)(_discover_csp_dataset)
def parse_csp_ods(ods_bytes):
"""Stream-parse the ODS workbook and aggregate counts by area/offence."""
counts = defaultdict(int)
table_name = None
with zipfile.ZipFile(io.BytesIO(ods_bytes)) as archive:
with archive.open("content.xml") as stream:
for event, element in ET.iterparse(stream, events=("start", "end")):
if event == "start" and element.tag == TABLE_TAG:
table_name = element.get(NAME_ATTR)
elif event == "end" and element.tag == ROW_TAG:
if table_name and re.match(r"^\d{4}_\d{2}$", table_name):
values = []
for cell in element:
if cell.tag != CELL_TAG:
continue
text = "".join(node.text or "" for node in cell.iter(P_TAG))
if text:
values.append(text)
elif int(cell.get(REPEAT_COLS) or 1) > 1:
break
if len(values) >= 9 and values[0] != "Financial Year":
try:
count = int(float(values[8]))
except (TypeError, ValueError):
count = 0
year = values[0]
quarter = values[1]
force = values[2]
area = values[3]
group = values[5]
counts[(year, quarter, force, area, group)] += count
element.clear()
elif event == "end" and element.tag == TABLE_TAG:
table_name = None
element.clear()
rows = [
{
"period": period,
"year_end": 2000 + int(period.split("/")[1]),
"quarter": int(quarter) if str(quarter).isdigit() else 0,
"force": force,
"area": area,
"offence_group": group,
"count": count,
}
for (period, quarter, force, area, group), count in counts.items()
]
return pd.DataFrame(rows)
def load_crime_data(url, label, end_year):
"""Download and cache the area-level crime table."""
cache_file = CACHE_DIR / f"crime_csp_{end_year}.csv"
stamp = CACHE_DIR / f"crime_csp_{end_year}.url"
if cache_file.exists() and stamp.exists() and stamp.read_text().strip() == url:
return pd.read_csv(cache_file)
response = requests.get(url, headers=USER_AGENT, timeout=300)
response.raise_for_status()
frame = parse_csp_ods(response.content)
frame.to_csv(cache_file, index=False)
stamp.write_text(url)
return frame
def load_population():
"""Download and cache ONS mid-year population estimates by local authority."""
cache_file = CACHE_DIR / "population.csv"
if cache_file.exists() and time.time() - cache_file.stat().st_mtime < 120 * 86400:
return pd.read_csv(cache_file)
response = requests.get(ONS_POP_CSV, headers=USER_AGENT, timeout=120)
response.raise_for_status()
raw = pd.read_csv(io.BytesIO(response.content))
mask = (raw["Age"] == "Total") & (raw["Sex"] == "All")
raw = raw.loc[mask]
raw = raw[raw["administrative-geography"].astype(str).str.match(LAD_CODE_RE)]
frame = (
raw[["administrative-geography", "Geography", "v4_0"]]
.rename(columns={"administrative-geography": "area_code", "Geography": "ons_name", "v4_0": "population"})
.drop_duplicates("area_code")
)
frame["area_key"] = frame["ons_name"].map(normalize)
frame.to_csv(cache_file, index=False)
return frame
def warm_cache():
"""Pre-compute both caches (used at image build time)."""
url, label, end_year = _discover_csp_dataset()
load_crime_data(url, label, end_year)
load_population()
return url, label
def build_area_table(crime, population, group, year_end):
"""Area-level totals for one offence group and year, with rates."""
subset = crime[crime["year_end"] == year_end]
if group != "All offences":
subset = subset[subset["offence_group"] == group]
totals = (
subset.groupby(["area", "force"], as_index=False)["count"].sum()
)
totals["area_key"] = totals["area"].map(normalize)
pop = population[["area_key", "population", "ons_name"]].drop_duplicates("area_key")
totals = totals.merge(pop, on="area_key", how="left")
totals["rate_per_100k"] = totals["count"] / totals["population"] * 100_000
return totals
def format_int(value):
if pd.isna(value):
return "n/a"
return f"{int(value):,}"
def main():
st.set_page_config(
page_title="UK Crime Tracker",
page_icon="map",
layout="wide",
)
st.title("UK Crime Tracker")
st.caption(
"Police-recorded crime by local area across England and Wales, shown in "
"proportion to population. Data: Home Office recorded crime open data; "
"population: ONS mid-year estimates."
)
try:
with st.spinner("Discovering the latest Home Office release..."):
url, label, end_year = discover_csp_dataset()
with st.spinner("Loading crime data (first run can take ~1 minute)..."):
crime = load_crime_data(url, label, end_year)
with st.spinner("Loading ONS population estimates..."):
population = load_population()
except Exception as exc: # noqa: BLE001
st.error(f"Could not load data: {exc}")
st.stop()
crime = crime[~crime["area"].str.lower().str.startswith(("unassigned", "unallocated"))]
groups = sorted(g for g in crime["offence_group"].dropna().unique())
years = sorted(crime["year_end"].unique())
latest_year = int(max(years))
total_area_names = crime["area"].nunique()
st.markdown(
f"**Data release:** year ending March {label} | "
f"**Areas covered:** {total_area_names} | "
f"**Offence groups:** {len(groups)}"
)
st.divider()
controls = st.columns([2, 1, 1])
with controls[0]:
area = st.selectbox("Search for an area", sorted(crime["area"].unique()))
with controls[1]:
group = st.selectbox("Offence group", ["All offences"] + groups)
with controls[2]:
min_population = st.number_input(
"Minimum population for rankings", min_value=0, value=50_000, step=10_000
)
latest = build_area_table(crime, population, group, latest_year)
previous = build_area_table(crime, population, group, latest_year - 1)
selected = latest[latest["area"] == area]
selected_prev = previous[previous["area"] == area]
ranked = latest.dropna(subset=["rate_per_100k"])
ranked = ranked[ranked["population"] >= min_population]
ranked = ranked.sort_values("rate_per_100k", ascending=False).reset_index(drop=True)
ranked["rank"] = ranked.index + 1
if not selected.empty:
row = selected.iloc[0]
prev_count = selected_prev["count"].sum() if not selected_prev.empty else None
count = int(row["count"])
yoy = None
if prev_count:
yoy = (count - prev_count) / prev_count * 100
metric_cols = st.columns(4)
metric_cols[0].metric(f"Recorded crimes ({label})", format_int(count))
if pd.isna(row["population"]):
metric_cols[1].metric("Rate per 100,000", "no population match")
else:
metric_cols[1].metric("Rate per 100,000", f"{row['rate_per_100k']:,.0f}")
metric_cols[2].metric("Population", format_int(row["population"]))
if yoy is None:
metric_cols[3].metric("Year-on-year", "n/a")
else:
metric_cols[3].metric("Year-on-year", f"{yoy:+.1f}%", f"{count - prev_count:+,} crimes")
rank_row = ranked[ranked["area"] == area]
national = latest.dropna(subset=["population"])
if not rank_row.empty:
st.markdown(
f"**{area}** ranks **{int(rank_row.iloc[0]['rank'])}** of "
f"{len(ranked)} areas by {group.lower()} rate per 100,000 "
f"(areas with population >= {int(min_population):,})."
)
if not national.empty:
national_rate = national["count"].sum() / national["population"].sum() * 100_000
st.caption(
f"England and Wales benchmark for {group.lower()}: "
f"{national_rate:,.0f} per 100,000."
)
else:
st.warning("No data found for the selected area.")
st.divider()
st.subheader("Top 10 areas by crime rate")
top = ranked.head(10).iloc[::-1]
if top.empty:
st.info("No areas meet the population threshold.")
else:
figure = go.Figure(
go.Bar(
x=top["rate_per_100k"],
y=top["area"],
orientation="h",
marker_color="#c8102e",
customdata=top[["count", "population"]].to_numpy(),
hovertemplate=(
"%{y}
Rate: %{x:,.0f} per 100,000"
"
Crimes: %{customdata[0]:,.0f}"
"
Population: %{customdata[1]:,.0f}"
),
)
)
figure.update_layout(
xaxis_title="Crimes per 100,000 residents",
yaxis_title="",
height=460,
margin=dict(l=10, r=10, t=10, b=10),
)
st.plotly_chart(figure, width="stretch")
area_history = crime[crime["area"] == area]
if group != "All offences":
area_history = area_history[area_history["offence_group"] == group]
trend = (
area_history.groupby(["year_end", "quarter"], as_index=False)["count"].sum()
)
trend = trend[trend["quarter"] > 0].sort_values(["year_end", "quarter"])
if not trend.empty:
trend["label"] = (
trend["year_end"].sub(1).astype(str) + " Q" + trend["quarter"].astype(str)
)
st.subheader(f"Quarterly trend: {area}")
line = go.Figure(
go.Scatter(
x=trend["label"],
y=trend["count"],
mode="lines+markers",
line=dict(color="#1d2b53", width=2.5),
hovertemplate="%{x}
Crimes: %{y:,}",
)
)
line.update_layout(
xaxis_title="Quarter",
yaxis_title="Recorded crimes",
height=360,
margin=dict(l=10, r=10, t=10, b=10),
)
st.plotly_chart(line, width="stretch")
breakdown = (
crime[(crime["area"] == area) & (crime["year_end"] == latest_year)]
.groupby("offence_group", as_index=False)["count"]
.sum()
.sort_values("count", ascending=False)
)
if not breakdown.empty:
st.subheader(f"Offence mix for {area}, year ending March {label}")
pie = go.Figure(
go.Pie(
labels=breakdown["offence_group"],
values=breakdown["count"],
hole=0.45,
hovertemplate="%{label}
%{value:,} (%{percent})",
)
)
pie.update_layout(height=420, margin=dict(l=10, r=10, t=10, b=10))
st.plotly_chart(pie, width="stretch")
st.divider()
st.subheader("All areas")
table = latest[["area", "force", "count", "population", "rate_per_100k"]].copy()
table = table.sort_values("rate_per_100k", ascending=False)
table = table.rename(
columns={
"area": "Area",
"force": "Police force",
"count": "Recorded crimes",
"population": "Population",
"rate_per_100k": "Rate per 100,000",
}
)
st.dataframe(table, width="stretch", hide_index=True)
matched = table["Rate per 100,000"].notna().sum()
st.caption(
f"{matched} of {len(table)} areas matched to ONS population estimates; "
"unmatched areas still show recorded crime counts. Home Office open data "
"covers England and Wales."
)
if __name__ == "__main__":
main()