feat: Phase 2 — leadership 大屏 (/overview) + district normalization + drawer a11y
Phase 2 of the UX modernization. Three conflict-free workstreams. Leadership 驾驶舱 (/overview): - Wuhan 13-district Leaflet choropleth (public/wuhan_districts.geojson, keyed on name, darker=higher per 高风险高亮), legend, hover/click-zoom - 全部/门诊/住院 Segmented toggle drives choropleth + Top-5 district bar - literal "数据截至2023-12" as-of badge (D3 honesty); raw spinner → LoadingState - decompose OverviewDashboard 501→273; 6 components + 2 helpers under components/overview/ District normalization (backend data boundary): - case_loader.normalize_district + load_cases_by_district_daily collapse the 26 dirty labels (武昌/武昌区…) → 13 canonical; analysis/grid/insights repointed (fixes a grid-merge row-drop bug as a bonus); in-memory, schema unchanged Shell a11y (code-review carryover): - drawer is now a proper modal: ESC, body scroll-lock, focus-in + focus-trap cycle + focus-restore, role=dialog/aria-modal/aria-label, hamburger aria-expanded - SideNav expanded state lifted to AppShell so rail+drawer stay in sync - RouteErrorBoundary around <Outlet/> keeps shell chrome on page/chunk failure Gates: tsc 0 · vitest 64 · e2e 19/19 (17 user-flows + 2 overview) · build ok · backend pytest 6 new + 48 regression green Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -25,12 +25,42 @@ _cache: dict[str, Optional[pd.DataFrame | datetime]] = {
|
||||
"outpatient": None,
|
||||
"inpatient": None,
|
||||
"combined": None,
|
||||
"cases_by_district_daily": None,
|
||||
"loaded_at": None,
|
||||
}
|
||||
|
||||
# Guards the lazy build so concurrent callers don't duplicate the load/concat.
|
||||
_load_lock = threading.RLock()
|
||||
|
||||
# Canonical Wuhan administrative districts (13), matching the `name` field in
|
||||
# Datas/武汉市.geojson. All district roll-ups must collapse to exactly these.
|
||||
CANONICAL_DISTRICTS = [
|
||||
'江岸区', '江汉区', '硚口区', '汉阳区', '武昌区', '青山区', '洪山区',
|
||||
'东西湖区', '汉南区', '蔡甸区', '江夏区', '黄陂区', '新洲区',
|
||||
]
|
||||
# Bare (suffix-less) base name -> canonical 区-suffixed name.
|
||||
_DISTRICT_BASE_TO_CANONICAL = {d[:-1]: d for d in CANONICAL_DISTRICTS}
|
||||
_DISTRICT_SUFFIXES = ('区', '县', '市')
|
||||
|
||||
|
||||
def normalize_district(name: str) -> str:
|
||||
"""Map a district label to its canonical 区-suffixed form.
|
||||
|
||||
The case parquet carries both bare ("武昌") and suffixed ("武昌区") spellings
|
||||
of the same district, which double-counts in any roll-up. This collapses
|
||||
them: known bare names map to their canonical form; already-suffixed names
|
||||
pass through unchanged; anything else gets a "区" appended.
|
||||
"""
|
||||
if name is None:
|
||||
return name
|
||||
name = str(name).strip()
|
||||
if name in _DISTRICT_BASE_TO_CANONICAL:
|
||||
return _DISTRICT_BASE_TO_CANONICAL[name]
|
||||
if name.endswith(_DISTRICT_SUFFIXES):
|
||||
return name
|
||||
return f"{name}区"
|
||||
|
||||
|
||||
# Wuhan district mapping
|
||||
WUHAN_DISTRICTS = {
|
||||
'江岸区': ['江岸'],
|
||||
@@ -170,3 +200,40 @@ def get_inpatient_data() -> pd.DataFrame:
|
||||
"""Return the cached inpatient dataframe"""
|
||||
load_data()
|
||||
return _cache["inpatient"] # type: ignore[return-value]
|
||||
|
||||
|
||||
def load_cases_by_district_daily() -> pd.DataFrame:
|
||||
"""Load processed/cases_by_district_daily.parquet with districts normalized.
|
||||
|
||||
The on-disk parquet carries both bare and 区-suffixed spellings of each
|
||||
district (26 labels = 13 districts × 2 spellings), so any groupby on the
|
||||
raw `district` column double-counts. This is the single data-access
|
||||
boundary: it normalizes labels to the canonical 13 and re-aggregates
|
||||
(sum of outpatient_count / inpatient_count / total_cases per
|
||||
normalized district + date), so every downstream consumer
|
||||
(analysis / grid / insights) sees clean, deduped 13-district data.
|
||||
|
||||
Returns a copy with columns [date, district, outpatient_count,
|
||||
inpatient_count, total_cases]. Raises FileNotFoundError if the parquet
|
||||
is missing (callers handle this as they did before).
|
||||
"""
|
||||
path = PROCESSED_DIR / "cases_by_district_daily.parquet"
|
||||
cached = _cache.get("cases_by_district_daily")
|
||||
if cached is not None:
|
||||
return cast(pd.DataFrame, cached).copy()
|
||||
|
||||
with _load_lock:
|
||||
cached = _cache.get("cases_by_district_daily")
|
||||
if cached is not None:
|
||||
return cast(pd.DataFrame, cached).copy()
|
||||
|
||||
df = pd.read_parquet(path)
|
||||
df["district"] = df["district"].map(normalize_district)
|
||||
agg = (
|
||||
df.groupby(["date", "district"], as_index=False)[
|
||||
["outpatient_count", "inpatient_count", "total_cases"]
|
||||
]
|
||||
.sum()
|
||||
)
|
||||
_cache["cases_by_district_daily"] = agg # type: ignore[assignment]
|
||||
return agg.copy()
|
||||
|
||||
@@ -12,6 +12,7 @@ import pandas as pd
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from config import DATA_DIR, RISK_HIGH, PROJECT_ROOT, WUHAN_BOUNDS, LAT_STEP, LON_STEP
|
||||
from data.case_loader import load_cases_by_district_daily
|
||||
from utils.date_helpers import get_latest_date
|
||||
from utils.geojson import parse_geojson_file, load_districts
|
||||
from utils.geo import point_in_polygon
|
||||
@@ -179,18 +180,16 @@ def _district_avg_aqi() -> dict:
|
||||
def _district_total_cases() -> dict:
|
||||
"""Real total recorded cases per district from cases_by_district_daily.
|
||||
|
||||
District labels in the case file are inconsistent ("武昌" vs "武昌区"),
|
||||
so names are normalized by stripping the "区" suffix and summed, then
|
||||
keyed by the canonical mapping name (with "区"). Returns {district: cases}.
|
||||
District labels are normalized to the canonical 13 区-suffixed names at the
|
||||
data-access boundary (data.case_loader), so this is a plain per-district
|
||||
sum. Returns {district: cases}.
|
||||
"""
|
||||
path = PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet"
|
||||
if not path.exists():
|
||||
try:
|
||||
df = load_cases_by_district_daily()
|
||||
except FileNotFoundError:
|
||||
return {}
|
||||
df = pd.read_parquet(path, columns=["district", "total_cases"])
|
||||
df = df.copy()
|
||||
df["base"] = df["district"].str.replace("区", "", regex=False)
|
||||
by_base = df.groupby("base")["total_cases"].sum()
|
||||
return {f"{base}区": int(v) for base, v in by_base.items()}
|
||||
by_district = df.groupby("district")["total_cases"].sum()
|
||||
return {str(d): int(v) for d, v in by_district.items()}
|
||||
|
||||
|
||||
@lru_cache(maxsize=8)
|
||||
|
||||
@@ -21,6 +21,7 @@ from models import (
|
||||
MultiDayPredictionRequest,
|
||||
MultiDayPredictionResponse,
|
||||
)
|
||||
from data.case_loader import load_cases_by_district_daily
|
||||
|
||||
router = APIRouter(prefix="/api", tags=["grid"])
|
||||
|
||||
@@ -43,14 +44,13 @@ def _compute_historical_aggregation(
|
||||
) -> HistoricalAggregationResponse:
|
||||
"""Run the full pandas aggregation pipeline (called in thread pool)."""
|
||||
try:
|
||||
cases_df = _load_parquet(PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet")
|
||||
cases_df = load_cases_by_district_daily()
|
||||
except FileNotFoundError:
|
||||
return HistoricalAggregationResponse(
|
||||
aggregations=[], total_records=0,
|
||||
date_range=(start.strftime("%Y-%m-%d"), end.strftime("%Y-%m-%d")),
|
||||
timestamp=datetime.now().isoformat(),
|
||||
)
|
||||
cases_df = cases_df.copy()
|
||||
cases_df['date'] = pd.to_datetime(cases_df['date'])
|
||||
|
||||
filtered_cases = cases_df[
|
||||
@@ -196,7 +196,7 @@ def _grids_geojson_body(date: str, district: Optional[str], risk_level: Optional
|
||||
if district:
|
||||
merged = merged[merged['district_name'].str.contains(district.replace('区', ''), na=False, regex=False)]
|
||||
|
||||
cases_df = _load_parquet(PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet").copy()
|
||||
cases_df = load_cases_by_district_daily()
|
||||
cases_df['date'] = pd.to_datetime(cases_df['date']).dt.strftime('%Y-%m-%d')
|
||||
cases_df = cases_df[cases_df['date'] == date]
|
||||
|
||||
@@ -364,7 +364,7 @@ def _compute_grid_history(grid_id: str, days: int) -> dict:
|
||||
|
||||
district = grid_info.iloc[0]['district_name']
|
||||
|
||||
cases_df = _load_parquet(PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet")
|
||||
cases_df = load_cases_by_district_daily()
|
||||
cases_df['date'] = pd.to_datetime(cases_df['date'])
|
||||
|
||||
end_date = datetime.now()
|
||||
|
||||
@@ -11,6 +11,7 @@ from pydantic import BaseModel, Field
|
||||
from typing import Dict, List, Literal
|
||||
|
||||
from config import DATA_DIR, RISK_HIGH, PROJECT_ROOT
|
||||
from data.case_loader import load_cases_by_district_daily
|
||||
from models import (
|
||||
InsightsResponse,
|
||||
InsightTrend,
|
||||
@@ -491,22 +492,22 @@ async def get_insights_cards():
|
||||
|
||||
cases_path = PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet"
|
||||
if cases_path.exists():
|
||||
cases_df = _cached_parquet(str(cases_path))
|
||||
# Districts already normalized to the canonical 13 区-suffixed names.
|
||||
cases_df = load_cases_by_district_daily()
|
||||
cases_df["date"] = pd.to_datetime(cases_df["date"])
|
||||
latest_case_date = cases_df["date"].max()
|
||||
latest_cases = cases_df[cases_df["date"] == latest_case_date].copy()
|
||||
latest_cases["base_district"] = latest_cases["district"].str.replace("区", "")
|
||||
district_daily = latest_cases.groupby("base_district")["total_cases"].sum().sort_values(ascending=False)
|
||||
latest_cases = cases_df[cases_df["date"] == latest_case_date]
|
||||
district_daily = latest_cases.groupby("district")["total_cases"].sum().sort_values(ascending=False)
|
||||
total_daily = int(district_daily.sum())
|
||||
top_name = district_daily.index[0]
|
||||
top_val = int(district_daily.iloc[0])
|
||||
num_districts = len(district_daily)
|
||||
|
||||
week_ago = latest_case_date - pd.Timedelta(days=6)
|
||||
week_cases = cases_df[cases_df["date"] >= week_ago].copy()
|
||||
week_cases["base_district"] = week_cases["district"].str.replace("区", "")
|
||||
week_cases = cases_df[cases_df["date"] >= week_ago]
|
||||
daily_totals = week_cases.groupby("date")["total_cases"].sum()
|
||||
avg_daily = int(daily_totals.mean())
|
||||
week_district = week_cases.groupby("base_district")["total_cases"].sum().sort_values(ascending=False)
|
||||
week_district = week_cases.groupby("district")["total_cases"].sum().sort_values(ascending=False)
|
||||
week_top_val = int(week_district.iloc[0])
|
||||
|
||||
date_str = latest_case_date.strftime("%m月%d日")
|
||||
@@ -515,8 +516,8 @@ async def get_insights_cards():
|
||||
title=f"日病例统计 ({date_str})",
|
||||
description=(
|
||||
f"最近统计日({date_str})全市{num_districts}个区共记录{total_daily}例儿童呼吸道疾病病例,"
|
||||
f"{top_name}区{top_val}例为当日最高。近7日日均{avg_daily}例,"
|
||||
f"{week_district.index[0]}区累计{week_top_val}例居首。"
|
||||
f"{top_name}{top_val}例为当日最高。近7日日均{avg_daily}例,"
|
||||
f"{week_district.index[0]}累计{week_top_val}例居首。"
|
||||
),
|
||||
type="warning",
|
||||
metric="日病例",
|
||||
|
||||
82
backend/tests/test_district_normalization.py
Normal file
82
backend/tests/test_district_normalization.py
Normal file
@@ -0,0 +1,82 @@
|
||||
"""Tests for district label normalization at the case-loader boundary.
|
||||
|
||||
The processed/cases_by_district_daily.parquet carries both bare ("武昌") and
|
||||
区-suffixed ("武昌区") spellings of each district (26 labels = 13 districts × 2
|
||||
spellings), which double-counts in any roll-up. data.case_loader normalizes
|
||||
these to the canonical 13 区-suffixed names and re-aggregates. These tests pin
|
||||
that behavior.
|
||||
"""
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
# Ensure the backend package root is importable at collection time (mirrors the
|
||||
# sys.path handling other modules rely on once the app is imported).
|
||||
BACKEND_ROOT = Path(__file__).parent.parent
|
||||
if str(BACKEND_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(BACKEND_ROOT))
|
||||
|
||||
from data.case_loader import ( # noqa: E402
|
||||
CANONICAL_DISTRICTS,
|
||||
normalize_district,
|
||||
load_cases_by_district_daily,
|
||||
)
|
||||
|
||||
PROJECT_ROOT = Path(__file__).parent.parent.parent
|
||||
RAW_PARQUET = PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet"
|
||||
|
||||
|
||||
def test_normalize_district_known_bare_forms():
|
||||
"""Every known bare form maps to its canonical 区-suffixed name."""
|
||||
cases = {
|
||||
"武昌": "武昌区", "汉阳": "汉阳区", "江岸": "江岸区", "硚口": "硚口区",
|
||||
"青山": "青山区", "洪山": "洪山区", "东西湖": "东西湖区", "汉南": "汉南区",
|
||||
"蔡甸": "蔡甸区", "江夏": "江夏区", "黄陂": "黄陂区", "新洲": "新洲区",
|
||||
"江汉": "江汉区",
|
||||
}
|
||||
for bare, canonical in cases.items():
|
||||
assert normalize_district(bare) == canonical
|
||||
|
||||
|
||||
def test_normalize_district_already_suffixed_passes_through():
|
||||
for d in CANONICAL_DISTRICTS:
|
||||
assert normalize_district(d) == d
|
||||
|
||||
|
||||
def test_canonical_set_is_exactly_thirteen():
|
||||
assert len(CANONICAL_DISTRICTS) == 13
|
||||
assert len(set(CANONICAL_DISTRICTS)) == 13
|
||||
|
||||
|
||||
@pytest.mark.skipif(not RAW_PARQUET.exists(), reason="case parquet not present")
|
||||
def test_loader_collapses_to_thirteen_canonical_districts():
|
||||
df = load_cases_by_district_daily()
|
||||
districts = set(df["district"].unique())
|
||||
|
||||
# (a) exactly 13 unique districts, all canonical
|
||||
assert len(districts) == 13, f"expected 13 districts, got {len(districts)}: {sorted(districts)}"
|
||||
assert districts == set(CANONICAL_DISTRICTS)
|
||||
|
||||
# (b) no bare / unsuffixed duplicates remain
|
||||
for name in districts:
|
||||
assert name.endswith(("区", "县", "市")), f"unsuffixed district leaked: {name}"
|
||||
|
||||
|
||||
@pytest.mark.skipif(not RAW_PARQUET.exists(), reason="case parquet not present")
|
||||
def test_loader_preserves_totals_no_rows_dropped_or_double_counted():
|
||||
"""Sum integrity: normalized total == raw parquet total."""
|
||||
raw = pd.read_parquet(RAW_PARQUET)
|
||||
normalized = load_cases_by_district_daily()
|
||||
|
||||
assert int(normalized["total_cases"].sum()) == int(raw["total_cases"].sum())
|
||||
assert int(normalized["outpatient_count"].sum()) == int(raw["outpatient_count"].sum())
|
||||
assert int(normalized["inpatient_count"].sum()) == int(raw["inpatient_count"].sum())
|
||||
|
||||
|
||||
@pytest.mark.skipif(not RAW_PARQUET.exists(), reason="case parquet not present")
|
||||
def test_raw_parquet_actually_has_dirty_labels():
|
||||
"""Sanity: the raw file really has the 26-label problem we are fixing."""
|
||||
raw = pd.read_parquet(RAW_PARQUET)
|
||||
assert raw["district"].nunique() > 13
|
||||
Reference in New Issue
Block a user