feat: Phase 2 — leadership 大屏 (/overview) + district normalization + drawer a11y

Phase 2 of the UX modernization. Three conflict-free workstreams.

Leadership 驾驶舱 (/overview):
- Wuhan 13-district Leaflet choropleth (public/wuhan_districts.geojson, keyed
  on name, darker=higher per 高风险高亮), legend, hover/click-zoom
- 全部/门诊/住院 Segmented toggle drives choropleth + Top-5 district bar
- literal "数据截至2023-12" as-of badge (D3 honesty); raw spinner → LoadingState
- decompose OverviewDashboard 501→273; 6 components + 2 helpers under components/overview/

District normalization (backend data boundary):
- case_loader.normalize_district + load_cases_by_district_daily collapse the
  26 dirty labels (武昌/武昌区…) → 13 canonical; analysis/grid/insights repointed
  (fixes a grid-merge row-drop bug as a bonus); in-memory, schema unchanged

Shell a11y (code-review carryover):
- drawer is now a proper modal: ESC, body scroll-lock, focus-in + focus-trap
  cycle + focus-restore, role=dialog/aria-modal/aria-label, hamburger aria-expanded
- SideNav expanded state lifted to AppShell so rail+drawer stay in sync
- RouteErrorBoundary around <Outlet/> keeps shell chrome on page/chunk failure

Gates: tsc 0 · vitest 64 · e2e 19/19 (17 user-flows + 2 overview) · build ok
· backend pytest 6 new + 48 regression green

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-06-21 20:24:26 +08:00
parent 3db3b12480
commit 9a94156acc
22 changed files with 1378 additions and 358 deletions

View File

@@ -25,12 +25,42 @@ _cache: dict[str, Optional[pd.DataFrame | datetime]] = {
"outpatient": None,
"inpatient": None,
"combined": None,
"cases_by_district_daily": None,
"loaded_at": None,
}
# Guards the lazy build so concurrent callers don't duplicate the load/concat.
_load_lock = threading.RLock()
# Canonical Wuhan administrative districts (13), matching the `name` field in
# Datas/武汉市.geojson. All district roll-ups must collapse to exactly these.
CANONICAL_DISTRICTS = [
'江岸区', '江汉区', '硚口区', '汉阳区', '武昌区', '青山区', '洪山区',
'东西湖区', '汉南区', '蔡甸区', '江夏区', '黄陂区', '新洲区',
]
# Bare (suffix-less) base name -> canonical 区-suffixed name.
_DISTRICT_BASE_TO_CANONICAL = {d[:-1]: d for d in CANONICAL_DISTRICTS}
_DISTRICT_SUFFIXES = ('', '', '')
def normalize_district(name: str) -> str:
"""Map a district label to its canonical 区-suffixed form.
The case parquet carries both bare ("武昌") and suffixed ("武昌区") spellings
of the same district, which double-counts in any roll-up. This collapses
them: known bare names map to their canonical form; already-suffixed names
pass through unchanged; anything else gets a "" appended.
"""
if name is None:
return name
name = str(name).strip()
if name in _DISTRICT_BASE_TO_CANONICAL:
return _DISTRICT_BASE_TO_CANONICAL[name]
if name.endswith(_DISTRICT_SUFFIXES):
return name
return f"{name}"
# Wuhan district mapping
WUHAN_DISTRICTS = {
'江岸区': ['江岸'],
@@ -170,3 +200,40 @@ def get_inpatient_data() -> pd.DataFrame:
"""Return the cached inpatient dataframe"""
load_data()
return _cache["inpatient"] # type: ignore[return-value]
def load_cases_by_district_daily() -> pd.DataFrame:
"""Load processed/cases_by_district_daily.parquet with districts normalized.
The on-disk parquet carries both bare and 区-suffixed spellings of each
district (26 labels = 13 districts × 2 spellings), so any groupby on the
raw `district` column double-counts. This is the single data-access
boundary: it normalizes labels to the canonical 13 and re-aggregates
(sum of outpatient_count / inpatient_count / total_cases per
normalized district + date), so every downstream consumer
(analysis / grid / insights) sees clean, deduped 13-district data.
Returns a copy with columns [date, district, outpatient_count,
inpatient_count, total_cases]. Raises FileNotFoundError if the parquet
is missing (callers handle this as they did before).
"""
path = PROCESSED_DIR / "cases_by_district_daily.parquet"
cached = _cache.get("cases_by_district_daily")
if cached is not None:
return cast(pd.DataFrame, cached).copy()
with _load_lock:
cached = _cache.get("cases_by_district_daily")
if cached is not None:
return cast(pd.DataFrame, cached).copy()
df = pd.read_parquet(path)
df["district"] = df["district"].map(normalize_district)
agg = (
df.groupby(["date", "district"], as_index=False)[
["outpatient_count", "inpatient_count", "total_cases"]
]
.sum()
)
_cache["cases_by_district_daily"] = agg # type: ignore[assignment]
return agg.copy()

View File

@@ -12,6 +12,7 @@ import pandas as pd
from pydantic import BaseModel, Field
from config import DATA_DIR, RISK_HIGH, PROJECT_ROOT, WUHAN_BOUNDS, LAT_STEP, LON_STEP
from data.case_loader import load_cases_by_district_daily
from utils.date_helpers import get_latest_date
from utils.geojson import parse_geojson_file, load_districts
from utils.geo import point_in_polygon
@@ -179,18 +180,16 @@ def _district_avg_aqi() -> dict:
def _district_total_cases() -> dict:
"""Real total recorded cases per district from cases_by_district_daily.
District labels in the case file are inconsistent ("武昌" vs "武昌区"),
so names are normalized by stripping the "" suffix and summed, then
keyed by the canonical mapping name (with ""). Returns {district: cases}.
District labels are normalized to the canonical 13 区-suffixed names at the
data-access boundary (data.case_loader), so this is a plain per-district
sum. Returns {district: cases}.
"""
path = PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet"
if not path.exists():
try:
df = load_cases_by_district_daily()
except FileNotFoundError:
return {}
df = pd.read_parquet(path, columns=["district", "total_cases"])
df = df.copy()
df["base"] = df["district"].str.replace("", "", regex=False)
by_base = df.groupby("base")["total_cases"].sum()
return {f"{base}": int(v) for base, v in by_base.items()}
by_district = df.groupby("district")["total_cases"].sum()
return {str(d): int(v) for d, v in by_district.items()}
@lru_cache(maxsize=8)

View File

@@ -21,6 +21,7 @@ from models import (
MultiDayPredictionRequest,
MultiDayPredictionResponse,
)
from data.case_loader import load_cases_by_district_daily
router = APIRouter(prefix="/api", tags=["grid"])
@@ -43,14 +44,13 @@ def _compute_historical_aggregation(
) -> HistoricalAggregationResponse:
"""Run the full pandas aggregation pipeline (called in thread pool)."""
try:
cases_df = _load_parquet(PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet")
cases_df = load_cases_by_district_daily()
except FileNotFoundError:
return HistoricalAggregationResponse(
aggregations=[], total_records=0,
date_range=(start.strftime("%Y-%m-%d"), end.strftime("%Y-%m-%d")),
timestamp=datetime.now().isoformat(),
)
cases_df = cases_df.copy()
cases_df['date'] = pd.to_datetime(cases_df['date'])
filtered_cases = cases_df[
@@ -196,7 +196,7 @@ def _grids_geojson_body(date: str, district: Optional[str], risk_level: Optional
if district:
merged = merged[merged['district_name'].str.contains(district.replace('', ''), na=False, regex=False)]
cases_df = _load_parquet(PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet").copy()
cases_df = load_cases_by_district_daily()
cases_df['date'] = pd.to_datetime(cases_df['date']).dt.strftime('%Y-%m-%d')
cases_df = cases_df[cases_df['date'] == date]
@@ -364,7 +364,7 @@ def _compute_grid_history(grid_id: str, days: int) -> dict:
district = grid_info.iloc[0]['district_name']
cases_df = _load_parquet(PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet")
cases_df = load_cases_by_district_daily()
cases_df['date'] = pd.to_datetime(cases_df['date'])
end_date = datetime.now()

View File

@@ -11,6 +11,7 @@ from pydantic import BaseModel, Field
from typing import Dict, List, Literal
from config import DATA_DIR, RISK_HIGH, PROJECT_ROOT
from data.case_loader import load_cases_by_district_daily
from models import (
InsightsResponse,
InsightTrend,
@@ -491,22 +492,22 @@ async def get_insights_cards():
cases_path = PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet"
if cases_path.exists():
cases_df = _cached_parquet(str(cases_path))
# Districts already normalized to the canonical 13 区-suffixed names.
cases_df = load_cases_by_district_daily()
cases_df["date"] = pd.to_datetime(cases_df["date"])
latest_case_date = cases_df["date"].max()
latest_cases = cases_df[cases_df["date"] == latest_case_date].copy()
latest_cases["base_district"] = latest_cases["district"].str.replace("", "")
district_daily = latest_cases.groupby("base_district")["total_cases"].sum().sort_values(ascending=False)
latest_cases = cases_df[cases_df["date"] == latest_case_date]
district_daily = latest_cases.groupby("district")["total_cases"].sum().sort_values(ascending=False)
total_daily = int(district_daily.sum())
top_name = district_daily.index[0]
top_val = int(district_daily.iloc[0])
num_districts = len(district_daily)
week_ago = latest_case_date - pd.Timedelta(days=6)
week_cases = cases_df[cases_df["date"] >= week_ago].copy()
week_cases["base_district"] = week_cases["district"].str.replace("", "")
week_cases = cases_df[cases_df["date"] >= week_ago]
daily_totals = week_cases.groupby("date")["total_cases"].sum()
avg_daily = int(daily_totals.mean())
week_district = week_cases.groupby("base_district")["total_cases"].sum().sort_values(ascending=False)
week_district = week_cases.groupby("district")["total_cases"].sum().sort_values(ascending=False)
week_top_val = int(week_district.iloc[0])
date_str = latest_case_date.strftime("%m月%d")
@@ -515,8 +516,8 @@ async def get_insights_cards():
title=f"日病例统计 ({date_str})",
description=(
f"最近统计日({date_str})全市{num_districts}个区共记录{total_daily}例儿童呼吸道疾病病例,"
f"{top_name}{top_val}例为当日最高。近7日日均{avg_daily}例,"
f"{week_district.index[0]}累计{week_top_val}例居首。"
f"{top_name}{top_val}例为当日最高。近7日日均{avg_daily}例,"
f"{week_district.index[0]}累计{week_top_val}例居首。"
),
type="warning",
metric="日病例",

View File

@@ -0,0 +1,82 @@
"""Tests for district label normalization at the case-loader boundary.
The processed/cases_by_district_daily.parquet carries both bare ("武昌") and
区-suffixed ("武昌区") spellings of each district (26 labels = 13 districts × 2
spellings), which double-counts in any roll-up. data.case_loader normalizes
these to the canonical 13 区-suffixed names and re-aggregates. These tests pin
that behavior.
"""
import sys
from pathlib import Path
import pandas as pd
import pytest
# Ensure the backend package root is importable at collection time (mirrors the
# sys.path handling other modules rely on once the app is imported).
BACKEND_ROOT = Path(__file__).parent.parent
if str(BACKEND_ROOT) not in sys.path:
sys.path.insert(0, str(BACKEND_ROOT))
from data.case_loader import ( # noqa: E402
CANONICAL_DISTRICTS,
normalize_district,
load_cases_by_district_daily,
)
PROJECT_ROOT = Path(__file__).parent.parent.parent
RAW_PARQUET = PROJECT_ROOT / "processed" / "cases_by_district_daily.parquet"
def test_normalize_district_known_bare_forms():
"""Every known bare form maps to its canonical 区-suffixed name."""
cases = {
"武昌": "武昌区", "汉阳": "汉阳区", "江岸": "江岸区", "硚口": "硚口区",
"青山": "青山区", "洪山": "洪山区", "东西湖": "东西湖区", "汉南": "汉南区",
"蔡甸": "蔡甸区", "江夏": "江夏区", "黄陂": "黄陂区", "新洲": "新洲区",
"江汉": "江汉区",
}
for bare, canonical in cases.items():
assert normalize_district(bare) == canonical
def test_normalize_district_already_suffixed_passes_through():
for d in CANONICAL_DISTRICTS:
assert normalize_district(d) == d
def test_canonical_set_is_exactly_thirteen():
assert len(CANONICAL_DISTRICTS) == 13
assert len(set(CANONICAL_DISTRICTS)) == 13
@pytest.mark.skipif(not RAW_PARQUET.exists(), reason="case parquet not present")
def test_loader_collapses_to_thirteen_canonical_districts():
df = load_cases_by_district_daily()
districts = set(df["district"].unique())
# (a) exactly 13 unique districts, all canonical
assert len(districts) == 13, f"expected 13 districts, got {len(districts)}: {sorted(districts)}"
assert districts == set(CANONICAL_DISTRICTS)
# (b) no bare / unsuffixed duplicates remain
for name in districts:
assert name.endswith(("", "", "")), f"unsuffixed district leaked: {name}"
@pytest.mark.skipif(not RAW_PARQUET.exists(), reason="case parquet not present")
def test_loader_preserves_totals_no_rows_dropped_or_double_counted():
"""Sum integrity: normalized total == raw parquet total."""
raw = pd.read_parquet(RAW_PARQUET)
normalized = load_cases_by_district_daily()
assert int(normalized["total_cases"].sum()) == int(raw["total_cases"].sum())
assert int(normalized["outpatient_count"].sum()) == int(raw["outpatient_count"].sum())
assert int(normalized["inpatient_count"].sum()) == int(raw["inpatient_count"].sum())
@pytest.mark.skipif(not RAW_PARQUET.exists(), reason="case parquet not present")
def test_raw_parquet_actually_has_dirty_labels():
"""Sanity: the raw file really has the 26-label problem we are fixing."""
raw = pd.read_parquet(RAW_PARQUET)
assert raw["district"].nunique() > 13