feat: Initial CBPOA commit — 武汉儿童呼吸疾病风险评估系统
Context: Build a spatial risk assessment system correlating air quality data with children's respiratory disease incidence across Wuhan. Approach: FastAPI backend serving PostGIS spatial queries, React frontend with Deck.gl maps, and a PyTorch SpatialTemporalGCN pipeline for multi-day (1d/3d/7d) risk prediction. Changes: - backend/ — FastAPI API with auth (JWT), alerts, risk analysis, geocoded case data, grid statistics, and report endpoints - frontend/ — React dashboard with interactive risk maps, alert monitoring, district comparison charts, and timeline player - models/ — SpatialTemporalGCN model with trained weights and ONNX export for inference - scripts/ — ETL pipeline for weather + medical data, grid generation, feature engineering, training, and daily inference - deploy/ — Docker Compose configs for backend, frontend, and MLflow - docs/ — API docs, deployment guide, user guide, and code review Impact: Enables spatial risk visualization, alert monitoring, and ML-driven health risk forecasting for environmental health teams.
This commit is contained in:
370
backend/routers/cases.py
Normal file
370
backend/routers/cases.py
Normal file
@@ -0,0 +1,370 @@
|
||||
"""
|
||||
医疗病例数据 API 路由
|
||||
|
||||
提供门诊和住院数据的统计、趋势、区域分布等接口
|
||||
"""
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Query
|
||||
from pydantic import BaseModel
|
||||
from typing import Optional
|
||||
from datetime import datetime, date
|
||||
import pandas as pd
|
||||
import re
|
||||
from pathlib import Path
|
||||
import json
|
||||
|
||||
DATE_PATTERN = re.compile(r"^\d{4}-\d{2}-\d{2}$")
|
||||
|
||||
router = APIRouter(prefix="/api/cases", tags=["cases"])
|
||||
|
||||
# 数据缓存
|
||||
_cache = {
|
||||
"outpatient": None,
|
||||
"inpatient": None,
|
||||
"loaded_at": None,
|
||||
}
|
||||
|
||||
# 武汉市区映射
|
||||
WUHAN_DISTRICTS = {
|
||||
'江岸区': ['江岸'],
|
||||
'江汉区': ['江汉'],
|
||||
'武昌区': ['武昌'],
|
||||
'洪山区': ['洪山'],
|
||||
'汉阳区': ['汉阳'],
|
||||
'东西湖区': ['东西湖'],
|
||||
'黄陂区': ['黄陂'],
|
||||
'硚口区': ['硚口'],
|
||||
'江夏区': ['江夏'],
|
||||
'青山区': ['青山'],
|
||||
'新洲区': ['新洲'],
|
||||
'蔡甸区': ['蔡甸'],
|
||||
'东湖新技术开发区': ['东湖新技术开发区', '光谷'],
|
||||
'经开(汉南)区': ['经开', '汉南', '经济开发区'],
|
||||
'东湖生态旅游风景区': ['东湖生态旅游风景区']
|
||||
}
|
||||
|
||||
PROJECT_ROOT = Path(__file__).parent.parent.parent
|
||||
DATA_DIR = PROJECT_ROOT / "Datas"
|
||||
|
||||
|
||||
def _extract_district(addr: str) -> str:
|
||||
"""从地址提取武汉市区名"""
|
||||
if pd.isna(addr):
|
||||
return '未知'
|
||||
addr = str(addr)
|
||||
for district, keywords in WUHAN_DISTRICTS.items():
|
||||
for kw in keywords:
|
||||
if kw in addr:
|
||||
return district
|
||||
return '其他'
|
||||
|
||||
|
||||
def _load_data():
|
||||
"""加载并缓存数据"""
|
||||
if _cache["loaded_at"] is not None:
|
||||
return
|
||||
|
||||
try:
|
||||
# 加载门诊数据
|
||||
df_out = pd.read_excel(DATA_DIR / "view_门诊.xlsx")
|
||||
df_out['date'] = pd.to_datetime(df_out['门诊日期_re'])
|
||||
df_out['district'] = df_out['现住址区'].fillna('未知')
|
||||
_cache["outpatient"] = df_out
|
||||
|
||||
# 加载住院数据
|
||||
df_in = pd.read_excel(DATA_DIR / "view_住院.xlsx")
|
||||
df_in['date'] = pd.to_datetime(df_in['入院日期_re'])
|
||||
df_in['district'] = df_in['现住址_脱敏'].apply(_extract_district)
|
||||
_cache["inpatient"] = df_in
|
||||
|
||||
_cache["loaded_at"] = datetime.now()
|
||||
except Exception as e:
|
||||
raise RuntimeError(f"数据加载失败:{str(e)}")
|
||||
|
||||
|
||||
def _get_combined_data():
|
||||
"""获取合并的病例数据"""
|
||||
_load_data()
|
||||
|
||||
df_out = _cache["outpatient"][['date', 'district', '初诊', '主诉']].copy()
|
||||
df_out['type'] = 'outpatient'
|
||||
df_out['diagnosis'] = df_out['初诊']
|
||||
|
||||
df_in = _cache["inpatient"][['date', 'district', '诊断名称']].copy()
|
||||
df_in['type'] = 'inpatient'
|
||||
df_in['diagnosis'] = df_in['诊断名称']
|
||||
df_in['主诉'] = None
|
||||
|
||||
return pd.concat([df_out, df_in], ignore_index=True)
|
||||
|
||||
|
||||
# ============== Response Models ==============
|
||||
|
||||
class StatsResponse(BaseModel):
|
||||
"""统计数据响应"""
|
||||
total_outpatient: int
|
||||
total_inpatient: int
|
||||
date_range: dict
|
||||
top_districts: list
|
||||
top_diagnoses: list
|
||||
|
||||
|
||||
class TrendPoint(BaseModel):
|
||||
"""趋势数据点"""
|
||||
date: str
|
||||
outpatient: int
|
||||
inpatient: int
|
||||
total: int
|
||||
|
||||
|
||||
class TrendResponse(BaseModel):
|
||||
"""趋势数据响应"""
|
||||
trend: list[TrendPoint]
|
||||
summary: dict
|
||||
|
||||
|
||||
class DistrictData(BaseModel):
|
||||
"""区域数据"""
|
||||
district: str
|
||||
outpatient: int
|
||||
inpatient: int
|
||||
total: int
|
||||
outpatient_ratio: float
|
||||
inpatient_ratio: float
|
||||
|
||||
|
||||
class DistrictsResponse(BaseModel):
|
||||
"""区域分布响应"""
|
||||
districts: list[DistrictData]
|
||||
total: int
|
||||
|
||||
|
||||
class RealtimeData(BaseModel):
|
||||
"""实时数据"""
|
||||
today_outpatient: int
|
||||
today_inpatient: int
|
||||
today_total: int
|
||||
last_7d_avg: int
|
||||
change_ratio: float
|
||||
status: str
|
||||
|
||||
|
||||
# ============== API Endpoints ==============
|
||||
|
||||
@router.get("/stats", response_model=StatsResponse, summary="获取病例统计数据")
|
||||
async def get_cases_stats():
|
||||
"""
|
||||
获取病例总体统计信息
|
||||
|
||||
- 总门诊量、总住院量
|
||||
- 数据日期范围
|
||||
- 就诊量前 10 的区域
|
||||
- 最常见诊断前 10
|
||||
"""
|
||||
_load_data()
|
||||
|
||||
df_out = _cache["outpatient"]
|
||||
df_in = _cache["inpatient"]
|
||||
|
||||
# 计算统计
|
||||
total_outpatient = len(df_out)
|
||||
total_inpatient = len(df_in)
|
||||
|
||||
# 日期范围
|
||||
min_date = min(df_out['date'].min(), df_in['date'].min())
|
||||
max_date = max(df_out['date'].max(), df_in['date'].max())
|
||||
|
||||
# 区域统计
|
||||
out_districts = df_out[df_out['district'] != '未知']['district'].value_counts().head(10)
|
||||
in_districts = df_in[df_in['district'] != '其他']['district'].value_counts().head(10)
|
||||
|
||||
combined_districts = pd.concat([out_districts, in_districts]).groupby(level=0).sum().nlargest(10)
|
||||
top_districts = [{"district": d, "count": int(c)} for d, c in combined_districts.items()]
|
||||
|
||||
# 诊断统计
|
||||
out_diagnoses = df_out['初诊'].value_counts().head(10)
|
||||
in_diagnoses = df_in['诊断名称'].value_counts().head(10)
|
||||
|
||||
top_diagnoses = [
|
||||
{"diagnosis": str(d), "outpatient": int(out_diagnoses.get(d, 0)), "inpatient": int(in_diagnoses.get(d, 0))}
|
||||
for d in set(list(out_diagnoses.index[:5]) + list(in_diagnoses.index[:5]))
|
||||
][:10]
|
||||
|
||||
return StatsResponse(
|
||||
total_outpatient=total_outpatient,
|
||||
total_inpatient=total_inpatient,
|
||||
date_range={
|
||||
"start": min_date.strftime("%Y-%m-%d"),
|
||||
"end": max_date.strftime("%Y-%m-%d")
|
||||
},
|
||||
top_districts=top_districts,
|
||||
top_diagnoses=top_diagnoses
|
||||
)
|
||||
|
||||
|
||||
@router.get("/trend", response_model=TrendResponse, summary="获取病例趋势数据")
|
||||
async def get_cases_trend(
|
||||
start_date: Optional[str] = Query(None, description="开始日期 (YYYY-MM-DD)"),
|
||||
end_date: Optional[str] = Query(None, description="结束日期 (YYYY-MM-DD)"),
|
||||
group_by: str = Query("day", description="分组粒度:day, week, month"),
|
||||
):
|
||||
"""
|
||||
获取病例时间趋势数据
|
||||
|
||||
- 支持按日、周、月分组
|
||||
- 可指定日期范围
|
||||
- 返回门诊、住院、总计趋势
|
||||
"""
|
||||
if start_date and not DATE_PATTERN.match(start_date):
|
||||
raise HTTPException(status_code=400, detail="Invalid start_date format. Use YYYY-MM-DD")
|
||||
if end_date and not DATE_PATTERN.match(end_date):
|
||||
raise HTTPException(status_code=400, detail="Invalid end_date format. Use YYYY-MM-DD")
|
||||
|
||||
df = _get_combined_data()
|
||||
|
||||
# 日期过滤
|
||||
if start_date:
|
||||
df = df[df['date'] >= pd.to_datetime(start_date)]
|
||||
if end_date:
|
||||
df = df[df['date'] <= pd.to_datetime(end_date)]
|
||||
|
||||
# 分组
|
||||
if group_by == "week":
|
||||
df['period'] = df['date'].dt.to_period('W').dt.start_time
|
||||
elif group_by == "month":
|
||||
df['period'] = df['date'].dt.to_period('M').dt.start_time
|
||||
else:
|
||||
df['period'] = df['date'].dt.date
|
||||
|
||||
# 聚合
|
||||
out_trend = df[df['type'] == 'outpatient'].groupby('period').size()
|
||||
in_trend = df[df['type'] == 'inpatient'].groupby('period').size()
|
||||
|
||||
periods = sorted(set(out_trend.index.tolist() + in_trend.index.tolist()))
|
||||
|
||||
trend = []
|
||||
total_out = total_in = 0
|
||||
for p in periods:
|
||||
out_count = int(out_trend.get(p, 0))
|
||||
in_count = int(in_trend.get(p, 0))
|
||||
total_out += out_count
|
||||
total_in += in_count
|
||||
trend.append(TrendPoint(
|
||||
date=pd.Timestamp(p).strftime("%Y-%m-%d"),
|
||||
outpatient=out_count,
|
||||
inpatient=in_count,
|
||||
total=out_count + in_count
|
||||
))
|
||||
|
||||
return TrendResponse(
|
||||
trend=trend,
|
||||
summary={
|
||||
"total_outpatient": total_out,
|
||||
"total_inpatient": total_in,
|
||||
"period_count": len(periods),
|
||||
"avg_daily_outpatient": round(total_out / max(len(periods), 1), 2),
|
||||
"avg_daily_inpatient": round(total_in / max(len(periods), 1), 2),
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@router.get("/districts", response_model=DistrictsResponse, summary="获取区域分布数据")
|
||||
async def get_cases_districts(
|
||||
case_type: Optional[str] = Query(None, description="病例类型:outpatient, inpatient, all"),
|
||||
min_count: int = Query(10, description="最小病例数过滤"),
|
||||
):
|
||||
"""
|
||||
获取病例区域分布数据
|
||||
|
||||
- 支持按病例类型筛选
|
||||
- 可设置最小病例数过滤
|
||||
- 返回各区门诊、住院量及占比
|
||||
"""
|
||||
df = _get_combined_data()
|
||||
|
||||
# 类型过滤
|
||||
if case_type == "outpatient":
|
||||
df = df[df['type'] == 'outpatient']
|
||||
elif case_type == "inpatient":
|
||||
df = df[df['type'] == 'inpatient']
|
||||
|
||||
# 过滤未知区域
|
||||
df = df[(df['district'] != '未知') & (df['district'] != '其他')]
|
||||
|
||||
# 聚合
|
||||
district_stats = df.groupby(['district', 'type']).size().unstack(fill_value=0)
|
||||
|
||||
if 'outpatient' not in district_stats.columns:
|
||||
district_stats['outpatient'] = 0
|
||||
if 'inpatient' not in district_stats.columns:
|
||||
district_stats['inpatient'] = 0
|
||||
|
||||
district_stats['total'] = district_stats['outpatient'] + district_stats['inpatient']
|
||||
|
||||
# 过滤
|
||||
district_stats = district_stats[district_stats['total'] >= min_count]
|
||||
district_stats = district_stats.sort_values('total', ascending=False)
|
||||
|
||||
total = int(district_stats['total'].sum())
|
||||
|
||||
districts = []
|
||||
for district, row in district_stats.iterrows():
|
||||
districts.append(DistrictData(
|
||||
district=district,
|
||||
outpatient=int(row['outpatient']),
|
||||
inpatient=int(row['inpatient']),
|
||||
total=int(row['total']),
|
||||
outpatient_ratio=round(row['outpatient'] / row['total'] * 100, 2) if row['total'] > 0 else 0,
|
||||
inpatient_ratio=round(row['inpatient'] / row['total'] * 100, 2) if row['total'] > 0 else 0
|
||||
))
|
||||
|
||||
return DistrictsResponse(districts=districts, total=total)
|
||||
|
||||
|
||||
@router.get("/realtime", response_model=RealtimeData, summary="获取实时数据")
|
||||
async def get_cases_realtime():
|
||||
"""
|
||||
获取实时病例数据
|
||||
|
||||
- 今日就诊量
|
||||
- 近 7 日平均值
|
||||
- 变化率
|
||||
- 状态评估 (正常/偏高/偏低)
|
||||
"""
|
||||
df = _get_combined_data()
|
||||
|
||||
today = pd.Timestamp.today().normalize()
|
||||
last_7d = today - pd.Timedelta(days=7)
|
||||
|
||||
# 今日数据
|
||||
today_data = df[df['date'] >= today]
|
||||
today_total = len(today_data)
|
||||
today_out = len(today_data[today_data['type'] == 'outpatient'])
|
||||
today_in = len(today_data[today_data['type'] == 'inpatient'])
|
||||
|
||||
# 近 7 日平均
|
||||
last_7d_data = df[(df['date'] >= last_7d) & (df['date'] < today)]
|
||||
last_7d_avg = round(len(last_7d_data) / 7, 2) if len(last_7d_data) > 0 else 0
|
||||
|
||||
# 变化率
|
||||
if last_7d_avg > 0:
|
||||
change_ratio = round((today_total - last_7d_avg) / last_7d_avg * 100, 2)
|
||||
else:
|
||||
change_ratio = 0.0
|
||||
|
||||
# 状态评估
|
||||
if change_ratio > 20:
|
||||
status = "偏高"
|
||||
elif change_ratio < -20:
|
||||
status = "偏低"
|
||||
else:
|
||||
status = "正常"
|
||||
|
||||
return RealtimeData(
|
||||
today_outpatient=today_out,
|
||||
today_inpatient=today_in,
|
||||
today_total=today_total,
|
||||
last_7d_avg=last_7d_avg,
|
||||
change_ratio=change_ratio,
|
||||
status=status
|
||||
)
|
||||
Reference in New Issue
Block a user