fix: 数据库与数据管线 6 个 P1 + 5 个 P2 审查问题修复

P1-1 [context_builder] build_context 共享 session,切片函数传 db 参数,
        回测 20 场并发连接需求从 100+ 降至每场 1 个
P1-2 [bzzoiro] 预加载改为按 raw_events 日期范围 ±30 天按需加载
P1-3 [understat] 批量查询球队 + 比赛,从 1140 次往返降到 3 次
P1-4 [injuries] 批量幂等检查 + 分批 flush,IntegrityError 逐条回退
P1-5 [predict] 删除 threading.Lock,dict 操作原子无需同步锁
P1-6 [models] 添加 (match_id, provider, model) 唯一约束 + 迁移

P2-1 [unit_of_work] get_uow 返回类型改为 AsyncIterator[AsyncSession]
P2-2 [normalize] _parse_date 失败时记录 warning 避免静默丢数据
P2-3 [repositories] find_by_teams_and_date 改用 match_date_date 等值匹配
P2-4 [migration] 幽灵列 cutoff_at 已在 0006 迁移删除(已有)
P2-5 [migration] injuries 约束命名对齐 ORM,UniqueConstraint → 唯一索引
This commit is contained in:
shangfangjian
2026-09-16 03:09:39 +08:00
parent 983b620659
commit ff0045ad93
11 changed files with 458 additions and 123 deletions
+57 -11
View File
@@ -10,9 +10,10 @@ import json
import logging
import random
import re
from datetime import datetime, timezone
from datetime import datetime, timedelta, timezone
from sqlalchemy import func, select
from sqlalchemy import select
from sqlalchemy.orm import selectinload
from src.core.http_client import get_client
from src.data.config import FDCO_TO_UNDERSTAT, LEAGUE_NAMES
@@ -76,6 +77,18 @@ async def fetch_understat(league_code: str, season: int) -> list[dict]:
return data
def _match_key(home_team_id: int, away_team_id: int, match_date) -> tuple[int, int, str]:
"""比赛去重键:(主队, 客队, 天级日期 ISO 字符串)。
统一在这里构造,避免"预加载时用 str(date)、写入时用 isoformat()"这类
隐式格式依赖 —— 两者当前恰好相等,但一旦有人改动其一就会静默失配,
导致所有比赛被判为不存在而重复插入。
"""
if hasattr(match_date, "date") and callable(match_date.date):
match_date = match_date.date()
return (home_team_id, away_team_id, match_date.isoformat() if match_date is not None else "")
@register
class UnderstatSource:
"""understat xG 数据源(实现 DataSource 协议)。"""
@@ -86,8 +99,10 @@ class UnderstatSource:
"""采集 understat xG → 回填到现有 Match。只回填 xG 字段,不创建新 Match。
注意: 本方法不控制事务(commit/rollback),由调用方通过 UnitOfWork 控制。
P1-3: 批量查询优化,将单赛季 380 场 × 3 次 DB 往返降为 3 次查询。
"""
from src.db.repositories import LeagueRepository, MatchRepository, TeamRepository
from src.db.repositories import LeagueRepository, TeamRepository
result = {"updated": 0, "skipped": 0, "unmatched": 0, "errors": []}
@@ -101,7 +116,6 @@ class UnderstatSource:
# 使用 Repository
league_repo = LeagueRepository(db)
team_repo = TeamRepository(db)
match_repo = MatchRepository(db)
# 查联赛
league_obj = await league_repo.get_by_code(league)
@@ -109,6 +123,9 @@ class UnderstatSource:
result["errors"].append(f"league {league} not found in DB")
return result
# === 批量优化: 一次规范化,收集球队名和日期 ===
normalized_matches: list = []
all_team_names: set[str] = set()
for raw in raw_matches:
if not raw.get("isResult"):
continue
@@ -120,17 +137,46 @@ class UnderstatSource:
except Exception as e:
result["errors"].append(f"normalize: {e}")
continue
normalized_matches.append((nm, raw))
all_team_names.add(nm.home_team)
all_team_names.add(nm.away_team)
# 匹配已有 Match(天级) - 使用 Repository
home_team = await team_repo.get_by_name(nm.home_team)
away_team = await team_repo.get_by_name(nm.away_team)
if home_team is None or away_team is None:
if not normalized_matches:
return result
# === 批量查询球队(1 次 DB 往返) ===
team_name_to_id = {}
if all_team_names:
teams = await team_repo.get_all_by_names(list(all_team_names))
team_name_to_id = {name: team.id for name, team in teams.items()}
# === 批量查询已有比赛(1 次 DB 往返,按日期范围) ===
match_dict: dict[tuple, Match] = {}
dates = [nm.date for nm, _ in normalized_matches if nm.date is not None]
if dates:
min_dt = min(dates) - timedelta(days=30)
max_dt = max(dates) + timedelta(days=30)
stmt = (
select(Match)
.options(selectinload(Match.stats))
.where(Match.league_id == league_obj.id)
.where(Match.match_date >= min_dt)
.where(Match.match_date <= max_dt)
)
for m in (await db.execute(stmt)).scalars():
key = _match_key(m.home_team_id, m.away_team_id, m.match_date_date)
match_dict[key] = m
# === 内存匹配 + 回填 xG ===
for nm, raw in normalized_matches:
home_team_id = team_name_to_id.get(nm.home_team)
away_team_id = team_name_to_id.get(nm.away_team)
if home_team_id is None or away_team_id is None:
result["unmatched"] += 1
continue
existing = await match_repo.find_by_teams_and_date(
league_obj.id, home_team.id, away_team.id, nm.date
)
match_key = _match_key(home_team_id, away_team_id, nm.date)
existing = match_dict.get(match_key)
if existing is None:
result["unmatched"] += 1
continue