fix: 数据库与数据管线 6 个 P1 + 5 个 P2 审查问题修复
P1-1 [context_builder] build_context 共享 session,切片函数传 db 参数,
回测 20 场并发连接需求从 100+ 降至每场 1 个
P1-2 [bzzoiro] 预加载改为按 raw_events 日期范围 ±30 天按需加载
P1-3 [understat] 批量查询球队 + 比赛,从 1140 次往返降到 3 次
P1-4 [injuries] 批量幂等检查 + 分批 flush,IntegrityError 逐条回退
P1-5 [predict] 删除 threading.Lock,dict 操作原子无需同步锁
P1-6 [models] 添加 (match_id, provider, model) 唯一约束 + 迁移
P2-1 [unit_of_work] get_uow 返回类型改为 AsyncIterator[AsyncSession]
P2-2 [normalize] _parse_date 失败时记录 warning 避免静默丢数据
P2-3 [repositories] find_by_teams_and_date 改用 match_date_date 等值匹配
P2-4 [migration] 幽灵列 cutoff_at 已在 0006 迁移删除(已有)
P2-5 [migration] injuries 约束命名对齐 ORM,UniqueConstraint → 唯一索引
This commit is contained in:
+57
-11
@@ -10,9 +10,10 @@ import json
|
||||
import logging
|
||||
import random
|
||||
import re
|
||||
from datetime import datetime, timezone
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
from sqlalchemy import func, select
|
||||
from sqlalchemy import select
|
||||
from sqlalchemy.orm import selectinload
|
||||
|
||||
from src.core.http_client import get_client
|
||||
from src.data.config import FDCO_TO_UNDERSTAT, LEAGUE_NAMES
|
||||
@@ -76,6 +77,18 @@ async def fetch_understat(league_code: str, season: int) -> list[dict]:
|
||||
return data
|
||||
|
||||
|
||||
def _match_key(home_team_id: int, away_team_id: int, match_date) -> tuple[int, int, str]:
|
||||
"""比赛去重键:(主队, 客队, 天级日期 ISO 字符串)。
|
||||
|
||||
统一在这里构造,避免"预加载时用 str(date)、写入时用 isoformat()"这类
|
||||
隐式格式依赖 —— 两者当前恰好相等,但一旦有人改动其一就会静默失配,
|
||||
导致所有比赛被判为不存在而重复插入。
|
||||
"""
|
||||
if hasattr(match_date, "date") and callable(match_date.date):
|
||||
match_date = match_date.date()
|
||||
return (home_team_id, away_team_id, match_date.isoformat() if match_date is not None else "")
|
||||
|
||||
|
||||
@register
|
||||
class UnderstatSource:
|
||||
"""understat xG 数据源(实现 DataSource 协议)。"""
|
||||
@@ -86,8 +99,10 @@ class UnderstatSource:
|
||||
"""采集 understat xG → 回填到现有 Match。只回填 xG 字段,不创建新 Match。
|
||||
|
||||
注意: 本方法不控制事务(commit/rollback),由调用方通过 UnitOfWork 控制。
|
||||
|
||||
P1-3: 批量查询优化,将单赛季 380 场 × 3 次 DB 往返降为 3 次查询。
|
||||
"""
|
||||
from src.db.repositories import LeagueRepository, MatchRepository, TeamRepository
|
||||
from src.db.repositories import LeagueRepository, TeamRepository
|
||||
|
||||
result = {"updated": 0, "skipped": 0, "unmatched": 0, "errors": []}
|
||||
|
||||
@@ -101,7 +116,6 @@ class UnderstatSource:
|
||||
# 使用 Repository
|
||||
league_repo = LeagueRepository(db)
|
||||
team_repo = TeamRepository(db)
|
||||
match_repo = MatchRepository(db)
|
||||
|
||||
# 查联赛
|
||||
league_obj = await league_repo.get_by_code(league)
|
||||
@@ -109,6 +123,9 @@ class UnderstatSource:
|
||||
result["errors"].append(f"league {league} not found in DB")
|
||||
return result
|
||||
|
||||
# === 批量优化: 一次规范化,收集球队名和日期 ===
|
||||
normalized_matches: list = []
|
||||
all_team_names: set[str] = set()
|
||||
for raw in raw_matches:
|
||||
if not raw.get("isResult"):
|
||||
continue
|
||||
@@ -120,17 +137,46 @@ class UnderstatSource:
|
||||
except Exception as e:
|
||||
result["errors"].append(f"normalize: {e}")
|
||||
continue
|
||||
normalized_matches.append((nm, raw))
|
||||
all_team_names.add(nm.home_team)
|
||||
all_team_names.add(nm.away_team)
|
||||
|
||||
# 匹配已有 Match(天级) - 使用 Repository
|
||||
home_team = await team_repo.get_by_name(nm.home_team)
|
||||
away_team = await team_repo.get_by_name(nm.away_team)
|
||||
if home_team is None or away_team is None:
|
||||
if not normalized_matches:
|
||||
return result
|
||||
|
||||
# === 批量查询球队(1 次 DB 往返) ===
|
||||
team_name_to_id = {}
|
||||
if all_team_names:
|
||||
teams = await team_repo.get_all_by_names(list(all_team_names))
|
||||
team_name_to_id = {name: team.id for name, team in teams.items()}
|
||||
|
||||
# === 批量查询已有比赛(1 次 DB 往返,按日期范围) ===
|
||||
match_dict: dict[tuple, Match] = {}
|
||||
dates = [nm.date for nm, _ in normalized_matches if nm.date is not None]
|
||||
if dates:
|
||||
min_dt = min(dates) - timedelta(days=30)
|
||||
max_dt = max(dates) + timedelta(days=30)
|
||||
stmt = (
|
||||
select(Match)
|
||||
.options(selectinload(Match.stats))
|
||||
.where(Match.league_id == league_obj.id)
|
||||
.where(Match.match_date >= min_dt)
|
||||
.where(Match.match_date <= max_dt)
|
||||
)
|
||||
for m in (await db.execute(stmt)).scalars():
|
||||
key = _match_key(m.home_team_id, m.away_team_id, m.match_date_date)
|
||||
match_dict[key] = m
|
||||
|
||||
# === 内存匹配 + 回填 xG ===
|
||||
for nm, raw in normalized_matches:
|
||||
home_team_id = team_name_to_id.get(nm.home_team)
|
||||
away_team_id = team_name_to_id.get(nm.away_team)
|
||||
if home_team_id is None or away_team_id is None:
|
||||
result["unmatched"] += 1
|
||||
continue
|
||||
|
||||
existing = await match_repo.find_by_teams_and_date(
|
||||
league_obj.id, home_team.id, away_team.id, nm.date
|
||||
)
|
||||
match_key = _match_key(home_team_id, away_team_id, nm.date)
|
||||
existing = match_dict.get(match_key)
|
||||
if existing is None:
|
||||
result["unmatched"] += 1
|
||||
continue
|
||||
|
||||
Reference in New Issue
Block a user