"""一次性跑完所有 sync task(按依赖顺序)。 按 sort_order 触发:stock_basic → kline_daily → kline_index → kline_5min → industry_sector → sector_features → share_snapshot → market_regime 每个 task 跑完写日志 + 把 result 落到 logs/runall_summary.json。 注:以下 task **不**在 runall 链中 —— 各自有独立的 systemd timer: - tick_trade : mairui 21:00 发布 → market-sync-tick.timer (21:05) - moneyflow : mairui 21:30 发布 → market-sync-moneyflow.timer (21:35) 2026-07-08 教训: 之前 runall 是顺序串行,任一 task 永久卡住(kline_daily 在 5200/5204 卡死 4h)→ 后面 5 个 task 全部没机会跑,runall 进程被 systemd 4h 超时杀, dataset_registry 卡在 running 状态无人清理。修复: 1. 启动时调 recover_interrupted_dataset_registry() — 清上次被中断的卡死 2. 每个 task 用 multiprocessing.Process 跑 + 硬上限 PER_TASK_TIMEOUT_SEC, 超时直接 kill 子进程,继续下一个 task (不卡整条 runall) 3. 数据源/任务层也加了 timeout(xueqiu 20s + kline_daily as_completed 30s), 这里是最后一道防线 """ from __future__ import annotations import json import multiprocessing as mp import os import signal import sys import time from pathlib import Path from typing import Any _PROJECT_ROOT = Path(__file__).resolve().parent.parent if str(_PROJECT_ROOT) not in sys.path: sys.path.insert(0, str(_PROJECT_ROOT)) from app.core.utils.logging import setup_logging, get_logger setup_logging() logger = get_logger("runall") from app.core.datasource.registry import build_default_registry from app.core.sync.registry import seed_sync_registry build_default_registry() seed_sync_registry() from app.core.db import ops as db_ops from app.tasks import get_task # 启动时清理上一轮被中断的卡死状态(2026-07-08 教训) # 注意: 这里清的是"上次 runall 被杀时的 running",如果当前 runall 自己卡死, # 子进程被 kill 后 recover 仍要再调一次 — 见末尾的 finally。 try: n_recovered = db_ops.recover_interrupted_dataset_registry() if n_recovered: logger.warning("[runall] 启动时清掉 %d 个上轮卡死的 running 任务", n_recovered) except Exception as e: logger.warning("[runall] 启动清理失败(非致命): %s", e) # 顺序执行,按 sort_order 走(便于排查 + 避免外部 API 限流) # 注:tick_trade / moneyflow 由独立 timer 跑(21:05 / 21:35) TASKS = [ "stock_basic", "kline_index", "kline_daily", "kline_5min", "market_regime", ] # 不跳任何 task — 全量跑 SKIP: set[str] = set() # 每个 task 的硬上限(秒)。覆盖数据源/任务层都失败的最坏情况: # kline_daily: 35 min (正常 18 min, 留余量) # kline_5min: 75 min (全量首次跑很久, 增量 ~30 min) # share_snapshot: 15 min # 其他小表: 10 min PER_TASK_TIMEOUT_SEC = { "kline_daily": 35 * 60, "kline_5min": 75 * 60, "share_snapshot": 15 * 60, "market_regime": 5 * 60, "stock_basic": 15 * 60, "kline_index": 5 * 60, } DEFAULT_TASK_TIMEOUT = 20 * 60 def _run_task_in_subprocess(tid: str, result_queue: mp.Queue) -> None: """子进程入口: 跑 task.run() 并把 result 通过 queue 返回。 单独进程是为了父进程能用 SIGKILL 干掉它(timeout 时)而不污染主 runall 状态。 """ try: task = get_task(tid) kwargs = {"max_workers": 5} r = task.run(trigger_source="runall", **kwargs) result_queue.put(r) except Exception as e: result_queue.put({"status": "error", "message": f"{type(e).__name__}: {e}"}) results: dict[str, Any] = {} t_all = time.time() # 注意: 主执行块必须在 __name__ == '__main__' 内部,否则 multiprocessing spawn # 子进程重新 import 本模块时会再次执行进程启动逻辑,导致 RuntimeError: # "An attempt has been made to start a new process before the current process # has finished its bootstrapping phase." (2026-07-13 15:30 runall 全部失败) def _run_tasks(): global results for tid in TASKS: if tid in SKIP: results[tid] = {"status": "skipped", "message": "已 ok"} logger.info(f"[runall] {tid} 跳过(已 ok)") continue timeout_sec = PER_TASK_TIMEOUT_SEC.get(tid, DEFAULT_TASK_TIMEOUT) logger.info(f"[runall] >>> 开始 {tid} (hard timeout {timeout_sec}s)") t0 = time.time() # 子进程跑 task,父进程用 .join(timeout) 守门 ctx = mp.get_context("spawn") result_queue: mp.Queue = ctx.Queue() proc = ctx.Process(target=_run_task_in_subprocess, args=(tid, result_queue), name=f"runall-{tid}") proc.start() proc.join(timeout=timeout_sec) elapsed = round(time.time() - t0, 1) if proc.is_alive(): logger.error( f"[runall] !!! {tid} 超时 (>{timeout_sec}s, elapsed={elapsed}s), 强制 kill" ) proc.terminate() proc.join(5) if proc.is_alive(): proc.kill() proc.join(2) results[tid] = { "status": "error", "message": f"runall timeout (>={timeout_sec}s), killed", "elapsed_sec": elapsed, } try: db_ops.recover_interrupted_dataset_registry() except Exception as e: logger.warning("[runall] 超时后清理 dataset_registry 失败: %s", e) else: try: r = result_queue.get(timeout=5) except Exception as e: r = {"status": "error", "message": f"无法获取子进程结果: {e}"} r["elapsed_sec"] = elapsed results[tid] = r logger.info( f"[runall] <<< {tid} 完成 status={r.get('status')} elapsed={elapsed}s msg={r.get('message')}" ) time.sleep(5) if __name__ == '__main__': _run_tasks() # 全部完成清一次 stuck 状态 try: db_ops.recover_interrupted_dataset_registry() except Exception as e: logger.warning("[runall] 末尾清理失败(非致命): %s", e) elapsed_total = round(time.time() - t_all, 1) summary: dict[str, Any] = {"total_elapsed_sec": elapsed_total, "tasks": results} out = _PROJECT_ROOT / "logs" / "runall_summary.json" out.write_text(json.dumps(summary, ensure_ascii=False, indent=2)) logger.info(f"[runall] 全部完成 total={elapsed_total}s summary={out}") print(json.dumps(summary, ensure_ascii=False, indent=2))