构建状态增加平台终态只读核对,修复回调丢失后任务长期卡在构建中

工作流在脚本收尾前失败(拉取/编译/runner 异常)或 GENURL 不可达时,
结束回调会永久丢失,而等待页 30 秒轮询与列表页此前只读本地库,
任务只能等 6 小时本地超时,平台侧已失败也无法及时反映。

- GitHub/Gitea 后端新增只读 get_run(),仅在平台报告终态时返回结论,
  在途、非 200、网络异常、坏 JSON 一律返回 None;不取消、不写平台
- 等待页 30 秒轮询同步核对一次真实状态:20 秒节流、6 秒超时、异常静默
- 我的构建/后台仪表盘改为后台守护线程核对,页面渲染零等待
- 仅做非终态到终态的条件更新,回调抢先到达时不会被覆盖
- 停止/删除纯本地语义、6 小时本地超时兜底等现有逻辑保持不变
This commit is contained in:
naeeo
2026-09-30 12:46:05 +08:00
parent b0219603be
commit d78c071818
5 changed files with 203 additions and 12 deletions
+91 -2
View File
@@ -2,6 +2,8 @@ import io
import json
import os
import re
import threading
import time
import uuid
from datetime import timedelta
from pathlib import Path
@@ -33,8 +35,93 @@ from .models import GithubRun, STATUS_LABELS, STATUS_BADGE
# 终态状态集合
TERMINAL_STATUSES = ('success', 'failure', 'cancelled', 'timed_out', 'skipped')
# 正常构建约 30~45 分钟;超过该时长仍在进行中的,视为 runner 离线/作业僵死。
# 状态以构建脚本结束回调为准,页面不再向构建平台主动查询;超时兜底只做本地判断。
# 状态以构建脚本结束回调为第一来源;回调丢失时由 get_run 只读核对平台终态兜底,
# 超时判断只做本地兜底。
STALE_BUILD_HOURS = 6
# 主动核对节流:同一在途任务两次平台查询的最小间隔,避免轮询/刷页面打满 API
REFRESH_THROTTLE_SECONDS = 20
# 平台查询短超时,平台抖动时等待页轮询最多多等这几秒,不拖垮页面
REFRESH_TIMEOUT = 6
_refresh_stamps = {}
_refresh_lock = threading.Lock()
def _claim_refresh_slot(run_pk):
"""节流:到间隔才占位成功;先占位再请求,避免并发页面/线程重复打平台。"""
now = time.monotonic()
with _refresh_lock:
if now - _refresh_stamps.get(run_pk, 0) >= REFRESH_THROTTLE_SECONDS:
_refresh_stamps[run_pk] = now
return True
return False
def _apply_remote_terminal(gh_run, info):
"""平台明确终态且本地仍在途时落库(条件更新,避免覆盖刚到达的回调)。返回是否变化。"""
if info is None or not info.finished:
return False
new_status = info.status
if new_status not in TERMINAL_STATUSES or new_status == gh_run.status:
return False
updated = GithubRun.objects.filter(pk=gh_run.pk) \
.exclude(status__in=TERMINAL_STATUSES) \
.update(status=new_status, updated_at=timezone.now())
if updated:
gh_run.status = new_status
return bool(updated)
def refresh_active_run(gh_run):
"""同步向构建平台核对单个在途任务(等待页 30 秒轮询 / 状态接口用)。
20 秒节流、6 秒超时;任何网络异常静默。仅当平台报告终态时落库,
工作流结束回调仍是第一状态来源,本核对只兜底回调丢失。
"""
if gh_run.is_finished() or not gh_run.github_run_id:
return False
if not _claim_refresh_slot(gh_run.pk):
return False
try:
backend = build_backends.get_backend(gh_run.backend or 'github')
info = backend.get_run(gh_run.github_run_id, timeout=REFRESH_TIMEOUT)
except Exception as e:
print(f"核对构建状态出错:{e}")
return False
return _apply_remote_terminal(gh_run, info)
def _refresh_runs_in_background(run_pks):
"""后台线程体:逐个核对终态;与请求/渲染完全解耦,异常不影响任何页面。"""
try:
for pk in run_pks:
gh_run = GithubRun.objects.filter(pk=pk).first()
if gh_run is None or gh_run.is_finished() or not gh_run.github_run_id:
continue
try:
backend = build_backends.get_backend(gh_run.backend or 'github')
info = backend.get_run(gh_run.github_run_id, timeout=REFRESH_TIMEOUT)
_apply_remote_terminal(gh_run, info)
except Exception as e:
print(f"后台核对构建状态出错:{e}")
finally:
# 子线程使用独立数据库连接,结束时必须关闭
from django.db import connections
connections.close_all()
def schedule_refresh_for_runs(gh_runs):
"""列表/仪表盘用:为在途任务安排后台核对,页面渲染零等待。
与等待页共用 20 秒节流;无在途任务或刚核对过时不创建线程。
"""
due_pks = [
r.pk for r in gh_runs
if not r.is_finished() and r.github_run_id and _claim_refresh_slot(r.pk)
]
if not due_pks:
return
threading.Thread(
target=_refresh_runs_in_background, args=(due_pks,), daemon=True).start()
def mark_stale_builds(gh_runs=None):
@@ -422,7 +509,9 @@ def _get_run_status(uuid_val):
backend = build_backends.get_backend(gh_run.backend or 'github')
github_log_url = backend.log_url(gh_run.github_run_id) if gh_run.github_run_id else ''
# 状态以构建脚本结束回调为准;这里只做纯本地的 6 小时超时兜底,不请求构建平台
# 状态以构建脚本结束回调为准;此处先做平台终态只读核对(20 秒节流、6 秒超时,
# 回调丢失时能及时发现平台侧已失败/取消),再做纯本地的 6 小时超时兜底
refresh_active_run(gh_run)
mark_run_stale_if_needed(gh_run)
return {