fix: dispatch detect start to mainland nodes

This commit is contained in:
Your Name
2026-04-19 02:18:42 +08:00
parent b9c29481b5
commit 1adfeed1ab
6 changed files with 297 additions and 5 deletions

View File

@@ -15,6 +15,7 @@ from app.services.detect_job_service import (
)
from app.services.detect_service import get_detect_status
from app.services.detect_run_service import create_detect_run_snapshot, finalize_detect_run, mark_detect_run_stopping
from app.services.ops_job_service import create_ops_job, list_managed_nodes
from app.services.settings_service import get_settings_payload, resolve_thread_count
from app.services.worker_control_service import send_worker_command, start_worker
@@ -68,6 +69,172 @@ def _build_settings_summary(settings_payload: dict) -> dict:
}
def _mainland_detect_targets() -> dict[str, list[dict]]:
controllers: list[dict] = []
workers: list[dict] = []
for node in list_managed_nodes():
if not bool(node.get("is_enabled", True)):
continue
if str(node.get("region") or "").strip() != "mainland":
continue
if not str(node.get("last_seen_at") or "").strip():
continue
role = str(node.get("role") or "").strip()
if role == "control":
controllers.append(node)
elif role == "worker":
workers.append(node)
return {"controllers": controllers, "workers": workers}
def _queue_remote_detect_job(
*,
node_code: str,
action: str,
job_summary: dict,
cycle_token: str,
requested_by: str = "api",
payload: dict | None = None,
) -> dict:
ok, message, data = create_ops_job(
{
"action": action,
"target_node_code": node_code,
"execution_mode": "remote-agent",
"requested_by": requested_by,
"auto_approve": True,
"run_now": False,
"payload": {
"job_id": int(job_summary.get("job_id") or 0),
"job_code": str(job_summary.get("job_code") or "").strip(),
"cycle_token": cycle_token,
**dict(payload or {}),
},
"metadata": {
"source": "detect.start",
"job_id": int(job_summary.get("job_id") or 0),
"job_code": str(job_summary.get("job_code") or "").strip(),
"cycle_token": cycle_token,
},
}
)
return {
"node_code": node_code,
"action": action,
"ok": ok,
"message": message,
"job": dict((data or {}).get("job") or {}),
}
def _dispatch_remote_detect_start(*, job_summary: dict, cycle_token: str) -> dict:
targets = _mainland_detect_targets()
queued: list[dict] = []
for node in targets["controllers"]:
node_code = str(node.get("node_code") or "").strip()
if not node_code:
continue
queued.append(
_queue_remote_detect_job(
node_code=node_code,
action="runtime.start_sync_agent",
job_summary=job_summary,
cycle_token=cycle_token,
)
)
queued.append(
_queue_remote_detect_job(
node_code=node_code,
action="runtime.pull_tasks",
job_summary=job_summary,
cycle_token=cycle_token,
payload={"limit": int(job_summary.get("items_pending") or job_summary.get("items_total") or 0) or 1000},
)
)
queued.append(
_queue_remote_detect_job(
node_code=node_code,
action="runtime.start_worker",
job_summary=job_summary,
cycle_token=cycle_token,
)
)
queued.append(
_queue_remote_detect_job(
node_code=node_code,
action="runtime.start_detection",
job_summary=job_summary,
cycle_token=cycle_token,
)
)
for node in targets["workers"]:
node_code = str(node.get("node_code") or "").strip()
if not node_code:
continue
queued.append(
_queue_remote_detect_job(
node_code=node_code,
action="runtime.start_worker",
job_summary=job_summary,
cycle_token=cycle_token,
)
)
queued.append(
_queue_remote_detect_job(
node_code=node_code,
action="runtime.start_detection",
job_summary=job_summary,
cycle_token=cycle_token,
)
)
success_jobs = [item for item in queued if item.get("ok")]
failed_jobs = [item for item in queued if not item.get("ok")]
return {
"target_summary": {
"controller_nodes": [str(item.get("node_code") or "") for item in targets["controllers"]],
"worker_nodes": [str(item.get("node_code") or "") for item in targets["workers"]],
},
"queued_jobs": queued,
"queued_total": len(success_jobs),
"failed_total": len(failed_jobs),
}
def _dispatch_remote_detect_stop(*, active_job: dict | None, cycle_token: str = "") -> dict:
targets = _mainland_detect_targets()
job_summary = active_job or {}
queued: list[dict] = []
for node in [*targets["controllers"], *targets["workers"]]:
node_code = str(node.get("node_code") or "").strip()
if not node_code:
continue
queued.append(
_queue_remote_detect_job(
node_code=node_code,
action="runtime.stop_detection",
job_summary=job_summary,
cycle_token=cycle_token,
payload={},
)
)
success_jobs = [item for item in queued if item.get("ok")]
failed_jobs = [item for item in queued if not item.get("ok")]
return {
"target_summary": {
"controller_nodes": [str(item.get("node_code") or "") for item in targets["controllers"]],
"worker_nodes": [str(item.get("node_code") or "") for item in targets["workers"]],
},
"queued_jobs": queued,
"queued_total": len(success_jobs),
"failed_total": len(failed_jobs),
}
@router.get("/detect/status", response_model=ApiResponse)
def detect_status() -> ApiResponse:
return ApiResponse(data=get_detect_status())
@@ -165,6 +332,7 @@ def start_detect() -> ApiResponse:
settings_payload = get_settings_payload()
settings_summary = _build_settings_summary(settings_payload)
if command_ok:
remote_dispatch = _dispatch_remote_detect_start(job_summary=job_summary, cycle_token=cycle_token)
create_detect_run_snapshot(
message=f"{message}{command_message}",
runtime={
@@ -177,12 +345,28 @@ def start_detect() -> ApiResponse:
progress=snapshot.get("progress", {}),
settings_summary=settings_summary,
)
append_detect_job_event(
job_summary["job_id"],
event_type="job_dispatch_remote_queued",
level="info" if int(remote_dispatch.get("failed_total", 0) or 0) == 0 else "warning",
message=(
f"已向大陆节点排队 {int(remote_dispatch.get('queued_total', 0) or 0)} 个远端检测动作"
if int(remote_dispatch.get("queued_total", 0) or 0) > 0
else "当前没有可排队的大陆远端检测动作"
),
payload={
"cycle_token": cycle_token,
"remote_dispatch": remote_dispatch,
},
)
else:
remote_dispatch = {"queued_jobs": [], "queued_total": 0, "failed_total": 0, "target_summary": {"controller_nodes": [], "worker_nodes": []}}
response_message = f"{message}{command_message}" if command_ok else command_message
result = _build_detect_action_result(
action="start",
ok=command_ok,
message=response_message,
data={"job": job_summary},
data={"job": job_summary, "remote_dispatch": remote_dispatch},
)
return ApiResponse(code=0 if command_ok else 1, message=response_message, data=result)
@@ -191,6 +375,8 @@ def start_detect() -> ApiResponse:
def stop_detect() -> ApiResponse:
active_job = get_active_detect_job_summary(event_limit=10)
ok, message = send_worker_command("stop_detection")
cycle_token = str((active_job or {}).get("current_cycle_token") or "").strip()
remote_dispatch = _dispatch_remote_detect_stop(active_job=active_job, cycle_token=cycle_token)
if active_job:
append_detect_job_event(
active_job["job_id"],
@@ -199,6 +385,17 @@ def stop_detect() -> ApiResponse:
message=message,
payload={"cycle_token": active_job.get("current_cycle_token", "")},
)
append_detect_job_event(
active_job["job_id"],
event_type="job_stop_remote_queued",
level="info" if int(remote_dispatch.get("failed_total", 0) or 0) == 0 else "warning",
message=(
f"已向大陆节点排队 {int(remote_dispatch.get('queued_total', 0) or 0)} 个停止检测动作"
if int(remote_dispatch.get("queued_total", 0) or 0) > 0
else "当前没有可排队的大陆停止检测动作"
),
payload={"cycle_token": cycle_token, "remote_dispatch": remote_dispatch},
)
snapshot = get_detect_status()
settings_payload = get_settings_payload()
settings_summary = _build_settings_summary(settings_payload)
@@ -228,5 +425,5 @@ def stop_detect() -> ApiResponse:
settings_summary=settings_summary,
active_job=active_job,
)
result = _build_detect_action_result(action="stop", ok=ok, message=message)
result = _build_detect_action_result(action="stop", ok=ok, message=message, data={"remote_dispatch": remote_dispatch})
return ApiResponse(code=0 if ok else 1, message=message, data=result)