from __future__ import annotations import json import socket import threading from datetime import datetime, timedelta from psycopg2 import errors from app.core.config import settings from app.core.db import db_read_retry, get_db _RUNTIME_SCHEMA_SQL = """ CREATE TABLE IF NOT EXISTS detect_worker_nodes ( node_code VARCHAR(64) PRIMARY KEY, region VARCHAR(32) NOT NULL DEFAULT 'unknown', role VARCHAR(32) NOT NULL DEFAULT 'worker', hostname VARCHAR(255) NOT NULL DEFAULT '', ip VARCHAR(64) NOT NULL DEFAULT '', status VARCHAR(32) NOT NULL DEFAULT 'unknown', worker_version VARCHAR(32) NOT NULL DEFAULT '', current_load INTEGER NOT NULL DEFAULT 0, metadata_json JSONB, last_heartbeat_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, create_time TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, update_time TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP ); CREATE TABLE IF NOT EXISTS detect_jobs ( id BIGSERIAL PRIMARY KEY, job_code VARCHAR(64) NOT NULL UNIQUE, source VARCHAR(64) NOT NULL DEFAULT 'manual', plan_hash VARCHAR(128) NOT NULL DEFAULT '', task_mode VARCHAR(32) NOT NULL DEFAULT 'domain_pipeline', step_code VARCHAR(64) NOT NULL DEFAULT '', status VARCHAR(32) NOT NULL DEFAULT 'pending', remark TEXT NOT NULL DEFAULT '', created_by VARCHAR(64) NOT NULL DEFAULT '', created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, started_at TIMESTAMP, finished_at TIMESTAMP ); CREATE TABLE IF NOT EXISTS detect_job_items ( id BIGSERIAL PRIMARY KEY, job_id BIGINT NOT NULL REFERENCES detect_jobs(id) ON DELETE CASCADE, domain_id BIGINT NOT NULL, step_code VARCHAR(64) NOT NULL DEFAULT '', status VARCHAR(32) NOT NULL DEFAULT 'pending', claimed_by VARCHAR(64) NOT NULL DEFAULT '', claim_token VARCHAR(64) NOT NULL DEFAULT '', lease_expires_at TIMESTAMP, attempt_count INTEGER NOT NULL DEFAULT 0, last_error TEXT NOT NULL DEFAULT '', result_version VARCHAR(64) NOT NULL DEFAULT '', step_payload_json JSONB, result_payload_json JSONB, started_at TIMESTAMP, finished_at TIMESTAMP, updated_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, create_time TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP ); CREATE INDEX IF NOT EXISTS idx_detect_job_items_status_lease ON detect_job_items(status, lease_expires_at); CREATE INDEX IF NOT EXISTS idx_detect_job_items_claim_ready ON detect_job_items(status, create_time, id) WHERE step_code <> ''; CREATE INDEX IF NOT EXISTS idx_detect_job_items_claim_job_ready ON detect_job_items(job_id, status, create_time, id) WHERE step_code <> ''; CREATE INDEX IF NOT EXISTS idx_detect_job_items_claim_step_ready ON detect_job_items(status, step_code, lease_expires_at, create_time, id) WHERE step_code <> '' AND status IN ('pending', 'failed'); CREATE INDEX IF NOT EXISTS idx_detect_job_items_claim_job_step_ready ON detect_job_items(job_id, status, step_code, lease_expires_at, create_time, id) WHERE step_code <> '' AND status IN ('pending', 'failed'); ALTER TABLE detect_jobs ADD COLUMN IF NOT EXISTS task_mode VARCHAR(32) NOT NULL DEFAULT 'domain_pipeline', ADD COLUMN IF NOT EXISTS step_code VARCHAR(64) NOT NULL DEFAULT ''; ALTER TABLE detect_job_items ADD COLUMN IF NOT EXISTS step_code VARCHAR(64) NOT NULL DEFAULT '', ADD COLUMN IF NOT EXISTS step_payload_json JSONB, ADD COLUMN IF NOT EXISTS result_payload_json JSONB; ALTER TABLE detect_job_items DROP CONSTRAINT IF EXISTS uq_detect_job_items_job_domain; CREATE UNIQUE INDEX IF NOT EXISTS idx_detect_job_items_job_domain_step ON detect_job_items(job_id, domain_id, step_code); CREATE TABLE IF NOT EXISTS detect_run_events ( id BIGSERIAL PRIMARY KEY, job_id BIGINT REFERENCES detect_jobs(id) ON DELETE SET NULL, job_item_id BIGINT REFERENCES detect_job_items(id) ON DELETE SET NULL, node_code VARCHAR(64) NOT NULL DEFAULT '', event_type VARCHAR(64) NOT NULL DEFAULT '', level VARCHAR(16) NOT NULL DEFAULT 'info', message TEXT NOT NULL DEFAULT '', payload_json JSONB, created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP ); CREATE INDEX IF NOT EXISTS idx_detect_run_events_job_created ON detect_run_events(job_id, created_at DESC); CREATE TABLE IF NOT EXISTS detect_sync_records ( id BIGSERIAL PRIMARY KEY, sync_type VARCHAR(32) NOT NULL DEFAULT '', source_region VARCHAR(32) NOT NULL DEFAULT '', target_region VARCHAR(32) NOT NULL DEFAULT '', status VARCHAR(32) NOT NULL DEFAULT 'pending', payload_json JSONB, error_message TEXT NOT NULL DEFAULT '', created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP, updated_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP ); """ _STALE_AFTER_SECONDS = 90 _OFFLINE_AFTER_MINUTES = 5 _IMPORTED_RUNTIME_STALE_AFTER_MINUTES = 10 _IMPORTED_RUNTIME_OFFLINE_AFTER_MINUTES = 30 _PRUNE_IMPORTED_AFTER_MINUTES = 30 _PRUNE_GENERAL_AFTER_HOURS = 6 _RUNTIME_SCHEMA_READY = False _RUNTIME_SCHEMA_LOCK = threading.Lock() _RUNTIME_SCHEMA_ADVISORY_LOCK_ID = 62021001 _RUNTIME_SCHEMA_INDEX_ADVISORY_LOCK_ID = 62021002 _DISABLED_MANAGED_NODE_CACHE: set[str] = set() _DISABLED_MANAGED_NODE_CACHE_EXPIRES_AT = 0.0 _RUNTIME_REQUIRED_TABLES = ( "detect_worker_nodes", "detect_jobs", "detect_job_items", "detect_run_events", "detect_sync_records", ) _RUNTIME_REQUIRED_COLUMNS = { "detect_jobs": {"task_mode", "step_code"}, "detect_job_items": {"step_code", "step_payload_json", "result_payload_json"}, } _RUNTIME_REQUIRED_INDEXES = ( "idx_detect_job_items_job_domain_step", "idx_detect_job_items_claim_step_ready", "idx_detect_job_items_claim_job_step_ready", "idx_detect_sync_records_scope_created", "idx_detect_sync_records_source_record_created", "idx_detect_sync_records_source_record_hash_created", "idx_detect_sync_records_runtime_push_lookup", ) _RUNTIME_REQUIRED_INDEX_TABLES = { "idx_detect_job_items_job_domain_step": "detect_job_items", "idx_detect_job_items_claim_step_ready": "detect_job_items", "idx_detect_job_items_claim_job_step_ready": "detect_job_items", "idx_detect_sync_records_scope_created": "detect_sync_records", "idx_detect_sync_records_source_record_created": "detect_sync_records", "idx_detect_sync_records_source_record_hash_created": "detect_sync_records", "idx_detect_sync_records_runtime_push_lookup": "detect_sync_records", } _RUNTIME_REQUIRED_INDEX_DDL = { "idx_detect_job_items_job_domain_step": """ CREATE UNIQUE INDEX CONCURRENTLY IF NOT EXISTS idx_detect_job_items_job_domain_step ON detect_job_items(job_id, domain_id, step_code) """, "idx_detect_job_items_claim_step_ready": """ CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_detect_job_items_claim_step_ready ON detect_job_items(status, step_code, lease_expires_at, create_time, id) WHERE step_code <> '' AND status IN ('pending', 'failed') """, "idx_detect_job_items_claim_job_step_ready": """ CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_detect_job_items_claim_job_step_ready ON detect_job_items(job_id, status, step_code, lease_expires_at, create_time, id) WHERE step_code <> '' AND status IN ('pending', 'failed') """, "idx_detect_sync_records_scope_created": """ CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_detect_sync_records_scope_created ON detect_sync_records(sync_type, source_region, target_region, created_at DESC, id DESC) """, "idx_detect_sync_records_source_record_created": """ CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_detect_sync_records_source_record_created ON detect_sync_records( sync_type, source_region, target_region, ((payload_json->>'source_record_id')), created_at DESC, id DESC ) """, "idx_detect_sync_records_source_record_hash_created": """ CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_detect_sync_records_source_record_hash_created ON detect_sync_records( sync_type, source_region, target_region, ((payload_json->>'source_record_id')), ((payload_json->>'projection_hash')), created_at DESC, id DESC ) """, "idx_detect_sync_records_runtime_push_lookup": """ CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_detect_sync_records_runtime_push_lookup ON detect_sync_records( source_region, target_region, ((payload_json->>'sync_type')), ((payload_json->>'source_record_id')), created_at DESC, id DESC ) WHERE sync_type = 'runtime_push' """, } def _resolve_local_ip() -> str: try: return socket.gethostbyname(socket.gethostname()) except Exception: return "" def _decode_json(value: object) -> dict: if isinstance(value, dict): return value if value in (None, ""): return {} try: return json.loads(value) except Exception: return {} def _parse_runtime_timestamp(value: object) -> datetime | None: raw = str(value or "").strip() if not raw: return None normalized = raw.replace("Z", "+00:00") try: return datetime.fromisoformat(normalized) except Exception: return None def _control_node_supports_worker(*, region: object, metadata: dict | None) -> bool: normalized_region = str(region or "").strip() runtime_metadata = dict(metadata or {}) active_threads = int(runtime_metadata.get("active_threads", 0) or 0) max_threads = int(runtime_metadata.get("max_threads", 0) or 0) if normalized_region != "mainland": return False return bool( runtime_metadata.get("worker_online", False) or runtime_metadata.get("detect_participating", False) or active_threads > 0 or max_threads > 0 ) def _metadata_idle_without_runtime_work(metadata: dict | None) -> bool: runtime_metadata = dict(metadata or {}) phase = str( runtime_metadata.get("phase") or runtime_metadata.get("phase_label") or "" ).strip().lower() if phase not in {"idle", "completed", "stopped"}: return False active_job_code = str(runtime_metadata.get("active_job_code") or "").strip() counters = ( int(runtime_metadata.get("job_items_total", 0) or 0), int(runtime_metadata.get("job_items_claimed", 0) or 0), int(runtime_metadata.get("job_items_running", 0) or 0), int(runtime_metadata.get("job_items_completed", 0) or 0), int(runtime_metadata.get("job_items_failed", 0) or 0), ) if active_job_code: return False return not any(value > 0 for value in counters) def _load_managed_node_overlays() -> dict[str, dict]: try: with get_db() as conn: with conn.cursor() as cur: cur.execute( """ SELECT node_code, metadata_json, last_seen_at FROM ops_managed_nodes WHERE is_enabled = TRUE """ ) rows = cur.fetchall() except Exception: return {} overlays: dict[str, dict] = {} for row in rows: node_code = str(row[0] or "").strip() if not node_code: continue overlays[node_code] = { "metadata": _decode_json(row[1]), "last_seen_at": row[2], } return overlays def _load_disabled_managed_node_codes() -> set[str]: global _DISABLED_MANAGED_NODE_CACHE, _DISABLED_MANAGED_NODE_CACHE_EXPIRES_AT now_ts = datetime.now().timestamp() if now_ts < _DISABLED_MANAGED_NODE_CACHE_EXPIRES_AT: return set(_DISABLED_MANAGED_NODE_CACHE) try: with get_db() as conn: with conn.cursor() as cur: cur.execute( """ SELECT node_code FROM ops_managed_nodes WHERE is_enabled = FALSE """ ) rows = cur.fetchall() except Exception: return set(_DISABLED_MANAGED_NODE_CACHE) disabled_codes = { str(row[0] or "").strip() for row in list(rows or []) if str(row[0] or "").strip() } _DISABLED_MANAGED_NODE_CACHE = disabled_codes _DISABLED_MANAGED_NODE_CACHE_EXPIRES_AT = now_ts + 5.0 return set(disabled_codes) def ensure_runtime_schema() -> None: global _RUNTIME_SCHEMA_READY if _RUNTIME_SCHEMA_READY: return with _RUNTIME_SCHEMA_LOCK: if _RUNTIME_SCHEMA_READY: return with get_db() as conn: with conn.cursor() as cur: if _runtime_schema_basics_present(cur): missing_indexes = list(_runtime_missing_indexes(cur)) if not missing_indexes: _RUNTIME_SCHEMA_READY = True return else: missing_indexes = [] if missing_indexes: _ensure_runtime_schema_indexes(missing_indexes) with conn.cursor() as cur: if _runtime_schema_basics_present(cur) and not list(_runtime_missing_indexes(cur)): _RUNTIME_SCHEMA_READY = True return else: try: with conn.cursor() as cur: cur.execute("SELECT pg_advisory_xact_lock(%s)", (_RUNTIME_SCHEMA_ADVISORY_LOCK_ID,)) cur.execute(_RUNTIME_SCHEMA_SQL) conn.commit() except Exception as exc: recoverable = isinstance(exc, (errors.DeadlockDetected, errors.LockNotAvailable)) try: conn.rollback() except Exception: pass if not recoverable: raise with conn.cursor() as cur: if not _runtime_schema_basics_present(cur): raise _RUNTIME_SCHEMA_READY = True def _runtime_schema_basics_present(cur) -> bool: for table_name in _RUNTIME_REQUIRED_TABLES: cur.execute("SELECT to_regclass(%s)", (f"public.{table_name}",)) row = cur.fetchone() if not row or not row[0]: return False for table_name, required_columns in _RUNTIME_REQUIRED_COLUMNS.items(): cur.execute( """ SELECT column_name FROM information_schema.columns WHERE table_schema = 'public' AND table_name = %s """, (table_name,), ) existing_columns = {str(row[0] or "").strip() for row in list(cur.fetchall() or [])} if not set(required_columns).issubset(existing_columns): return False return True def _runtime_missing_indexes(cur): states = _runtime_index_states(cur) for index_name in _RUNTIME_REQUIRED_INDEXES: if not bool((states.get(index_name) or {}).get("valid")): yield index_name def _runtime_index_states(cur) -> dict[str, dict[str, bool]]: table_names = sorted(set(_RUNTIME_REQUIRED_INDEX_TABLES.values())) cur.execute( """ SELECT idx.relname AS index_name, pg_index.indisvalid AS is_valid, pg_index.indisready AS is_ready, pg_index.indislive AS is_live FROM pg_class AS idx JOIN pg_index ON pg_index.indexrelid = idx.oid JOIN pg_class AS tbl ON tbl.oid = pg_index.indrelid JOIN pg_namespace AS ns ON ns.oid = tbl.relnamespace WHERE ns.nspname = 'public' AND tbl.relname = ANY(%s) AND idx.relname = ANY(%s) """, (table_names, list(_RUNTIME_REQUIRED_INDEXES)), ) states = { index_name: {"valid": False, "ready": False, "live": False} for index_name in _RUNTIME_REQUIRED_INDEXES } for row in list(cur.fetchall() or []): index_name = str(row[0] or "").strip() if index_name not in states: continue states[index_name] = { "valid": bool(row[1]), "ready": bool(row[2]), "live": bool(row[3]), } return states def _ensure_runtime_schema_indexes(index_names: list[str] | tuple[str, ...]) -> None: normalized_indexes = [ index_name for index_name in list(index_names or []) if str(index_name or "").strip() in _RUNTIME_REQUIRED_INDEX_DDL ] if not normalized_indexes: return with get_db() as conn: conn.autocommit = True with conn.cursor() as cur: cur.execute("SELECT pg_try_advisory_lock(%s)", (_RUNTIME_SCHEMA_INDEX_ADVISORY_LOCK_ID,)) row = cur.fetchone() if not bool((row or [False])[0]): return try: current_states = _runtime_index_states(cur) for index_name in normalized_indexes: index_state = current_states.get(index_name) or {} if bool(index_state.get("valid")): continue if bool(index_state.get("ready")) or bool(index_state.get("live")): cur.execute(f"DROP INDEX CONCURRENTLY IF EXISTS {index_name}") cur.execute(_RUNTIME_REQUIRED_INDEX_DDL[index_name]) finally: cur.execute("SELECT pg_advisory_unlock(%s)", (_RUNTIME_SCHEMA_INDEX_ADVISORY_LOCK_ID,)) def register_node_heartbeat( *, node_code: str, region: str, role: str, status: str, current_load: int = 0, worker_version: str = "0.1.0", metadata: dict | None = None, hostname_override: str | None = None, ip_override: str | None = None, ) -> None: hostname = str(hostname_override or "").strip() or socket.gethostname() ip = str(ip_override or "").strip() or _resolve_local_ip() with get_db() as conn: with conn.cursor() as cur: cur.execute( """ INSERT INTO detect_worker_nodes ( node_code, region, role, hostname, ip, status, worker_version, current_load, metadata_json, last_heartbeat_at, update_time ) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, CURRENT_TIMESTAMP, CURRENT_TIMESTAMP) ON CONFLICT (node_code) DO UPDATE SET region = EXCLUDED.region, role = EXCLUDED.role, hostname = EXCLUDED.hostname, ip = EXCLUDED.ip, status = EXCLUDED.status, worker_version = EXCLUDED.worker_version, current_load = EXCLUDED.current_load, metadata_json = EXCLUDED.metadata_json, last_heartbeat_at = CURRENT_TIMESTAMP, update_time = CURRENT_TIMESTAMP """, ( node_code, region, role, hostname, ip, status, worker_version, max(0, int(current_load or 0)), json.dumps(metadata or {}, ensure_ascii=False), ), ) conn.commit() def cleanup_imported_runtime_nodes(*, region: str, role: str, keep_node_code: str) -> None: normalized_region = str(region or "").strip() or "unknown" normalized_role = str(role or "").strip() or "unknown" preserved_node_code = str(keep_node_code or "").strip() if not preserved_node_code: return with get_db() as conn: with conn.cursor() as cur: cur.execute( """ DELETE FROM detect_worker_nodes WHERE region = %s AND role = %s AND node_code <> %s AND ( node_code = %s OR (metadata_json->>'service') = 'runtime-ingest' ) """, ( normalized_region, normalized_role, preserved_node_code, f"{normalized_region}-{normalized_role}-imported", ), ) conn.commit() def cleanup_imported_runtime_nodes_many(*, region: str, role: str, keep_node_codes: list[str] | tuple[str, ...] | set[str]) -> None: normalized_region = str(region or "").strip() or "unknown" normalized_role = str(role or "").strip() or "unknown" preserved_node_codes = sorted( { str(node_code or "").strip() for node_code in (keep_node_codes or []) if str(node_code or "").strip() } ) if not preserved_node_codes: return placeholders = ", ".join(["%s"] * len(preserved_node_codes)) with get_db() as conn: with conn.cursor() as cur: cur.execute( f""" DELETE FROM detect_worker_nodes WHERE region = %s AND role = %s AND ( node_code = %s OR (metadata_json->>'service') = 'runtime-ingest' ) AND node_code NOT IN ({placeholders}) """, ( normalized_region, normalized_role, f"{normalized_region}-{normalized_role}-imported", *preserved_node_codes, ), ) conn.commit() def prune_expired_runtime_nodes() -> None: imported_cutoff = datetime.now() - timedelta(minutes=_PRUNE_IMPORTED_AFTER_MINUTES) general_cutoff = datetime.now() - timedelta(hours=_PRUNE_GENERAL_AFTER_HOURS) with get_db() as conn: with conn.cursor() as cur: cur.execute( """ DELETE FROM detect_worker_nodes WHERE ( (metadata_json->>'service') = 'runtime-ingest' AND last_heartbeat_at < %s ) OR ( COALESCE(metadata_json->>'service', '') <> 'runtime-ingest' AND last_heartbeat_at < %s ) """, (imported_cutoff, general_cutoff), ) conn.commit() def register_local_control_heartbeat() -> None: from app.services.detect_service import get_detect_status detect_status = get_detect_status() worker_online = bool(detect_status.get("worker_online", False)) active_job = dict(detect_status.get("active_job") or {}) node_stats = list(active_job.get("node_stats") or []) local_bucket = next( (item for item in node_stats if str(item.get("node_code") or "").strip() == settings.node_code), {}, ) items_total = int(local_bucket.get("items_total", 0) or 0) items_claimed = int(local_bucket.get("items_claimed", 0) or 0) items_running = int(local_bucket.get("items_running", 0) or 0) items_completed = int(local_bucket.get("items_completed", 0) or 0) active_threads = int(detect_status.get("active_thread_count", 0) or 0) max_threads = int(detect_status.get("max_thread_count", 0) or 0) current_load = max(items_running, active_threads, 0) detect_participating = bool(worker_online and (items_total > 0 or current_load > 0)) node_status = "busy" if current_load > 0 else "online" register_node_heartbeat( node_code=settings.node_code, region=settings.node_region, role=settings.node_role, status=node_status, current_load=current_load, metadata={ "service": "domain-api", "api_host": settings.api_host, "api_port": settings.api_port, "worker_mode": settings.worker_mode, "worker_online": worker_online, "worker_process_count": int(detect_status.get("worker_process_count", 0) or 0), "detect_participating": detect_participating, "active_job_code": str(active_job.get("job_code") or ""), "active_job_status": str(active_job.get("status") or ""), "job_items_total": items_total, "job_items_claimed": items_claimed, "job_items_running": items_running, "job_items_completed": items_completed, "active_threads": active_threads, "max_threads": max_threads, "available_proxy_count": int(detect_status.get("available_proxy_count", 0) or 0), "proxy_runtime_label": str(detect_status.get("proxy_runtime_label") or "").strip(), "proxy_runtime_reason": str(detect_status.get("proxy_runtime_reason") or "").strip(), "proxy_last_refresh_status": str(detect_status.get("proxy_last_refresh_status") or "").strip(), "proxy_last_refresh_time": str(detect_status.get("proxy_last_refresh_time") or "").strip(), "proxy_last_refresh_source_count": int(detect_status.get("proxy_last_refresh_source_count", 0) or 0), "proxy_last_refresh_total_items": int(detect_status.get("proxy_last_refresh_total_items", 0) or 0), "proxy_last_validated_count": int(detect_status.get("proxy_last_validated_count", 0) or 0), "proxy_last_available_count": int(detect_status.get("proxy_last_available_count", 0) or 0), "phase_label": str(detect_status.get("phase_label") or ""), "phase_detail": str(detect_status.get("phase_detail") or ""), "updated_at": datetime.now().isoformat(timespec="seconds"), }, ) def _resolve_effective_runtime_heartbeat( *, metadata: dict | None, last_heartbeat_at: datetime | None, update_time: datetime | None, ) -> datetime | None: effective_last_heartbeat = last_heartbeat_at runtime_metadata = dict(metadata or {}) if str(runtime_metadata.get("service") or "").strip() != "runtime-ingest": return effective_last_heartbeat metadata_updated_at = _parse_runtime_timestamp(runtime_metadata.get("updated_at")) for candidate in (update_time, metadata_updated_at): if not candidate: continue if effective_last_heartbeat is None or candidate > effective_last_heartbeat: effective_last_heartbeat = candidate return effective_last_heartbeat def _normalize_node_status(raw_status: str, last_heartbeat_at: datetime | None, *, metadata: dict | None = None) -> str: status = str(raw_status or "").strip() or "unknown" if not last_heartbeat_at: return status now = datetime.now(last_heartbeat_at.tzinfo) if last_heartbeat_at.tzinfo else datetime.now() age = now - last_heartbeat_at runtime_metadata = dict(metadata or {}) service_name = str(runtime_metadata.get("service") or "").strip() stale_after = timedelta(seconds=_STALE_AFTER_SECONDS) offline_after = timedelta(minutes=_OFFLINE_AFTER_MINUTES) if service_name == "runtime-ingest": stale_after = timedelta(minutes=_IMPORTED_RUNTIME_STALE_AFTER_MINUTES) offline_after = timedelta(minutes=_IMPORTED_RUNTIME_OFFLINE_AFTER_MINUTES) if age > offline_after: return "offline" if age > stale_after: return "stale" return status @db_read_retry() def get_cluster_snapshot() -> dict: prune_expired_runtime_nodes() register_local_control_heartbeat() managed_overlays = _load_managed_node_overlays() disabled_node_codes = _load_disabled_managed_node_codes() with get_db() as conn: with conn.cursor() as cur: cur.execute( """ SELECT node_code, region, role, hostname, ip, status, worker_version, current_load, metadata_json, last_heartbeat_at, update_time FROM detect_worker_nodes ORDER BY CASE WHEN status = 'busy' THEN 0 WHEN status = 'online' THEN 1 ELSE 2 END, region ASC, role ASC, node_code ASC LIMIT 100 """ ) rows = [ row for row in list(cur.fetchall() or []) if str((row or [""])[0] or "").strip() not in disabled_node_codes ] cur.execute( """ SELECT node_code, region, role, hostname, ip, status, worker_version, current_load, metadata_json, last_heartbeat_at, update_time FROM detect_worker_nodes ORDER BY CASE WHEN status = 'busy' THEN 0 WHEN status = 'online' THEN 1 ELSE 2 END, region ASC, role ASC, node_code ASC """ ) summary_rows = [ row for row in list(cur.fetchall() or []) if str((row or [""])[0] or "").strip() not in disabled_node_codes ] cur.execute("SELECT count(*) FROM detect_jobs") jobs_total = cur.fetchone()[0] cur.execute("SELECT count(*) FROM detect_job_items WHERE status IN ('pending', 'claimed', 'running')") active_items = cur.fetchone()[0] def _build_node_payload(row: tuple) -> dict: node_code = str(row[0] or "").strip() metadata = _decode_json(row[8]) current_load = int(row[7] or 0) sanitized_idle_runtime = _metadata_idle_without_runtime_work(metadata) if sanitized_idle_runtime: current_load = 0 metadata["active_threads"] = 0 metadata["detect_participating"] = False metadata["sanitized_runtime_state"] = "idle_phase_zeroed" runtime_last_heartbeat = row[9] runtime_update_time = row[10] managed_overlay = managed_overlays.get(node_code) or {} managed_last_seen = managed_overlay.get("last_seen_at") effective_last_heartbeat = _resolve_effective_runtime_heartbeat( metadata=metadata, last_heartbeat_at=runtime_last_heartbeat, update_time=runtime_update_time, ) overlay_is_newer = bool( managed_last_seen and (not effective_last_heartbeat or managed_last_seen > effective_last_heartbeat) ) effective_last_heartbeat = managed_last_seen if overlay_is_newer else effective_last_heartbeat normalized_status = _normalize_node_status(row[5], effective_last_heartbeat, metadata=metadata) if sanitized_idle_runtime and normalized_status == "busy": normalized_status = "online" if overlay_is_newer and normalized_status in {"offline", "stale"}: normalized_status = "busy" if current_load > 0 else "online" if managed_last_seen: metadata["agent_last_seen_at"] = managed_last_seen.isoformat(sep=" ", timespec="seconds") if overlay_is_newer: metadata["cluster_status_source"] = "managed-agent-overlay" return { "node_code": node_code, "region": row[1], "role": row[2], "hostname": row[3], "ip": row[4], "status": normalized_status, "worker_version": row[6], "current_load": current_load, "metadata": metadata, "last_heartbeat_at": effective_last_heartbeat.isoformat(sep=" ", timespec="seconds") if effective_last_heartbeat else "", } nodes = [_build_node_payload(row) for row in rows] summary_nodes = [_build_node_payload(row) for row in summary_rows] status_counts: dict[str, int] = {} role_counts: dict[str, int] = {} region_counts: dict[str, int] = {} busy_nodes: list[str] = [] stale_nodes: list[str] = [] offline_nodes: list[str] = [] online_worker_nodes = 0 dedicated_online_worker_nodes = 0 online_control_nodes = 0 for node in summary_nodes: node_status = str(node.get("status") or "unknown") node_role = str(node.get("role") or "unknown") node_region = str(node.get("region") or "unknown") node_code = str(node.get("node_code") or "") metadata = node.get("metadata") or {} node_current_load = int(node.get("current_load", 0) or 0) effective_worker = False if node_role == "control" and not _control_node_supports_worker(region=node_region, metadata=metadata): node_current_load = 0 node["current_load"] = 0 if node_status == "busy": node_status = "online" node["status"] = "online" status_counts[node_status] = status_counts.get(node_status, 0) + 1 role_counts[node_role] = role_counts.get(node_role, 0) + 1 region_counts[node_region] = region_counts.get(node_region, 0) + 1 if node_status == "busy": busy_nodes.append(node_code) if node_status == "stale": stale_nodes.append(node_code) if node_status == "offline": offline_nodes.append(node_code) if node_role == "worker" and node_status in {"online", "busy"}: dedicated_online_worker_nodes += 1 effective_worker = True elif node_role == "control" and node_status in {"online", "busy"}: if _control_node_supports_worker(region=node_region, metadata=metadata): effective_worker = True if effective_worker: online_worker_nodes += 1 if node_role == "control" and node_status in {"online", "busy"}: online_control_nodes += 1 node["is_effective_worker"] = effective_worker node["detect_participating"] = bool( effective_worker and (metadata.get("detect_participating", False) or node_current_load > 0) ) summary_node_map = { str(item.get("node_code") or "").strip(): item for item in summary_nodes if str(item.get("node_code") or "").strip() } for node in nodes: summary_node = summary_node_map.get(str(node.get("node_code") or "").strip()) if not summary_node: continue node["current_load"] = summary_node.get("current_load", node.get("current_load", 0)) node["status"] = summary_node.get("status", node.get("status", "unknown")) node["metadata"] = summary_node.get("metadata", node.get("metadata") or {}) node["last_heartbeat_at"] = summary_node.get("last_heartbeat_at", node.get("last_heartbeat_at", "")) node["is_effective_worker"] = bool(summary_node.get("is_effective_worker", False)) node["detect_participating"] = bool(summary_node.get("detect_participating", False)) return { "nodes": nodes, "nodes_total": len(summary_nodes), "jobs_total": jobs_total, "active_job_items": active_items, "summary": { "status_counts": status_counts, "role_counts": role_counts, "region_counts": region_counts, "busy_nodes": busy_nodes, "stale_nodes": stale_nodes, "offline_nodes": offline_nodes, "online_worker_nodes": online_worker_nodes, "dedicated_online_worker_nodes": dedicated_online_worker_nodes, "online_control_nodes": online_control_nodes, }, }