Host reconciliation mostly done

This commit is contained in:
2025-11-17 13:50:59 +10:30
parent 3cb2a0f2fb
commit e43edf5c53
10 changed files with 917 additions and 10 deletions
+7
View File
@@ -41,6 +41,13 @@ def register_socketio_handlers(socketio):
logger.info(f"[{worker_id}] Task {task_id} acknowledged. Exec time: {task.execution_time:.2f}s")
# Check if this is a reconcile_and_delete task completion
if task.task_type == "reconcile_and_delete" and result.get("success"):
logger.info(f"[{worker_id}] Reconcile and delete task completed successfully, initiating host online transition")
# Import the function here to avoid circular imports
from websocket_server.worker_manager import handle_reconcile_and_delete_completion
handle_reconcile_and_delete_completion(worker_id)
redis_client = get_redis_client()
redis_client.set(f"worker_status_{worker_id}", "idle")
+9 -1
View File
@@ -17,6 +17,7 @@ def assign_task_to_worker(worker_id):
"""
Attempt to assign a pending task to a connected worker.
Ensures locking, dependency satisfaction, and Redis status tracking.
Prioritizes reconciliation tasks.
"""
redis_client = get_redis_client()
max_lock_retries = 5
@@ -60,6 +61,13 @@ def assign_task_to_worker(worker_id):
redis_client.set(f"worker_status_{worker_id}", "idle")
return
# Prioritize reconciliation tasks
ready.sort(key=lambda t: (
0 if t.task_type == "reconcile_and_delete" else 1, # reconcile_and_delete has highest priority
1 if t.task_type == "pod-update" else 2, # pod-update has medium priority
t.creation_time # Then by creation time
))
# New isolated session for locking and updating
with get_db_session() as session:
selected = session.query(Task).filter(
@@ -69,7 +77,7 @@ def assign_task_to_worker(worker_id):
if selected:
if ENABLE_TASK_ASSIGNMENT_DEBUG:
logger.debug(f"[{worker_id}] Task selected: {selected.id}")
logger.debug(f"[{worker_id}] Task selected: {selected.id} (type: {selected.task_type})")
break
if ENABLE_TASK_ASSIGNMENT_DEBUG:
+161 -1
View File
@@ -5,10 +5,14 @@ import uuid
import time
import threading
import logging
import json
from websocket_server.shared_state import connected_workers, worker_lock, ping_tracker
from websocket_server.config import get_redis_client,PING_INTERVAL_SECONDS, ASSIGN_INTERVAL_SECONDS, LIVENESS_CHECK_INTERVAL_SECONDS, PING_EXPIRY_SECONDS, ENABLE_WEBSOCKET_PING_DEBUG, ENABLE_TASK_ASSIGNMENT_DEBUG
from websocket_server.task_assigner import assign_task_to_worker
from websocket_server.events import base
from websocket_server.models import Task
from sqlalchemy.orm import sessionmaker
from websocket_server.config import engine
logger = logging.getLogger("websocket_server")
@@ -21,6 +25,9 @@ worker_threads = {}
redis = get_redis_client()
# Create session factory for database operations
Session = sessionmaker(bind=engine)
# Last timestamps
last_ping_time = 0
last_assign_time = 0
@@ -28,7 +35,7 @@ last_assign_time = 0
def notify_worker_online(worker_id):
"""
Inform the API server that a worker has connected.
Inform the API server that a worker has connected and start reconciliation.
"""
logger.debug(f"[{worker_id}] Notifying API server: online")
payload = {"status": "online"}
@@ -37,6 +44,10 @@ def notify_worker_online(worker_id):
try:
response = requests.put(f"{api_server_url}/workload_hosts/{worker_id}", json=payload, headers=headers)
logger.info(f"[{worker_id}] API server acknowledged online state")
# Start host reconciliation process
start_host_reconciliation(worker_id)
except Exception as e:
logger.error(f"[{worker_id}] Failed to notify API server of online status: {e}")
@@ -57,6 +68,155 @@ def notify_worker_disconnect(worker_id):
logger.error(f"[{worker_id}] Failed to notify API server of disconnect: {e}")
def start_host_reconciliation(worker_id):
"""
Start the host reconciliation process when a worker connects.
This moves the host status to 'reconciling' and initiates the reconciliation workflow.
"""
logger.info(f"[{worker_id}] Starting host reconciliation process")
try:
# Step 1: Move host status to 'reconciling'
update_host_status(worker_id, "reconciling")
# Step 2: Get containers that should be on this host
expected_containers = get_expected_containers_for_host(worker_id)
# Step 3: Send pod-update tasks for each pod containing containers for this host
send_pod_update_tasks(worker_id, expected_containers)
# Step 4: Send reconcile_and_delete task
send_reconcile_and_delete_task(worker_id, expected_containers)
logger.info(f"[{worker_id}] Host reconciliation process initiated")
except Exception as e:
logger.error(f"[{worker_id}] Error during host reconciliation: {e}")
# If reconciliation fails, move host back to online status
update_host_status(worker_id, "online")
def update_host_status(worker_id, status):
"""
Update the host status in the API server.
"""
logger.debug(f"[{worker_id}] Updating host status to: {status}")
payload = {"status": status}
headers = {"Content-Type": "application/json"}
try:
response = requests.put(f"{api_server_url}/workload_hosts/{worker_id}", json=payload, headers=headers)
if response.status_code == 200:
logger.info(f"[{worker_id}] Host status updated to: {status}")
else:
logger.warning(f"[{worker_id}] Failed to update host status: {response.status_code} - {response.text}")
except Exception as e:
logger.error(f"[{worker_id}] Error updating host status: {e}")
def get_expected_containers_for_host(worker_id):
"""
Get all containers that should be running on this host from the API server.
"""
logger.debug(f"[{worker_id}] Fetching expected containers for host")
try:
response = requests.get(f"{api_server_url}/workload_hosts/{worker_id}/container_workloads")
if response.status_code == 200:
data = response.json()
if data.get('success') and 'data' in data:
containers_data = data['data']
logger.info(f"[{worker_id}] Found {len(containers_data.get('job_details', {}).get('containers', []))} expected containers")
return containers_data
else:
logger.warning(f"[{worker_id}] API response format unexpected: {data}")
return {"job_details": {"containers": []}}
else:
logger.error(f"[{worker_id}] Failed to fetch containers: {response.status_code}")
return {"job_details": {"containers": []}}
except Exception as e:
logger.error(f"[{worker_id}] Error fetching expected containers: {e}")
return {"job_details": {"containers": []}}
def send_pod_update_tasks(worker_id, expected_containers):
"""
Send pod-update tasks for each pod that contains containers for this host.
"""
containers = expected_containers.get('job_details', {}).get('containers', [])
# Group containers by pod_id
pods = {}
for container in containers:
pod_id = container.get('pod_id', 'no-pod')
if pod_id not in pods:
pods[pod_id] = []
pods[pod_id].append(container)
logger.info(f"[{worker_id}] Sending pod-update tasks for {len(pods)} pods")
# Create a task for each pod
for pod_id, pod_containers in pods.items():
try:
task = Task(
worker_id=worker_id,
task_type="pod-update",
job_details=json.dumps({
"pod_id": pod_id,
"containers": pod_containers
}),
status="pending"
)
session = Session()
session.add(task)
session.commit()
session.close()
logger.debug(f"[{worker_id}] Created pod-update task for pod {pod_id} with {len(pod_containers)} containers")
except Exception as e:
logger.error(f"[{worker_id}] Error creating pod-update task for pod {pod_id}: {e}")
def send_reconcile_and_delete_task(worker_id, expected_containers):
"""
Send the final reconcile_and_delete task with the list of expected container IDs.
"""
containers = expected_containers.get('job_details', {}).get('containers', [])
expected_container_ids = [container.get('container_id') for container in containers if container.get('container_id')]
logger.info(f"[{worker_id}] Creating reconcile_and_delete task for {len(expected_container_ids)} containers")
try:
task = Task(
worker_id=worker_id,
task_type="reconcile_and_delete",
job_details=json.dumps({
"expected_container_ids": expected_container_ids
}),
status="pending"
)
session = Session()
session.add(task)
session.commit()
session.close()
logger.debug(f"[{worker_id}] Created reconcile_and_delete task")
except Exception as e:
logger.error(f"[{worker_id}] Error creating reconcile_and_delete task: {e}")
def handle_reconcile_and_delete_completion(worker_id):
"""
Handle the completion of the reconcile_and_delete task by moving host to online status.
"""
logger.info(f"[{worker_id}] Reconcile and delete completed, moving host to online status")
update_host_status(worker_id, "online")
def worker_dispatch_flag_check():
from websocket_server.redis_utils import redis_subscribe, thread_stop_flags, worker_dispatch_flags