Feat: Retry diff Hosts on Workload failure
This commit is contained in:
@@ -0,0 +1,139 @@
|
||||
import configparser
|
||||
import io
|
||||
import logging
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Optional, Tuple
|
||||
|
||||
from runtime_urls import (
|
||||
SCHEDULER_CONF_PATH as _CONF_PATH,
|
||||
SCHEDULER_RETRY_BASE_DELAY_SECONDS,
|
||||
SCHEDULER_RETRY_BACKOFF_FACTOR,
|
||||
SCHEDULER_RETRY_MAX_DELAY_SECONDS,
|
||||
)
|
||||
|
||||
logger = logging.getLogger("websocket_server")
|
||||
|
||||
# Module-level cache so the file is only parsed once per process.
|
||||
_policy_cache: Optional[dict] = None
|
||||
|
||||
|
||||
def _load_policy(conf_path: str = _CONF_PATH) -> dict:
|
||||
"""Parse scheduler.conf and return {ERROR_TYPE_KEY: (action, max_tries)}."""
|
||||
global _policy_cache
|
||||
if _policy_cache is not None:
|
||||
return _policy_cache
|
||||
|
||||
# Pre-process the file so configparser handles it correctly:
|
||||
# - Drop option-like (key=value) lines that appear before the first [section]
|
||||
# (the file has a format-doc line at the top that would confuse configparser)
|
||||
# - Strip leading whitespace from real option lines inside sections
|
||||
# (configparser treats indented lines as value continuations otherwise)
|
||||
lines: list[str] = []
|
||||
in_section = False
|
||||
try:
|
||||
with open(conf_path) as f:
|
||||
for line in f:
|
||||
stripped = line.lstrip()
|
||||
if stripped.startswith("["):
|
||||
in_section = True
|
||||
looks_like_option = "=" in stripped and not stripped.startswith(("#", ";", "["))
|
||||
if not in_section:
|
||||
# Before first [section]: keep comments and blanks, drop option lines
|
||||
if not looks_like_option:
|
||||
lines.append(line)
|
||||
else:
|
||||
lines.append(stripped if looks_like_option else line)
|
||||
except FileNotFoundError:
|
||||
logger.warning("scheduler.conf not found at %s — all errors default to skip", conf_path)
|
||||
_policy_cache = {}
|
||||
return _policy_cache
|
||||
|
||||
cfg = configparser.RawConfigParser(inline_comment_prefixes=("#",))
|
||||
cfg.optionxform = str # preserve case
|
||||
cfg.read_file(io.StringIO("".join(lines)))
|
||||
|
||||
policy: dict = {}
|
||||
for section in cfg.sections():
|
||||
for key, raw_value in cfg.items(section):
|
||||
error_key = key.strip().upper()
|
||||
action_str = raw_value.strip().split()[0] if raw_value.strip() else "skip"
|
||||
parts = action_str.split(";")
|
||||
action = parts[0].lower()
|
||||
if action in ("skip", "skim"):
|
||||
policy[error_key] = ("skip", 0)
|
||||
elif action == "retry":
|
||||
count_str = parts[1] if len(parts) > 1 else "n"
|
||||
max_tries: Optional[int] = None if count_str == "n" else int(count_str)
|
||||
policy[error_key] = ("retry", max_tries)
|
||||
else:
|
||||
policy[error_key] = ("skip", 0)
|
||||
|
||||
_policy_cache = policy
|
||||
return policy
|
||||
|
||||
|
||||
class HandleRetry:
|
||||
"""
|
||||
Look up the scheduler.conf policy for a given ErrorType and apply it to a
|
||||
Workload ORM object within an existing DB session.
|
||||
|
||||
Usage (inside ack.py's get_db_session context):
|
||||
outcome = HandleRetry().apply(session, workload, error_msg, error_type)
|
||||
# outcome is "retry" or "skip"
|
||||
"""
|
||||
|
||||
def __init__(self, conf_path: str = _CONF_PATH):
|
||||
self._policy = _load_policy(conf_path)
|
||||
|
||||
def get_policy(self, error_type: str) -> Tuple[str, Optional[int]]:
|
||||
"""Return (action, max_tries) for the given error_type string."""
|
||||
key = (error_type or "UNKNOWN").strip().upper()
|
||||
return self._policy.get(key, ("skip", 0))
|
||||
|
||||
def apply(self, session, workload, error_msg: str, error_type: str) -> str:
|
||||
"""Apply the scheduler.conf policy to a failed workload in-place.
|
||||
|
||||
skip → mark error, no host retry.
|
||||
retry;N → exclude the failed host and re-queue; failed-to-spawn once >N hosts tried.
|
||||
retry;n → same but unbounded (only the empty-host-pool check in the beat task stops it).
|
||||
|
||||
Returns "retry" (re-queued as waiting-host) or "skip" (terminal).
|
||||
"""
|
||||
action, policy_max = self.get_policy(error_type)
|
||||
workload.allocation_attempts = (workload.allocation_attempts or 0) + 1
|
||||
workload.last_error_reason = error_msg[:512]
|
||||
workload.last_attempt_at = datetime.now(timezone.utc)
|
||||
|
||||
if action == "skip":
|
||||
workload.excluded_host_ids = []
|
||||
self._set_status(workload, "error")
|
||||
logger.info("Workload %s error_type=%s policy=skip → error", workload.id, error_type)
|
||||
return "skip"
|
||||
|
||||
excluded = list(workload.excluded_host_ids or [])
|
||||
if workload.workload_host_id and workload.workload_host_id not in excluded:
|
||||
excluded.append(workload.workload_host_id)
|
||||
workload.excluded_host_ids = excluded
|
||||
|
||||
if policy_max is not None and len(excluded) > policy_max:
|
||||
workload.excluded_host_ids = []
|
||||
self._set_status(workload, "failed-to-spawn")
|
||||
logger.warning("Workload %s error_type=%s policy=retry;%d → failed-to-spawn (%d hosts tried)",
|
||||
workload.id, error_type, policy_max, len(excluded))
|
||||
return "skip"
|
||||
|
||||
self._set_status(workload, "waiting-host")
|
||||
delay = int(min(SCHEDULER_RETRY_BASE_DELAY_SECONDS * (SCHEDULER_RETRY_BACKOFF_FACTOR ** (len(excluded) - 1)),
|
||||
SCHEDULER_RETRY_MAX_DELAY_SECONDS))
|
||||
workload.next_retry_at = datetime.now(timezone.utc) + timedelta(seconds=delay)
|
||||
logger.info("Workload %s error_type=%s policy=retry;%s → waiting-host (excluded %d, retry in %ds)",
|
||||
workload.id, error_type, policy_max if policy_max is not None else "n", len(excluded), delay)
|
||||
return "retry"
|
||||
|
||||
@staticmethod
|
||||
def _set_status(workload, status: str) -> None:
|
||||
try:
|
||||
workload.set_status(status)
|
||||
except Exception:
|
||||
workload._status = status
|
||||
|
||||
@@ -640,6 +640,7 @@ class Workload(BaseModel):
|
||||
next_retry_at = Column(DateTime, nullable=True)
|
||||
last_attempt_at = Column(DateTime, nullable=True)
|
||||
last_error_reason = Column(String(512), nullable=True)
|
||||
excluded_host_ids = Column(JSON, nullable=True, default=list) # hosts that failed this allocation cycle
|
||||
|
||||
pod_id = Column(String(36), ForeignKey("container_pods.id"), nullable=True)
|
||||
pod = relationship("ContainerPod", backref="workloads")
|
||||
@@ -717,7 +718,7 @@ class Workload(BaseModel):
|
||||
"provisioning": ["running", "error", "pulling"],
|
||||
"running": ["stopping", "error","pending-deleted","dead","stopped", "pulling","host_failed","waiting-host"],
|
||||
"pending-allocation": ["running", "error","allocated","pending-deleted","failed-to-spawn", "pulling", "waiting-host"],
|
||||
"allocated":["pending-deleted","running","dead","stopped", "launch_failed","failed-to-spawn", "pulling"],
|
||||
"allocated":["pending-deleted","running","dead","stopped", "launch_failed","failed-to-spawn", "pulling", "error", "waiting-host"],
|
||||
"waiting-host": ["running", "allocated", "failed-to-spawn", "pending-allocation"],
|
||||
"pending-allocated": ["*"],
|
||||
"dead": ["*"],
|
||||
|
||||
@@ -49,6 +49,19 @@ class ExcludeAllDisabledHosts(SchedulingFilter):
|
||||
return filtered_hosts
|
||||
|
||||
|
||||
class ExcludeHosts(SchedulingFilter):
|
||||
"""Drop hosts previously tried (and failed) for this workload, to force re-scheduling elsewhere."""
|
||||
def __init__(self, excluded_ids):
|
||||
self.excluded = {str(h) for h in (excluded_ids or [])}
|
||||
|
||||
def apply(self, hosts, requirements=None):
|
||||
if not self.excluded:
|
||||
return hosts
|
||||
kept = [h for h in hosts if str(h.id) not in self.excluded]
|
||||
logger.debug(f"ExcludeHosts: removed {len(hosts) - len(kept)}/{len(hosts)} (excluded={self.excluded})")
|
||||
return kept
|
||||
|
||||
|
||||
class CapableHostsFilter(SchedulingFilter):
|
||||
def __init__(self, capabilities, requires_all=True):
|
||||
if isinstance(capabilities, str):
|
||||
|
||||
@@ -65,7 +65,7 @@ def retry_waiting_host_vms(self) -> None:
|
||||
from app.scheduling_filters import (
|
||||
ExcludeAllOfflineHosts, ExcludeAllDisabledHosts,
|
||||
LibvirtCapableHosts, MostAvailableCapacity, RandomizeHostsOrder,
|
||||
HasSufficientResources, GPUCapableHosts,
|
||||
HasSufficientResources, GPUCapableHosts, ExcludeHosts,
|
||||
)
|
||||
from app.compute.controller.server_group_affinity import apply_affinity_filters
|
||||
|
||||
@@ -131,6 +131,24 @@ def retry_waiting_host_vms(self) -> None:
|
||||
set_waiting_host_and_next_retry_vm(vm, reason="no-eligible-hosts-insufficient-gpu")
|
||||
continue
|
||||
|
||||
# Exclude hosts already tried this cycle; if that empties an otherwise-eligible
|
||||
# pool, every candidate has failed → terminal (bounded by retry policy).
|
||||
eligible = filtered_hosts
|
||||
filtered_hosts = ExcludeHosts(vm.excluded_host_ids).apply(eligible)
|
||||
if not filtered_hosts:
|
||||
vm.set_status("failed-to-spawn")
|
||||
vm.last_error_reason = "all-candidate-hosts-failed"
|
||||
vm.excluded_host_ids = []
|
||||
db.session.add(vm)
|
||||
AuditEntry.log_event(
|
||||
object=vm,
|
||||
action="terminal_failure",
|
||||
description=f"VM {vm.id} failed-to-spawn: all {len(eligible)} candidate host(s) previously failed"
|
||||
)
|
||||
db.session.commit()
|
||||
logger.warning("VM %s: all candidate hosts previously failed, marked failed-to-spawn", vm.id)
|
||||
continue
|
||||
|
||||
selected_host = filtered_hosts[0]
|
||||
vm.workload_host_id = selected_host.id
|
||||
vm.allocation_attempts = 0
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
from enum import Enum
|
||||
|
||||
|
||||
class ErrorType(str, Enum):
|
||||
# scheduling.dispatch
|
||||
WORKER_UNREACHABLE = "WORKER_UNREACHABLE"
|
||||
WORKER_REJECTED_TASK = "WORKER_REJECTED_TASK"
|
||||
PORT_CREATION_FAILED = "PORT_CREATION_FAILED"
|
||||
NSCONTROLLER_DISPATCH_FAILED = "NSCONTROLLER_DISPATCH_FAILED"
|
||||
DNS_UPDATE_FAILED = "DNS_UPDATE_FAILED"
|
||||
GPU_ALLOCATION_FAILED = "GPU_ALLOCATION_FAILED"
|
||||
METADATA_PERSIST_FAILED = "METADATA_PERSIST_FAILED"
|
||||
# scheduling.compute
|
||||
LIBVIRT_CONNECTION_FAILED = "LIBVIRT_CONNECTION_FAILED"
|
||||
LIBVIRT_DEFINE_FAILED = "LIBVIRT_DEFINE_FAILED"
|
||||
LIBVIRT_START_FAILED = "LIBVIRT_START_FAILED"
|
||||
KVM_NOT_AVAILABLE = "KVM_NOT_AVAILABLE"
|
||||
NO_MACHINE_TYPE = "NO_MACHINE_TYPE"
|
||||
VFIO_ATTACH_TIMEOUT = "VFIO_ATTACH_TIMEOUT"
|
||||
VFIO_ATTACH_FAILED = "VFIO_ATTACH_FAILED"
|
||||
# scheduling.storage
|
||||
LOCAL_VOLUME_DIR_FAILED = "LOCAL_VOLUME_DIR_FAILED"
|
||||
LOCAL_VOLUME_CREATE_FAILED = "LOCAL_VOLUME_CREATE_FAILED"
|
||||
VOLUME_CONVERT_FAILED = "VOLUME_CONVERT_FAILED"
|
||||
VOLUME_RESIZE_FAILED = "VOLUME_RESIZE_FAILED"
|
||||
LVM_VG_NOT_CONFIGURED = "LVM_VG_NOT_CONFIGURED"
|
||||
LVM_VOLUME_CREATE_FAILED = "LVM_VOLUME_CREATE_FAILED"
|
||||
IMAGE_DOWNLOAD_FAILED = "IMAGE_DOWNLOAD_FAILED"
|
||||
CEPH_INIT_FAILED = "CEPH_INIT_FAILED"
|
||||
CEPH_CONNECT_FAILED = "CEPH_CONNECT_FAILED"
|
||||
CEPH_EXPORT_FAILED = "CEPH_EXPORT_FAILED"
|
||||
# scheduling.network
|
||||
OVS_PORT_CREATE_FAILED = "OVS_PORT_CREATE_FAILED"
|
||||
VXLAN_SETUP_FAILED = "VXLAN_SETUP_FAILED"
|
||||
SDN_UPDATE_FAILED = "SDN_UPDATE_FAILED"
|
||||
|
||||
UNKNOWN = "UNKNOWN"
|
||||
|
||||
|
||||
class SchedulingError(Exception):
|
||||
"""Raised by dispatch or worker code for a known scheduling failure category."""
|
||||
|
||||
def __init__(self, error_type: ErrorType, message: str = ""):
|
||||
self.error_type = error_type
|
||||
super().__init__(f"{error_type.value}: {message}" if message else error_type.value)
|
||||
|
||||
def to_result(self) -> dict:
|
||||
return {
|
||||
"success": False,
|
||||
"response": str(self),
|
||||
"error_type": self.error_type.value,
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
{
|
||||
"WORKER_UNREACHABLE": "WebSocket server POST to worker failed — worker offline or disconnected",
|
||||
"WORKER_REJECTED_TASK": "Worker returned non-201 status on task acceptance",
|
||||
"PORT_CREATION_FAILED": "network_obj.create_port() raised an error during dispatch",
|
||||
"NSCONTROLLER_DISPATCH_FAILED": "ensure_nscontroller_on_host() failed during dispatch",
|
||||
"DNS_UPDATE_FAILED": "send_dns_updates_for_vdc() failed",
|
||||
"GPU_ALLOCATION_FAILED": "schedule_gpu_for_vm() returned None — no GPU slot on host",
|
||||
"METADATA_PERSIST_FAILED": "VmMetadata DB insert failed during dispatch",
|
||||
|
||||
"LIBVIRT_CONNECTION_FAILED": "Could not open qemu:///system connection on the host",
|
||||
"LIBVIRT_DEFINE_FAILED": "defineXML failed — host state, lock, or permission issue",
|
||||
"LIBVIRT_START_FAILED": "domain.create() failed after XML was defined",
|
||||
"KVM_NOT_AVAILABLE": "No KVM domain found in libvirt capabilities on this host",
|
||||
"NO_MACHINE_TYPE": "No pc-q35 machine type available — old QEMU on host",
|
||||
"VFIO_ATTACH_TIMEOUT": "VFIO bind operation hung for >45 seconds",
|
||||
"VFIO_ATTACH_FAILED": "vfio-pci driver bind failed for PCI passthrough device",
|
||||
|
||||
"LOCAL_VOLUME_DIR_FAILED": "Could not create /var/lib/.../volumes directory on host",
|
||||
"LOCAL_VOLUME_CREATE_FAILED": "qemu-img create failed — likely insufficient disk space",
|
||||
"VOLUME_CONVERT_FAILED": "qemu-img convert failed during image preparation",
|
||||
"VOLUME_RESIZE_FAILED": "qemu-img resize failed",
|
||||
"LVM_VG_NOT_CONFIGURED": "LVM_VG_NAME missing in worker_settings.json",
|
||||
"LVM_VOLUME_CREATE_FAILED": "lvcreate failed — VG full or LVM error",
|
||||
"IMAGE_DOWNLOAD_FAILED": "HTTP fetch of boot image failed — may be host network issue",
|
||||
"CEPH_INIT_FAILED": "Ceph processor could not initialize on host — check Ceph settings",
|
||||
"CEPH_CONNECT_FAILED": "rados/rbd connection to Ceph cluster failed",
|
||||
"CEPH_EXPORT_FAILED": "rbd export operation failed",
|
||||
|
||||
"OVS_PORT_CREATE_FAILED": "OVS port setup failed on host",
|
||||
"VXLAN_SETUP_FAILED": "VxLAN interface creation failed on host",
|
||||
"SDN_UPDATE_FAILED": "NSController SDN push failed",
|
||||
|
||||
"UNKNOWN": "Unclassified worker error — check last_error_reason for raw message"
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
|
||||
###Format:
|
||||
###
|
||||
ATTRIBUTE_ERROR_HANDLE = retry|skim;n|integer
|
||||
##n means to retry recursively across other hosts
|
||||
##skim means to skim the retry; just fails
|
||||
##retry will try against other servers other than failure.
|
||||
|
||||
[scheduling.dispatch]
|
||||
WORKER_UNREACHABLE = retry;n # websocket server POST failed
|
||||
WORKER_REJECTED_TASK = retry;3 # worker returned non-201
|
||||
PORT_CREATION_FAILED = retry;3 # network_obj.create_port() raised
|
||||
NSCONTROLLER_DISPATCH_FAILED= retry;3 # ensure_nscontroller_on_host failed
|
||||
DNS_UPDATE_FAILED = retry;n # send_dns_updates_for_vdc failed (non-blocking today)
|
||||
GPU_ALLOCATION_FAILED = retry;n # schedule_gpu_for_vm returned None
|
||||
METADATA_PERSIST_FAILED = retry;3 # VmMetadata DB insert failed
|
||||
|
||||
[scheduling.compute]
|
||||
LIBVIRT_CONNECTION_FAILED = retry;n # can't open qemu:///system
|
||||
LIBVIRT_DEFINE_FAILED = retry;3 # defineXML failed (host state, lock, etc.)
|
||||
LIBVIRT_START_FAILED = retry;3 # domain.create() failed
|
||||
KVM_NOT_AVAILABLE = retry;n # no KVM domain in libvirt caps on this host
|
||||
NO_MACHINE_TYPE = retry;n # no pc-q35 type (old QEMU on this host)
|
||||
VFIO_ATTACH_TIMEOUT = retry;3 # VFIO bind hung >45s
|
||||
VFIO_ATTACH_FAILED = retry;3 # vfio-pci driver bind failed
|
||||
|
||||
|
||||
[scheduling.storage]
|
||||
LOCAL_VOLUME_DIR_FAILED = retry;n # can't create /var/lib/.../volumes dir
|
||||
LOCAL_VOLUME_CREATE_FAILED = retry;n # qemu-img create failed (disk space)
|
||||
VOLUME_CONVERT_FAILED = retry;n # qemu-img convert failed
|
||||
VOLUME_RESIZE_FAILED = retry;n # qemu-img resize failed
|
||||
LVM_VG_NOT_CONFIGURED = retry;n # LVM_VG_NAME missing in worker_settings
|
||||
LVM_VOLUME_CREATE_FAILED = retry;n # lvcreate failed
|
||||
IMAGE_DOWNLOAD_FAILED = retry;3 # HTTP fetch failed (could be host network)
|
||||
CEPH_INIT_FAILED = skip;n # Ceph processor couldn't initialize on host
|
||||
CEPH_CONNECT_FAILED = skip;n # rados/rbd connection failed
|
||||
CEPH_EXPORT_FAILED = skip;3 # rbd export failed
|
||||
|
||||
|
||||
|
||||
[scheduling.network]
|
||||
OVS_PORT_CREATE_FAILED = retry;3 # OVS port setup failed on host
|
||||
VXLAN_SETUP_FAILED = retry;3 # VxLAN interface creation failed
|
||||
SDN_UPDATE_FAILED = retry;n # nscontroller SDN push failed
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Add excluded_host_ids to workloads for retry host exclusion
|
||||
|
||||
Revision ID: 0002_workload_excluded_hosts
|
||||
Revises: 0001_workload_vm_retry_fields
|
||||
Create Date: 2026-06-11
|
||||
"""
|
||||
from alembic import op
|
||||
import sqlalchemy as sa
|
||||
|
||||
revision = '0002_workload_excluded_hosts'
|
||||
down_revision = '0001_workload_vm_retry_fields'
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade():
|
||||
with op.batch_alter_table('workloads', schema=None) as batch_op:
|
||||
batch_op.add_column(sa.Column('excluded_host_ids', sa.JSON(), nullable=True))
|
||||
|
||||
|
||||
def downgrade():
|
||||
with op.batch_alter_table('workloads', schema=None) as batch_op:
|
||||
batch_op.drop_column('excluded_host_ids')
|
||||
@@ -12,6 +12,11 @@ from dotenv import load_dotenv
|
||||
load_dotenv()
|
||||
|
||||
|
||||
# runtime_urls.py lives at the repo root, so its directory IS the project root.
|
||||
# This is a far more stable anchor than walking ".." from a nested module.
|
||||
PROJECT_ROOT = os.path.dirname(os.path.abspath(__file__))
|
||||
|
||||
|
||||
def _env(name: str, default: str) -> str:
|
||||
return os.getenv(name, default)
|
||||
|
||||
@@ -57,6 +62,7 @@ SCHEDULER_RETRY_BASE_DELAY_SECONDS = _env_int("SCHEDULER_RETRY_BASE_DELAY_SECOND
|
||||
SCHEDULER_RETRY_BACKOFF_FACTOR = _env_float("SCHEDULER_RETRY_BACKOFF_FACTOR", 2.0)
|
||||
SCHEDULER_RETRY_JITTER_SECONDS = _env_int("SCHEDULER_RETRY_JITTER_SECONDS", 30)
|
||||
SCHEDULER_RETRY_MAX_DELAY_SECONDS = _env_int("SCHEDULER_RETRY_MAX_DELAY_SECONDS", 1800)
|
||||
SCHEDULER_CONF_PATH = _env("SCHEDULER_CONF_PATH", os.path.join(PROJECT_ROOT, "conf", "scheduler.conf"))
|
||||
CLOUDFLARE_API_TOKEN = _env("CLOUDFLARE_API_TOKEN", "")
|
||||
CLOUDFLARE_ACCOUNT_ID = _env("CLOUDFLARE_ACCOUNT_ID", "")
|
||||
CLOUDFLARE_ZONE_ID = _env("CLOUDFLARE_ZONE_ID", "")
|
||||
|
||||
@@ -42,22 +42,37 @@ def register_socketio_handlers(socketio):
|
||||
|
||||
logger.info(f"[{worker_id}] Task {task_id} acknowledged. Exec time: {task.execution_time:.2f}s")
|
||||
|
||||
# On VM creation failure, mark the workload as error
|
||||
# On VM creation failure, classify the error and apply retry/skip policy
|
||||
if task.task_type == "virtual-machine-create" and not result.get("success"):
|
||||
try:
|
||||
from app.compute.utils.retry import HandleRetry
|
||||
job = json.loads(task.job_details) if task.job_details else {}
|
||||
workload_id = job.get("virtual_machine_id")
|
||||
error_msg = str(result.get("response", "Worker reported failure"))[:512]
|
||||
error_type = result.get("error_type", "UNKNOWN")
|
||||
raw_response = result.get("response", "Worker reported failure")
|
||||
if isinstance(raw_response, dict):
|
||||
error_msg = str(raw_response.get("error", raw_response))
|
||||
else:
|
||||
error_msg = str(raw_response)
|
||||
if workload_id:
|
||||
workload = session.query(Workload).filter(Workload.id == workload_id).first()
|
||||
if workload:
|
||||
workload.last_error_reason = error_msg
|
||||
workload.set_status("error")
|
||||
outcome = HandleRetry().apply(session, workload, error_msg, error_type)
|
||||
logger.warning(
|
||||
f"[{worker_id}] VM workload {workload_id} marked as error: {error_msg}"
|
||||
f"[{worker_id}] VM workload {workload_id} "
|
||||
f"error_type={error_type} outcome={outcome}: {error_msg}"
|
||||
)
|
||||
except Exception as inner:
|
||||
logger.exception(f"[{worker_id}] Failed to update workload status on VM failure: {inner}")
|
||||
logger.exception(f"[{worker_id}] Failed to handle workload retry on VM failure: {inner}")
|
||||
|
||||
# On VM creation success, clear retry exclusion memory
|
||||
elif task.task_type == "virtual-machine-create" and result.get("success"):
|
||||
job = json.loads(task.job_details) if task.job_details else {}
|
||||
wid = job.get("virtual_machine_id")
|
||||
if wid:
|
||||
wl = session.query(Workload).filter(Workload.id == wid).first()
|
||||
if wl and wl.excluded_host_ids:
|
||||
wl.excluded_host_ids = []
|
||||
|
||||
# Check if this is a reconcile_and_delete task completion
|
||||
if task.task_type == "reconcile_and_delete" and result.get("success"):
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#Worker.py
|
||||
import functools
|
||||
from logger import logger
|
||||
from worker_tasks.error_types import SchedulingError
|
||||
from worker_tasks.ovs_bridge_scanner import OVSBridgeScannerTask
|
||||
from worker_tasks.pci_device_scanner import PCIDeviceScannerTask
|
||||
|
||||
@@ -388,9 +389,12 @@ class WorkerClient:
|
||||
logger.debug(f"Added Docker event to queue: launch_failed for container {failed_id}.")
|
||||
|
||||
|
||||
except SchedulingError as e:
|
||||
logger.error(f"Error processing task {task_id} [{e.error_type.value}]: {e}")
|
||||
await self.send_task_result(task_id, e.to_result(), task_worker_id)
|
||||
except Exception as e:
|
||||
logger.error(f"Error processing task {task_id}: {e}")
|
||||
await self.send_task_result(task_id, {"success": False, "response": str(e)}, task_worker_id)
|
||||
await self.send_task_result(task_id, {"success": False, "response": str(e), "error_type": "UNKNOWN"}, task_worker_id)
|
||||
|
||||
async def send_task_result(self, task_id, result, worker_id):
|
||||
"""Send task result back to the server, waiting for reconnection if needed."""
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
# Standalone — no Flask/SQLAlchemy imports so the worker process can use this safely.
|
||||
from enum import Enum
|
||||
|
||||
|
||||
class ErrorType(str, Enum):
|
||||
# scheduling.dispatch
|
||||
WORKER_UNREACHABLE = "WORKER_UNREACHABLE"
|
||||
WORKER_REJECTED_TASK = "WORKER_REJECTED_TASK"
|
||||
PORT_CREATION_FAILED = "PORT_CREATION_FAILED"
|
||||
NSCONTROLLER_DISPATCH_FAILED = "NSCONTROLLER_DISPATCH_FAILED"
|
||||
DNS_UPDATE_FAILED = "DNS_UPDATE_FAILED"
|
||||
GPU_ALLOCATION_FAILED = "GPU_ALLOCATION_FAILED"
|
||||
METADATA_PERSIST_FAILED = "METADATA_PERSIST_FAILED"
|
||||
# scheduling.compute
|
||||
LIBVIRT_CONNECTION_FAILED = "LIBVIRT_CONNECTION_FAILED"
|
||||
LIBVIRT_DEFINE_FAILED = "LIBVIRT_DEFINE_FAILED"
|
||||
LIBVIRT_START_FAILED = "LIBVIRT_START_FAILED"
|
||||
KVM_NOT_AVAILABLE = "KVM_NOT_AVAILABLE"
|
||||
NO_MACHINE_TYPE = "NO_MACHINE_TYPE"
|
||||
VFIO_ATTACH_TIMEOUT = "VFIO_ATTACH_TIMEOUT"
|
||||
VFIO_ATTACH_FAILED = "VFIO_ATTACH_FAILED"
|
||||
# scheduling.storage
|
||||
LOCAL_VOLUME_DIR_FAILED = "LOCAL_VOLUME_DIR_FAILED"
|
||||
LOCAL_VOLUME_CREATE_FAILED = "LOCAL_VOLUME_CREATE_FAILED"
|
||||
VOLUME_CONVERT_FAILED = "VOLUME_CONVERT_FAILED"
|
||||
VOLUME_RESIZE_FAILED = "VOLUME_RESIZE_FAILED"
|
||||
LVM_VG_NOT_CONFIGURED = "LVM_VG_NOT_CONFIGURED"
|
||||
LVM_VOLUME_CREATE_FAILED = "LVM_VOLUME_CREATE_FAILED"
|
||||
IMAGE_DOWNLOAD_FAILED = "IMAGE_DOWNLOAD_FAILED"
|
||||
CEPH_INIT_FAILED = "CEPH_INIT_FAILED"
|
||||
CEPH_CONNECT_FAILED = "CEPH_CONNECT_FAILED"
|
||||
CEPH_EXPORT_FAILED = "CEPH_EXPORT_FAILED"
|
||||
# scheduling.network
|
||||
OVS_PORT_CREATE_FAILED = "OVS_PORT_CREATE_FAILED"
|
||||
VXLAN_SETUP_FAILED = "VXLAN_SETUP_FAILED"
|
||||
SDN_UPDATE_FAILED = "SDN_UPDATE_FAILED"
|
||||
|
||||
UNKNOWN = "UNKNOWN"
|
||||
|
||||
|
||||
class SchedulingError(Exception):
|
||||
"""Raised by worker task code for a known scheduling failure category.
|
||||
|
||||
The error_type value is sent back to the server as part of the ack result
|
||||
so HandleRetry can look up the policy in scheduler.conf.
|
||||
"""
|
||||
|
||||
def __init__(self, error_type: ErrorType, message: str = ""):
|
||||
self.error_type = error_type
|
||||
super().__init__(f"{error_type.value}: {message}" if message else error_type.value)
|
||||
|
||||
def to_result(self) -> dict:
|
||||
return {
|
||||
"success": False,
|
||||
"response": str(self),
|
||||
"error_type": self.error_type.value,
|
||||
}
|
||||
@@ -11,6 +11,7 @@ from logging import Logger
|
||||
import xml.etree.ElementTree as ET
|
||||
from xml.dom import minidom
|
||||
|
||||
from worker_tasks.error_types import ErrorType, SchedulingError
|
||||
from worker_tasks.volumes import VolumeProcessor
|
||||
from settings import settings # Import global settings
|
||||
from worker_tasks.libvirt_xml_builder import _generate_xml_from_config
|
||||
@@ -103,6 +104,8 @@ class LibvirtVirtualMachineTask:
|
||||
# Pass only the volumes list to the processor
|
||||
self.processed_volume_paths = self.volume_processor.process_volumes(self.virtual_machine_config["volumes"])
|
||||
self.logger.info(f"Volume processing complete. Paths: {self.processed_volume_paths}")
|
||||
except SchedulingError:
|
||||
raise # already classified — preserve error_type for HandleRetry
|
||||
except Exception as e:
|
||||
self.logger.error(f"Volume processing failed: {e}")
|
||||
raise ValueError(f"Volume processing failed: {e}") # Re-raise to stop task execution
|
||||
@@ -137,11 +140,13 @@ class LibvirtVirtualMachineTask:
|
||||
try:
|
||||
with libvirt.open("qemu:///system") as conn:
|
||||
if conn is None:
|
||||
raise RuntimeError("Failed to connect to libvirt.")
|
||||
raise SchedulingError(ErrorType.LIBVIRT_CONNECTION_FAILED, "Failed to connect to libvirt")
|
||||
caps_xml = conn.getCapabilities()
|
||||
except SchedulingError:
|
||||
raise
|
||||
except libvirt.libvirtError as e:
|
||||
self.logger.error(f"Libvirt connection error: {e}")
|
||||
raise RuntimeError(f"Failed to connect to libvirt: {e}")
|
||||
raise SchedulingError(ErrorType.LIBVIRT_CONNECTION_FAILED, str(e))
|
||||
|
||||
# Parse capabilities XML
|
||||
try:
|
||||
@@ -151,25 +156,25 @@ class LibvirtVirtualMachineTask:
|
||||
arch_elem = root.find(".//guest/arch[@name='x86_64']")
|
||||
if not arch_elem:
|
||||
self.logger.error("x86_64 architecture not found in libvirt capabilities.")
|
||||
raise ValueError("x86_64 architecture not supported by the host.")
|
||||
|
||||
raise SchedulingError(ErrorType.KVM_NOT_AVAILABLE, "x86_64 architecture not supported by this host")
|
||||
|
||||
# Check for KVM domain
|
||||
has_kvm = any(domain.get("type") == "kvm" for domain in arch_elem.findall("domain"))
|
||||
self.logger.debug(f"KVM domain found: {has_kvm}")
|
||||
|
||||
|
||||
if not has_kvm:
|
||||
self.logger.error("KVM domain not found within x86_64 architecture.")
|
||||
raise ValueError("KVM domain not supported for x86_64.")
|
||||
|
||||
raise SchedulingError(ErrorType.KVM_NOT_AVAILABLE, "KVM domain not found in libvirt capabilities")
|
||||
|
||||
# Find all pc-q35 machine types
|
||||
q35_machines = [
|
||||
machine.text for machine in arch_elem.findall("machine")
|
||||
if machine.text and machine.text.startswith("pc-q35-")
|
||||
]
|
||||
|
||||
|
||||
if not q35_machines:
|
||||
self.logger.error("No 'pc-q35' machine types available for KVM.")
|
||||
raise ValueError("No 'pc-q35' machine types available for KVM.")
|
||||
raise SchedulingError(ErrorType.NO_MACHINE_TYPE, "No pc-q35 machine types available on this host")
|
||||
|
||||
# Use a more robust version comparison with packaging.version
|
||||
from packaging import version
|
||||
@@ -223,8 +228,9 @@ class LibvirtVirtualMachineTask:
|
||||
self.logger.error(
|
||||
f"VFIO attach timed out after {VFIO_OP_TIMEOUT}s for {addr}"
|
||||
)
|
||||
raise RuntimeError(
|
||||
f"VFIO attach timed out for {addr}"
|
||||
raise SchedulingError(
|
||||
ErrorType.VFIO_ATTACH_TIMEOUT,
|
||||
f"VFIO attach timed out after {VFIO_OP_TIMEOUT}s for {addr}",
|
||||
)
|
||||
except Exception as e:
|
||||
self.logger.error(f"VFIO attach failed for {addr}: {e}")
|
||||
@@ -308,7 +314,7 @@ class LibvirtVirtualMachineTask:
|
||||
try:
|
||||
conn = libvirt.open("qemu:///system") # Connect to the hypervisor
|
||||
if conn is None:
|
||||
raise RuntimeError("Failed to open connection to libvirt.")
|
||||
raise SchedulingError(ErrorType.LIBVIRT_CONNECTION_FAILED, "Failed to open connection to libvirt")
|
||||
|
||||
# Lookup VirtualMachine by name
|
||||
self.logger.info("Looking up VirtualMachine Name")
|
||||
@@ -386,9 +392,15 @@ class LibvirtVirtualMachineTask:
|
||||
gpu_pci_addresses = (self.virtual_machine_config or {}).get("gpu_pci_addresses", [])
|
||||
if gpu_pci_addresses:
|
||||
self._attach_gpus_vfio(gpu_pci_addresses)
|
||||
domain = conn.defineXML(self.xml_config)
|
||||
try:
|
||||
domain = conn.defineXML(self.xml_config)
|
||||
except libvirt.libvirtError as e:
|
||||
raise SchedulingError(ErrorType.LIBVIRT_DEFINE_FAILED, str(e))
|
||||
if self.desired_state == "running":
|
||||
domain.create() # Start the VirtualMachine
|
||||
try:
|
||||
domain.create() # Start the VirtualMachine
|
||||
except libvirt.libvirtError as e:
|
||||
raise SchedulingError(ErrorType.LIBVIRT_START_FAILED, str(e))
|
||||
response_payload["response"]["status"] = "VirtualMachine created and running"
|
||||
else:
|
||||
response_payload["response"]["status"] = "VirtualMachine created and stopped"
|
||||
@@ -420,6 +432,8 @@ class LibvirtVirtualMachineTask:
|
||||
raise ValueError(f"Unsupported action: {self.action}")
|
||||
self.logger.info(f"Action '{self.action}' completed successfully for VirtualMachine '{self.virtual_machine_id}'.")
|
||||
|
||||
except SchedulingError:
|
||||
raise # Let workerClient classify and send back with error_type
|
||||
except Exception as e:
|
||||
self.logger.error(f"Error executing LibvirtVirtualMachineTask: {e}")
|
||||
response_payload["response"]["error"] = str(e)
|
||||
|
||||
@@ -11,6 +11,7 @@ from typing import Dict, List, Optional, Union, Any
|
||||
# from .volumes_ceph import CephVolumeProcessor
|
||||
# Import settings to get Ceph configuration
|
||||
from settings import settings
|
||||
from worker_tasks.error_types import ErrorType, SchedulingError
|
||||
|
||||
class VolumeProcessor:
|
||||
"""
|
||||
@@ -602,8 +603,10 @@ class VolumeProcessor:
|
||||
|
||||
self.logger.info(f"Creating LVM volume {volume_name} (ID: {volume_id})")
|
||||
|
||||
# Read VG name from settings — raises KeyError if not configured in worker_settings.json
|
||||
vg_name = settings.get_value("LVM_VG_NAME")
|
||||
try:
|
||||
vg_name = settings.get_value("LVM_VG_NAME")
|
||||
except KeyError:
|
||||
raise SchedulingError(ErrorType.LVM_VG_NOT_CONFIGURED, "LVM_VG_NAME not set in worker_settings.json")
|
||||
lv_name = f"{volume_name}_{volume_id}"
|
||||
|
||||
# Canonical LVM block device path: /dev/<vg>/<lv>
|
||||
@@ -617,7 +620,10 @@ class VolumeProcessor:
|
||||
|
||||
# Create logical volume
|
||||
self.logger.info(f"Creating LVM logical volume {lv_name} in VG {vg_name}")
|
||||
subprocess.run(['lvcreate', '-L', f"{size_gb}G", '-n', lv_name, vg_name], check=True)
|
||||
try:
|
||||
subprocess.run(['lvcreate', '-L', f"{size_gb}G", '-n', lv_name, vg_name], check=True)
|
||||
except subprocess.CalledProcessError as e:
|
||||
raise SchedulingError(ErrorType.LVM_VOLUME_CREATE_FAILED, f"lvcreate failed: {e}")
|
||||
self.logger.debug("LVM volume created")
|
||||
|
||||
# Handle different source types
|
||||
@@ -712,7 +718,7 @@ class VolumeProcessor:
|
||||
elif volume_type == 'ceph':
|
||||
# Ensure Ceph processor is initialized (lazy initialization)
|
||||
if not self.ceph_processor and not self._initialize_ceph_processor():
|
||||
raise ConnectionError("Ceph processor could not be initialized. Check Ceph settings and connectivity.")
|
||||
raise SchedulingError(ErrorType.CEPH_INIT_FAILED, "Ceph processor could not be initialized — check Ceph settings and connectivity")
|
||||
# Pass self (VolumeProcessor instance) to handle downloads if needed by Ceph processor
|
||||
result = self.ceph_processor.create_volume(volume_config, self)
|
||||
self.logger.debug(f"Result from ceph create_volume {result}")
|
||||
@@ -721,12 +727,12 @@ class VolumeProcessor:
|
||||
|
||||
self.logger.info(f"Volume processing completed successfully. Result: {result}")
|
||||
return result
|
||||
except (subprocess.CalledProcessError, ValueError, IOError, ConnectionError, FileNotFoundError) as e: # Added ConnectionError
|
||||
except SchedulingError:
|
||||
raise # Already classified — let workerClient send error_type
|
||||
except (subprocess.CalledProcessError, ValueError, IOError, ConnectionError, FileNotFoundError) as e:
|
||||
self.logger.error(f"Error processing volume {volume_config.get('id', 'unknown')}: {str(e)}")
|
||||
# The CephVolumeProcessor might raise rbd.Error or rados.Error, catch them if needed
|
||||
# Or rely on it converting them to standard exceptions like ConnectionError/IOError
|
||||
raise
|
||||
except Exception as e: # Catch unexpected errors
|
||||
except Exception as e:
|
||||
self.logger.error(f"Error processing volume: {str(e)}")
|
||||
raise
|
||||
|
||||
|
||||
Reference in New Issue
Block a user