Feat: Retry diff Hosts on Workload failure

This commit is contained in:
2026-06-11 13:38:56 +05:45
parent 38f5601945
commit c486b35e2a
14 changed files with 458 additions and 31 deletions
+139
View File
@@ -0,0 +1,139 @@
import configparser
import io
import logging
from datetime import datetime, timedelta, timezone
from typing import Optional, Tuple
from runtime_urls import (
SCHEDULER_CONF_PATH as _CONF_PATH,
SCHEDULER_RETRY_BASE_DELAY_SECONDS,
SCHEDULER_RETRY_BACKOFF_FACTOR,
SCHEDULER_RETRY_MAX_DELAY_SECONDS,
)
logger = logging.getLogger("websocket_server")
# Module-level cache so the file is only parsed once per process.
_policy_cache: Optional[dict] = None
def _load_policy(conf_path: str = _CONF_PATH) -> dict:
"""Parse scheduler.conf and return {ERROR_TYPE_KEY: (action, max_tries)}."""
global _policy_cache
if _policy_cache is not None:
return _policy_cache
# Pre-process the file so configparser handles it correctly:
# - Drop option-like (key=value) lines that appear before the first [section]
# (the file has a format-doc line at the top that would confuse configparser)
# - Strip leading whitespace from real option lines inside sections
# (configparser treats indented lines as value continuations otherwise)
lines: list[str] = []
in_section = False
try:
with open(conf_path) as f:
for line in f:
stripped = line.lstrip()
if stripped.startswith("["):
in_section = True
looks_like_option = "=" in stripped and not stripped.startswith(("#", ";", "["))
if not in_section:
# Before first [section]: keep comments and blanks, drop option lines
if not looks_like_option:
lines.append(line)
else:
lines.append(stripped if looks_like_option else line)
except FileNotFoundError:
logger.warning("scheduler.conf not found at %s — all errors default to skip", conf_path)
_policy_cache = {}
return _policy_cache
cfg = configparser.RawConfigParser(inline_comment_prefixes=("#",))
cfg.optionxform = str # preserve case
cfg.read_file(io.StringIO("".join(lines)))
policy: dict = {}
for section in cfg.sections():
for key, raw_value in cfg.items(section):
error_key = key.strip().upper()
action_str = raw_value.strip().split()[0] if raw_value.strip() else "skip"
parts = action_str.split(";")
action = parts[0].lower()
if action in ("skip", "skim"):
policy[error_key] = ("skip", 0)
elif action == "retry":
count_str = parts[1] if len(parts) > 1 else "n"
max_tries: Optional[int] = None if count_str == "n" else int(count_str)
policy[error_key] = ("retry", max_tries)
else:
policy[error_key] = ("skip", 0)
_policy_cache = policy
return policy
class HandleRetry:
"""
Look up the scheduler.conf policy for a given ErrorType and apply it to a
Workload ORM object within an existing DB session.
Usage (inside ack.py's get_db_session context):
outcome = HandleRetry().apply(session, workload, error_msg, error_type)
# outcome is "retry" or "skip"
"""
def __init__(self, conf_path: str = _CONF_PATH):
self._policy = _load_policy(conf_path)
def get_policy(self, error_type: str) -> Tuple[str, Optional[int]]:
"""Return (action, max_tries) for the given error_type string."""
key = (error_type or "UNKNOWN").strip().upper()
return self._policy.get(key, ("skip", 0))
def apply(self, session, workload, error_msg: str, error_type: str) -> str:
"""Apply the scheduler.conf policy to a failed workload in-place.
skip → mark error, no host retry.
retry;N → exclude the failed host and re-queue; failed-to-spawn once >N hosts tried.
retry;n → same but unbounded (only the empty-host-pool check in the beat task stops it).
Returns "retry" (re-queued as waiting-host) or "skip" (terminal).
"""
action, policy_max = self.get_policy(error_type)
workload.allocation_attempts = (workload.allocation_attempts or 0) + 1
workload.last_error_reason = error_msg[:512]
workload.last_attempt_at = datetime.now(timezone.utc)
if action == "skip":
workload.excluded_host_ids = []
self._set_status(workload, "error")
logger.info("Workload %s error_type=%s policy=skip → error", workload.id, error_type)
return "skip"
excluded = list(workload.excluded_host_ids or [])
if workload.workload_host_id and workload.workload_host_id not in excluded:
excluded.append(workload.workload_host_id)
workload.excluded_host_ids = excluded
if policy_max is not None and len(excluded) > policy_max:
workload.excluded_host_ids = []
self._set_status(workload, "failed-to-spawn")
logger.warning("Workload %s error_type=%s policy=retry;%d → failed-to-spawn (%d hosts tried)",
workload.id, error_type, policy_max, len(excluded))
return "skip"
self._set_status(workload, "waiting-host")
delay = int(min(SCHEDULER_RETRY_BASE_DELAY_SECONDS * (SCHEDULER_RETRY_BACKOFF_FACTOR ** (len(excluded) - 1)),
SCHEDULER_RETRY_MAX_DELAY_SECONDS))
workload.next_retry_at = datetime.now(timezone.utc) + timedelta(seconds=delay)
logger.info("Workload %s error_type=%s policy=retry;%s → waiting-host (excluded %d, retry in %ds)",
workload.id, error_type, policy_max if policy_max is not None else "n", len(excluded), delay)
return "retry"
@staticmethod
def _set_status(workload, status: str) -> None:
try:
workload.set_status(status)
except Exception:
workload._status = status
+2 -1
View File
@@ -640,6 +640,7 @@ class Workload(BaseModel):
next_retry_at = Column(DateTime, nullable=True)
last_attempt_at = Column(DateTime, nullable=True)
last_error_reason = Column(String(512), nullable=True)
excluded_host_ids = Column(JSON, nullable=True, default=list) # hosts that failed this allocation cycle
pod_id = Column(String(36), ForeignKey("container_pods.id"), nullable=True)
pod = relationship("ContainerPod", backref="workloads")
@@ -717,7 +718,7 @@ class Workload(BaseModel):
"provisioning": ["running", "error", "pulling"],
"running": ["stopping", "error","pending-deleted","dead","stopped", "pulling","host_failed","waiting-host"],
"pending-allocation": ["running", "error","allocated","pending-deleted","failed-to-spawn", "pulling", "waiting-host"],
"allocated":["pending-deleted","running","dead","stopped", "launch_failed","failed-to-spawn", "pulling"],
"allocated":["pending-deleted","running","dead","stopped", "launch_failed","failed-to-spawn", "pulling", "error", "waiting-host"],
"waiting-host": ["running", "allocated", "failed-to-spawn", "pending-allocation"],
"pending-allocated": ["*"],
"dead": ["*"],
+13
View File
@@ -49,6 +49,19 @@ class ExcludeAllDisabledHosts(SchedulingFilter):
return filtered_hosts
class ExcludeHosts(SchedulingFilter):
"""Drop hosts previously tried (and failed) for this workload, to force re-scheduling elsewhere."""
def __init__(self, excluded_ids):
self.excluded = {str(h) for h in (excluded_ids or [])}
def apply(self, hosts, requirements=None):
if not self.excluded:
return hosts
kept = [h for h in hosts if str(h.id) not in self.excluded]
logger.debug(f"ExcludeHosts: removed {len(hosts) - len(kept)}/{len(hosts)} (excluded={self.excluded})")
return kept
class CapableHostsFilter(SchedulingFilter):
def __init__(self, capabilities, requires_all=True):
if isinstance(capabilities, str):
+19 -1
View File
@@ -65,7 +65,7 @@ def retry_waiting_host_vms(self) -> None:
from app.scheduling_filters import (
ExcludeAllOfflineHosts, ExcludeAllDisabledHosts,
LibvirtCapableHosts, MostAvailableCapacity, RandomizeHostsOrder,
HasSufficientResources, GPUCapableHosts,
HasSufficientResources, GPUCapableHosts, ExcludeHosts,
)
from app.compute.controller.server_group_affinity import apply_affinity_filters
@@ -131,6 +131,24 @@ def retry_waiting_host_vms(self) -> None:
set_waiting_host_and_next_retry_vm(vm, reason="no-eligible-hosts-insufficient-gpu")
continue
# Exclude hosts already tried this cycle; if that empties an otherwise-eligible
# pool, every candidate has failed → terminal (bounded by retry policy).
eligible = filtered_hosts
filtered_hosts = ExcludeHosts(vm.excluded_host_ids).apply(eligible)
if not filtered_hosts:
vm.set_status("failed-to-spawn")
vm.last_error_reason = "all-candidate-hosts-failed"
vm.excluded_host_ids = []
db.session.add(vm)
AuditEntry.log_event(
object=vm,
action="terminal_failure",
description=f"VM {vm.id} failed-to-spawn: all {len(eligible)} candidate host(s) previously failed"
)
db.session.commit()
logger.warning("VM %s: all candidate hosts previously failed, marked failed-to-spawn", vm.id)
continue
selected_host = filtered_hosts[0]
vm.workload_host_id = selected_host.id
vm.allocation_attempts = 0
+52
View File
@@ -0,0 +1,52 @@
from enum import Enum
class ErrorType(str, Enum):
# scheduling.dispatch
WORKER_UNREACHABLE = "WORKER_UNREACHABLE"
WORKER_REJECTED_TASK = "WORKER_REJECTED_TASK"
PORT_CREATION_FAILED = "PORT_CREATION_FAILED"
NSCONTROLLER_DISPATCH_FAILED = "NSCONTROLLER_DISPATCH_FAILED"
DNS_UPDATE_FAILED = "DNS_UPDATE_FAILED"
GPU_ALLOCATION_FAILED = "GPU_ALLOCATION_FAILED"
METADATA_PERSIST_FAILED = "METADATA_PERSIST_FAILED"
# scheduling.compute
LIBVIRT_CONNECTION_FAILED = "LIBVIRT_CONNECTION_FAILED"
LIBVIRT_DEFINE_FAILED = "LIBVIRT_DEFINE_FAILED"
LIBVIRT_START_FAILED = "LIBVIRT_START_FAILED"
KVM_NOT_AVAILABLE = "KVM_NOT_AVAILABLE"
NO_MACHINE_TYPE = "NO_MACHINE_TYPE"
VFIO_ATTACH_TIMEOUT = "VFIO_ATTACH_TIMEOUT"
VFIO_ATTACH_FAILED = "VFIO_ATTACH_FAILED"
# scheduling.storage
LOCAL_VOLUME_DIR_FAILED = "LOCAL_VOLUME_DIR_FAILED"
LOCAL_VOLUME_CREATE_FAILED = "LOCAL_VOLUME_CREATE_FAILED"
VOLUME_CONVERT_FAILED = "VOLUME_CONVERT_FAILED"
VOLUME_RESIZE_FAILED = "VOLUME_RESIZE_FAILED"
LVM_VG_NOT_CONFIGURED = "LVM_VG_NOT_CONFIGURED"
LVM_VOLUME_CREATE_FAILED = "LVM_VOLUME_CREATE_FAILED"
IMAGE_DOWNLOAD_FAILED = "IMAGE_DOWNLOAD_FAILED"
CEPH_INIT_FAILED = "CEPH_INIT_FAILED"
CEPH_CONNECT_FAILED = "CEPH_CONNECT_FAILED"
CEPH_EXPORT_FAILED = "CEPH_EXPORT_FAILED"
# scheduling.network
OVS_PORT_CREATE_FAILED = "OVS_PORT_CREATE_FAILED"
VXLAN_SETUP_FAILED = "VXLAN_SETUP_FAILED"
SDN_UPDATE_FAILED = "SDN_UPDATE_FAILED"
UNKNOWN = "UNKNOWN"
class SchedulingError(Exception):
"""Raised by dispatch or worker code for a known scheduling failure category."""
def __init__(self, error_type: ErrorType, message: str = ""):
self.error_type = error_type
super().__init__(f"{error_type.value}: {message}" if message else error_type.value)
def to_result(self) -> dict:
return {
"success": False,
"response": str(self),
"error_type": self.error_type.value,
}
+34
View File
@@ -0,0 +1,34 @@
{
"WORKER_UNREACHABLE": "WebSocket server POST to worker failed — worker offline or disconnected",
"WORKER_REJECTED_TASK": "Worker returned non-201 status on task acceptance",
"PORT_CREATION_FAILED": "network_obj.create_port() raised an error during dispatch",
"NSCONTROLLER_DISPATCH_FAILED": "ensure_nscontroller_on_host() failed during dispatch",
"DNS_UPDATE_FAILED": "send_dns_updates_for_vdc() failed",
"GPU_ALLOCATION_FAILED": "schedule_gpu_for_vm() returned None — no GPU slot on host",
"METADATA_PERSIST_FAILED": "VmMetadata DB insert failed during dispatch",
"LIBVIRT_CONNECTION_FAILED": "Could not open qemu:///system connection on the host",
"LIBVIRT_DEFINE_FAILED": "defineXML failed — host state, lock, or permission issue",
"LIBVIRT_START_FAILED": "domain.create() failed after XML was defined",
"KVM_NOT_AVAILABLE": "No KVM domain found in libvirt capabilities on this host",
"NO_MACHINE_TYPE": "No pc-q35 machine type available — old QEMU on host",
"VFIO_ATTACH_TIMEOUT": "VFIO bind operation hung for >45 seconds",
"VFIO_ATTACH_FAILED": "vfio-pci driver bind failed for PCI passthrough device",
"LOCAL_VOLUME_DIR_FAILED": "Could not create /var/lib/.../volumes directory on host",
"LOCAL_VOLUME_CREATE_FAILED": "qemu-img create failed — likely insufficient disk space",
"VOLUME_CONVERT_FAILED": "qemu-img convert failed during image preparation",
"VOLUME_RESIZE_FAILED": "qemu-img resize failed",
"LVM_VG_NOT_CONFIGURED": "LVM_VG_NAME missing in worker_settings.json",
"LVM_VOLUME_CREATE_FAILED": "lvcreate failed — VG full or LVM error",
"IMAGE_DOWNLOAD_FAILED": "HTTP fetch of boot image failed — may be host network issue",
"CEPH_INIT_FAILED": "Ceph processor could not initialize on host — check Ceph settings",
"CEPH_CONNECT_FAILED": "rados/rbd connection to Ceph cluster failed",
"CEPH_EXPORT_FAILED": "rbd export operation failed",
"OVS_PORT_CREATE_FAILED": "OVS port setup failed on host",
"VXLAN_SETUP_FAILED": "VxLAN interface creation failed on host",
"SDN_UPDATE_FAILED": "NSController SDN push failed",
"UNKNOWN": "Unclassified worker error — check last_error_reason for raw message"
}
+45
View File
@@ -0,0 +1,45 @@
###Format:
###
ATTRIBUTE_ERROR_HANDLE = retry|skim;n|integer
##n means to retry recursively across other hosts
##skim means to skim the retry; just fails
##retry will try against other servers other than failure.
[scheduling.dispatch]
WORKER_UNREACHABLE = retry;n # websocket server POST failed
WORKER_REJECTED_TASK = retry;3 # worker returned non-201
PORT_CREATION_FAILED = retry;3 # network_obj.create_port() raised
NSCONTROLLER_DISPATCH_FAILED= retry;3 # ensure_nscontroller_on_host failed
DNS_UPDATE_FAILED = retry;n # send_dns_updates_for_vdc failed (non-blocking today)
GPU_ALLOCATION_FAILED = retry;n # schedule_gpu_for_vm returned None
METADATA_PERSIST_FAILED = retry;3 # VmMetadata DB insert failed
[scheduling.compute]
LIBVIRT_CONNECTION_FAILED = retry;n # can't open qemu:///system
LIBVIRT_DEFINE_FAILED = retry;3 # defineXML failed (host state, lock, etc.)
LIBVIRT_START_FAILED = retry;3 # domain.create() failed
KVM_NOT_AVAILABLE = retry;n # no KVM domain in libvirt caps on this host
NO_MACHINE_TYPE = retry;n # no pc-q35 type (old QEMU on this host)
VFIO_ATTACH_TIMEOUT = retry;3 # VFIO bind hung >45s
VFIO_ATTACH_FAILED = retry;3 # vfio-pci driver bind failed
[scheduling.storage]
LOCAL_VOLUME_DIR_FAILED = retry;n # can't create /var/lib/.../volumes dir
LOCAL_VOLUME_CREATE_FAILED = retry;n # qemu-img create failed (disk space)
VOLUME_CONVERT_FAILED = retry;n # qemu-img convert failed
VOLUME_RESIZE_FAILED = retry;n # qemu-img resize failed
LVM_VG_NOT_CONFIGURED = retry;n # LVM_VG_NAME missing in worker_settings
LVM_VOLUME_CREATE_FAILED = retry;n # lvcreate failed
IMAGE_DOWNLOAD_FAILED = retry;3 # HTTP fetch failed (could be host network)
CEPH_INIT_FAILED = skip;n # Ceph processor couldn't initialize on host
CEPH_CONNECT_FAILED = skip;n # rados/rbd connection failed
CEPH_EXPORT_FAILED = skip;3 # rbd export failed
[scheduling.network]
OVS_PORT_CREATE_FAILED = retry;3 # OVS port setup failed on host
VXLAN_SETUP_FAILED = retry;3 # VxLAN interface creation failed
SDN_UPDATE_FAILED = retry;n # nscontroller SDN push failed
@@ -0,0 +1,23 @@
"""Add excluded_host_ids to workloads for retry host exclusion
Revision ID: 0002_workload_excluded_hosts
Revises: 0001_workload_vm_retry_fields
Create Date: 2026-06-11
"""
from alembic import op
import sqlalchemy as sa
revision = '0002_workload_excluded_hosts'
down_revision = '0001_workload_vm_retry_fields'
branch_labels = None
depends_on = None
def upgrade():
with op.batch_alter_table('workloads', schema=None) as batch_op:
batch_op.add_column(sa.Column('excluded_host_ids', sa.JSON(), nullable=True))
def downgrade():
with op.batch_alter_table('workloads', schema=None) as batch_op:
batch_op.drop_column('excluded_host_ids')
+6
View File
@@ -12,6 +12,11 @@ from dotenv import load_dotenv
load_dotenv()
# runtime_urls.py lives at the repo root, so its directory IS the project root.
# This is a far more stable anchor than walking ".." from a nested module.
PROJECT_ROOT = os.path.dirname(os.path.abspath(__file__))
def _env(name: str, default: str) -> str:
return os.getenv(name, default)
@@ -57,6 +62,7 @@ SCHEDULER_RETRY_BASE_DELAY_SECONDS = _env_int("SCHEDULER_RETRY_BASE_DELAY_SECOND
SCHEDULER_RETRY_BACKOFF_FACTOR = _env_float("SCHEDULER_RETRY_BACKOFF_FACTOR", 2.0)
SCHEDULER_RETRY_JITTER_SECONDS = _env_int("SCHEDULER_RETRY_JITTER_SECONDS", 30)
SCHEDULER_RETRY_MAX_DELAY_SECONDS = _env_int("SCHEDULER_RETRY_MAX_DELAY_SECONDS", 1800)
SCHEDULER_CONF_PATH = _env("SCHEDULER_CONF_PATH", os.path.join(PROJECT_ROOT, "conf", "scheduler.conf"))
CLOUDFLARE_API_TOKEN = _env("CLOUDFLARE_API_TOKEN", "")
CLOUDFLARE_ACCOUNT_ID = _env("CLOUDFLARE_ACCOUNT_ID", "")
CLOUDFLARE_ZONE_ID = _env("CLOUDFLARE_ZONE_ID", "")
+21 -6
View File
@@ -42,22 +42,37 @@ def register_socketio_handlers(socketio):
logger.info(f"[{worker_id}] Task {task_id} acknowledged. Exec time: {task.execution_time:.2f}s")
# On VM creation failure, mark the workload as error
# On VM creation failure, classify the error and apply retry/skip policy
if task.task_type == "virtual-machine-create" and not result.get("success"):
try:
from app.compute.utils.retry import HandleRetry
job = json.loads(task.job_details) if task.job_details else {}
workload_id = job.get("virtual_machine_id")
error_msg = str(result.get("response", "Worker reported failure"))[:512]
error_type = result.get("error_type", "UNKNOWN")
raw_response = result.get("response", "Worker reported failure")
if isinstance(raw_response, dict):
error_msg = str(raw_response.get("error", raw_response))
else:
error_msg = str(raw_response)
if workload_id:
workload = session.query(Workload).filter(Workload.id == workload_id).first()
if workload:
workload.last_error_reason = error_msg
workload.set_status("error")
outcome = HandleRetry().apply(session, workload, error_msg, error_type)
logger.warning(
f"[{worker_id}] VM workload {workload_id} marked as error: {error_msg}"
f"[{worker_id}] VM workload {workload_id} "
f"error_type={error_type} outcome={outcome}: {error_msg}"
)
except Exception as inner:
logger.exception(f"[{worker_id}] Failed to update workload status on VM failure: {inner}")
logger.exception(f"[{worker_id}] Failed to handle workload retry on VM failure: {inner}")
# On VM creation success, clear retry exclusion memory
elif task.task_type == "virtual-machine-create" and result.get("success"):
job = json.loads(task.job_details) if task.job_details else {}
wid = job.get("virtual_machine_id")
if wid:
wl = session.query(Workload).filter(Workload.id == wid).first()
if wl and wl.excluded_host_ids:
wl.excluded_host_ids = []
# Check if this is a reconcile_and_delete task completion
if task.task_type == "reconcile_and_delete" and result.get("success"):
+5 -1
View File
@@ -1,6 +1,7 @@
#Worker.py
import functools
from logger import logger
from worker_tasks.error_types import SchedulingError
from worker_tasks.ovs_bridge_scanner import OVSBridgeScannerTask
from worker_tasks.pci_device_scanner import PCIDeviceScannerTask
@@ -388,9 +389,12 @@ class WorkerClient:
logger.debug(f"Added Docker event to queue: launch_failed for container {failed_id}.")
except SchedulingError as e:
logger.error(f"Error processing task {task_id} [{e.error_type.value}]: {e}")
await self.send_task_result(task_id, e.to_result(), task_worker_id)
except Exception as e:
logger.error(f"Error processing task {task_id}: {e}")
await self.send_task_result(task_id, {"success": False, "response": str(e)}, task_worker_id)
await self.send_task_result(task_id, {"success": False, "response": str(e), "error_type": "UNKNOWN"}, task_worker_id)
async def send_task_result(self, task_id, result, worker_id):
"""Send task result back to the server, waiting for reconnection if needed."""
+57
View File
@@ -0,0 +1,57 @@
# Standalone — no Flask/SQLAlchemy imports so the worker process can use this safely.
from enum import Enum
class ErrorType(str, Enum):
# scheduling.dispatch
WORKER_UNREACHABLE = "WORKER_UNREACHABLE"
WORKER_REJECTED_TASK = "WORKER_REJECTED_TASK"
PORT_CREATION_FAILED = "PORT_CREATION_FAILED"
NSCONTROLLER_DISPATCH_FAILED = "NSCONTROLLER_DISPATCH_FAILED"
DNS_UPDATE_FAILED = "DNS_UPDATE_FAILED"
GPU_ALLOCATION_FAILED = "GPU_ALLOCATION_FAILED"
METADATA_PERSIST_FAILED = "METADATA_PERSIST_FAILED"
# scheduling.compute
LIBVIRT_CONNECTION_FAILED = "LIBVIRT_CONNECTION_FAILED"
LIBVIRT_DEFINE_FAILED = "LIBVIRT_DEFINE_FAILED"
LIBVIRT_START_FAILED = "LIBVIRT_START_FAILED"
KVM_NOT_AVAILABLE = "KVM_NOT_AVAILABLE"
NO_MACHINE_TYPE = "NO_MACHINE_TYPE"
VFIO_ATTACH_TIMEOUT = "VFIO_ATTACH_TIMEOUT"
VFIO_ATTACH_FAILED = "VFIO_ATTACH_FAILED"
# scheduling.storage
LOCAL_VOLUME_DIR_FAILED = "LOCAL_VOLUME_DIR_FAILED"
LOCAL_VOLUME_CREATE_FAILED = "LOCAL_VOLUME_CREATE_FAILED"
VOLUME_CONVERT_FAILED = "VOLUME_CONVERT_FAILED"
VOLUME_RESIZE_FAILED = "VOLUME_RESIZE_FAILED"
LVM_VG_NOT_CONFIGURED = "LVM_VG_NOT_CONFIGURED"
LVM_VOLUME_CREATE_FAILED = "LVM_VOLUME_CREATE_FAILED"
IMAGE_DOWNLOAD_FAILED = "IMAGE_DOWNLOAD_FAILED"
CEPH_INIT_FAILED = "CEPH_INIT_FAILED"
CEPH_CONNECT_FAILED = "CEPH_CONNECT_FAILED"
CEPH_EXPORT_FAILED = "CEPH_EXPORT_FAILED"
# scheduling.network
OVS_PORT_CREATE_FAILED = "OVS_PORT_CREATE_FAILED"
VXLAN_SETUP_FAILED = "VXLAN_SETUP_FAILED"
SDN_UPDATE_FAILED = "SDN_UPDATE_FAILED"
UNKNOWN = "UNKNOWN"
class SchedulingError(Exception):
"""Raised by worker task code for a known scheduling failure category.
The error_type value is sent back to the server as part of the ack result
so HandleRetry can look up the policy in scheduler.conf.
"""
def __init__(self, error_type: ErrorType, message: str = ""):
self.error_type = error_type
super().__init__(f"{error_type.value}: {message}" if message else error_type.value)
def to_result(self) -> dict:
return {
"success": False,
"response": str(self),
"error_type": self.error_type.value,
}
+28 -14
View File
@@ -11,6 +11,7 @@ from logging import Logger
import xml.etree.ElementTree as ET
from xml.dom import minidom
from worker_tasks.error_types import ErrorType, SchedulingError
from worker_tasks.volumes import VolumeProcessor
from settings import settings # Import global settings
from worker_tasks.libvirt_xml_builder import _generate_xml_from_config
@@ -103,6 +104,8 @@ class LibvirtVirtualMachineTask:
# Pass only the volumes list to the processor
self.processed_volume_paths = self.volume_processor.process_volumes(self.virtual_machine_config["volumes"])
self.logger.info(f"Volume processing complete. Paths: {self.processed_volume_paths}")
except SchedulingError:
raise # already classified — preserve error_type for HandleRetry
except Exception as e:
self.logger.error(f"Volume processing failed: {e}")
raise ValueError(f"Volume processing failed: {e}") # Re-raise to stop task execution
@@ -137,11 +140,13 @@ class LibvirtVirtualMachineTask:
try:
with libvirt.open("qemu:///system") as conn:
if conn is None:
raise RuntimeError("Failed to connect to libvirt.")
raise SchedulingError(ErrorType.LIBVIRT_CONNECTION_FAILED, "Failed to connect to libvirt")
caps_xml = conn.getCapabilities()
except SchedulingError:
raise
except libvirt.libvirtError as e:
self.logger.error(f"Libvirt connection error: {e}")
raise RuntimeError(f"Failed to connect to libvirt: {e}")
raise SchedulingError(ErrorType.LIBVIRT_CONNECTION_FAILED, str(e))
# Parse capabilities XML
try:
@@ -151,25 +156,25 @@ class LibvirtVirtualMachineTask:
arch_elem = root.find(".//guest/arch[@name='x86_64']")
if not arch_elem:
self.logger.error("x86_64 architecture not found in libvirt capabilities.")
raise ValueError("x86_64 architecture not supported by the host.")
raise SchedulingError(ErrorType.KVM_NOT_AVAILABLE, "x86_64 architecture not supported by this host")
# Check for KVM domain
has_kvm = any(domain.get("type") == "kvm" for domain in arch_elem.findall("domain"))
self.logger.debug(f"KVM domain found: {has_kvm}")
if not has_kvm:
self.logger.error("KVM domain not found within x86_64 architecture.")
raise ValueError("KVM domain not supported for x86_64.")
raise SchedulingError(ErrorType.KVM_NOT_AVAILABLE, "KVM domain not found in libvirt capabilities")
# Find all pc-q35 machine types
q35_machines = [
machine.text for machine in arch_elem.findall("machine")
if machine.text and machine.text.startswith("pc-q35-")
]
if not q35_machines:
self.logger.error("No 'pc-q35' machine types available for KVM.")
raise ValueError("No 'pc-q35' machine types available for KVM.")
raise SchedulingError(ErrorType.NO_MACHINE_TYPE, "No pc-q35 machine types available on this host")
# Use a more robust version comparison with packaging.version
from packaging import version
@@ -223,8 +228,9 @@ class LibvirtVirtualMachineTask:
self.logger.error(
f"VFIO attach timed out after {VFIO_OP_TIMEOUT}s for {addr}"
)
raise RuntimeError(
f"VFIO attach timed out for {addr}"
raise SchedulingError(
ErrorType.VFIO_ATTACH_TIMEOUT,
f"VFIO attach timed out after {VFIO_OP_TIMEOUT}s for {addr}",
)
except Exception as e:
self.logger.error(f"VFIO attach failed for {addr}: {e}")
@@ -308,7 +314,7 @@ class LibvirtVirtualMachineTask:
try:
conn = libvirt.open("qemu:///system") # Connect to the hypervisor
if conn is None:
raise RuntimeError("Failed to open connection to libvirt.")
raise SchedulingError(ErrorType.LIBVIRT_CONNECTION_FAILED, "Failed to open connection to libvirt")
# Lookup VirtualMachine by name
self.logger.info("Looking up VirtualMachine Name")
@@ -386,9 +392,15 @@ class LibvirtVirtualMachineTask:
gpu_pci_addresses = (self.virtual_machine_config or {}).get("gpu_pci_addresses", [])
if gpu_pci_addresses:
self._attach_gpus_vfio(gpu_pci_addresses)
domain = conn.defineXML(self.xml_config)
try:
domain = conn.defineXML(self.xml_config)
except libvirt.libvirtError as e:
raise SchedulingError(ErrorType.LIBVIRT_DEFINE_FAILED, str(e))
if self.desired_state == "running":
domain.create() # Start the VirtualMachine
try:
domain.create() # Start the VirtualMachine
except libvirt.libvirtError as e:
raise SchedulingError(ErrorType.LIBVIRT_START_FAILED, str(e))
response_payload["response"]["status"] = "VirtualMachine created and running"
else:
response_payload["response"]["status"] = "VirtualMachine created and stopped"
@@ -420,6 +432,8 @@ class LibvirtVirtualMachineTask:
raise ValueError(f"Unsupported action: {self.action}")
self.logger.info(f"Action '{self.action}' completed successfully for VirtualMachine '{self.virtual_machine_id}'.")
except SchedulingError:
raise # Let workerClient classify and send back with error_type
except Exception as e:
self.logger.error(f"Error executing LibvirtVirtualMachineTask: {e}")
response_payload["response"]["error"] = str(e)
+14 -8
View File
@@ -11,6 +11,7 @@ from typing import Dict, List, Optional, Union, Any
# from .volumes_ceph import CephVolumeProcessor
# Import settings to get Ceph configuration
from settings import settings
from worker_tasks.error_types import ErrorType, SchedulingError
class VolumeProcessor:
"""
@@ -602,8 +603,10 @@ class VolumeProcessor:
self.logger.info(f"Creating LVM volume {volume_name} (ID: {volume_id})")
# Read VG name from settings — raises KeyError if not configured in worker_settings.json
vg_name = settings.get_value("LVM_VG_NAME")
try:
vg_name = settings.get_value("LVM_VG_NAME")
except KeyError:
raise SchedulingError(ErrorType.LVM_VG_NOT_CONFIGURED, "LVM_VG_NAME not set in worker_settings.json")
lv_name = f"{volume_name}_{volume_id}"
# Canonical LVM block device path: /dev/<vg>/<lv>
@@ -617,7 +620,10 @@ class VolumeProcessor:
# Create logical volume
self.logger.info(f"Creating LVM logical volume {lv_name} in VG {vg_name}")
subprocess.run(['lvcreate', '-L', f"{size_gb}G", '-n', lv_name, vg_name], check=True)
try:
subprocess.run(['lvcreate', '-L', f"{size_gb}G", '-n', lv_name, vg_name], check=True)
except subprocess.CalledProcessError as e:
raise SchedulingError(ErrorType.LVM_VOLUME_CREATE_FAILED, f"lvcreate failed: {e}")
self.logger.debug("LVM volume created")
# Handle different source types
@@ -712,7 +718,7 @@ class VolumeProcessor:
elif volume_type == 'ceph':
# Ensure Ceph processor is initialized (lazy initialization)
if not self.ceph_processor and not self._initialize_ceph_processor():
raise ConnectionError("Ceph processor could not be initialized. Check Ceph settings and connectivity.")
raise SchedulingError(ErrorType.CEPH_INIT_FAILED, "Ceph processor could not be initialized — check Ceph settings and connectivity")
# Pass self (VolumeProcessor instance) to handle downloads if needed by Ceph processor
result = self.ceph_processor.create_volume(volume_config, self)
self.logger.debug(f"Result from ceph create_volume {result}")
@@ -721,12 +727,12 @@ class VolumeProcessor:
self.logger.info(f"Volume processing completed successfully. Result: {result}")
return result
except (subprocess.CalledProcessError, ValueError, IOError, ConnectionError, FileNotFoundError) as e: # Added ConnectionError
except SchedulingError:
raise # Already classified — let workerClient send error_type
except (subprocess.CalledProcessError, ValueError, IOError, ConnectionError, FileNotFoundError) as e:
self.logger.error(f"Error processing volume {volume_config.get('id', 'unknown')}: {str(e)}")
# The CephVolumeProcessor might raise rbd.Error or rados.Error, catch them if needed
# Or rely on it converting them to standard exceptions like ConnectionError/IOError
raise
except Exception as e: # Catch unexpected errors
except Exception as e:
self.logger.error(f"Error processing volume: {str(e)}")
raise