Files
3cloud-backend/app/controller/api/workload_host_routes.py
T
JamesBhattarai 7ec2e4d47b Feat: Implement NAT for Private network
Added enable_nat bool for networks to allow nats
Implemented docker network bridge, to allow nat with tenancy seperated as `enable_icc:false`
2026-08-08 17:41:15 +02:00

823 lines
34 KiB
Python

from datetime import datetime
import uuid
from flask import request
from app import app, db, logger
from app.models.models import (WorkloadHost, Region, Label,
Workload, WorkloadHostOVSBridge, WorkloadHostPooledResource, WorkloadHostFixedResource, WorkloadResourceUsage)
from app.models.network import NetworkPort
from app.controller import api_bp
from app.utils.standard_responses import api_response
from app.utils.nscontroller_config import build_nscontroller_config
import ipaddress
from sqlalchemy import func
# WorkloadHost Routes
@api_bp.route('/workload_hosts', methods=['POST'])
def add_workload_host():
data = request.json
instance = WorkloadHost(
region_id=(data['region_id']),
system_manufacturer=data.get('system_manufacturer'),
system_model=data.get('system_model'),
physical_identifier=data.get('physical_identifier'),
dcim_identifier=data.get('dcim_identifier'),
installed_date=data.get('installed_date'),
installed_status=data.get('installed_status'),
hostname=data.get('hostname')
)
db.session.add(instance)
db.session.commit()
return api_response(data=instance.to_json(), status=201)
@api_bp.route('/workload_hosts/<workload_host_id>', methods=['PUT'])
def edit_workload_host(workload_host_id):
workload_host = WorkloadHost.query.get_or_404((workload_host_id))
data = request.json
# Check if status is being updated to offline
status_changing_to_offline = False
if "status" in data and data["status"] == "offline" and workload_host.status != "offline":
status_changing_to_offline = True
# Update the host attributes
for key, value in data.items():
if key=="status":
logger.debug(f"Updating host status to '{value}' for host {workload_host_id}")
workload_host.set_status(value)
else:
setattr(workload_host, key, value)
db.session.commit()
# If status changed to offline, schedule the host downtime handler with a delay
if status_changing_to_offline:
logger.info(f"Host {workload_host_id} status changed to offline, scheduling downtime handler in {app.config.get('HOST_DOWNTIME_CONFIRM', 5)} seconds")
from app.tasks.host_monitoring import handle_host_downtime
handle_host_downtime.apply_async(
args=[str(workload_host_id)],
countdown=app.config.get("HOST_DOWNTIME_CONFIRM", 5)
)
return api_response(data={"success": True})
@api_bp.route('/workload_hosts/<workload_host_id>', methods=['GET'])
def get_workload_host(workload_host_id):
workload_host: WorkloadHost = WorkloadHost.query.filter_by(id=workload_host_id,deleted=0).first_or_404()
host_data = _enrich_host_data(workload_host)
return api_response(data=host_data)
@api_bp.route('/workload_hosts/<workload_host_id>', methods=['DELETE'])
def delete_workload_host(workload_host_id):
workload_host = WorkloadHost.query.get_or_404((workload_host_id))
workload_host.soft_delete()
db.session.commit()
return api_response(message='WorkloadHost deleted successfully', status=200)
@api_bp.route('/workload_hosts', methods=['GET'])
def get_workload_hosts():
# TODO - Filter only hosts this user can see.
# Regular users cant see this info, admins cna only see hosts in regions they have access to
# Check for region_id query parameter
region_id = request.args.get('region_id')
# If region_id is provided, validate it
if region_id:
# Validate UUID format for region_id
try:
region_uuid = (region_id)
except (ValueError, TypeError) as e:
error_message = "Invalid UUID format for region_id."
logger.error(f"{error_message} Error: {str(e)}")
return api_response(success=False, message=error_message, status=400)
# Check if the region exists
region = Region.query.get(region_uuid)
if not region:
return api_response(success=False, message="Region not found", status=404)
# Filter workload hosts by region_id
workload_hosts = WorkloadHost.query.filter_by(region_id=region_id, deleted=0).all()
else:
# No region filter, get all workload hosts
workload_hosts = WorkloadHost.query.filter_by(deleted=0).all()
# Prepare response with resources
hosts_data = [_enrich_host_data(host) for host in workload_hosts]
return api_response(data=hosts_data)
def _enrich_host_data(host: WorkloadHost):
"""Helper function to enrich a WorkloadHost with its resource data."""
# Get pooled resources from table
pooled_resources = WorkloadHostPooledResource.query.filter_by(workload_host_id=host.id).all()
pooled_dict = {pr.resource_type: pr for pr in pooled_resources}
# Get utilization to compute used
utilization = host.get_resource_utilization()
res_dict = utilization.get('resources', {})
pooled_resources_data = []
for resource_type, props in res_dict.items():
pr = pooled_dict.get(resource_type)
if not pr:
continue
quantity_in_use = props.get("quantity_in_use", 0)
total_quantity = props.get("total_quantity", 0)
pooled_resources_data.append({
"id": str(pr.id),
"name": pr.name,
"resource_type": resource_type,
"total_quantity": total_quantity,
"quantity_in_use": quantity_in_use,
"quantity_available": total_quantity-quantity_in_use,
})
# Get fixed resources
fixed_resources = WorkloadHostFixedResource.query.filter_by(workload_host_id=host.id, deleted=False).all()
fixed_resources_data = []
for resource in fixed_resources:
fixed_resources_data.append({
"id": str(resource.id),
"name": resource.name,
"resource_type": resource.type,
"model": resource.model,
"manufacturer": resource.manufacturer,
"physical_address": resource.physical_address,
})
# Combine host data with resources
host_data = host.to_json()
host_data["pooled_resources"] = pooled_resources_data
host_data["fixed_resources"] = fixed_resources_data
return host_data
@api_bp.route("/workload_hosts/<workload_host_id>/ovs_bridges", methods=['GET'])
def get_all_ovs_bridges(workload_host_id):
workload_host = WorkloadHost.query.filter_by(id=workload_host_id, deleted=False).first_or_404()
bridges = [b.ovs_bridge_name for b in workload_host.ovs_bridges if not b.deleted]
return api_response(data=bridges)
@api_bp.route("/workload_hosts/<workload_host_id>/ovs_bridges", methods=['POST'])
def update_ovs_bridges(workload_host_id):
data = request.json
if not data or 'bridges' not in data:
return api_response(success=False, message="Missing 'bridges' list in request body", status=400)
bridges_input = data['bridges']
if not isinstance(bridges_input, list):
return api_response(success=False, message="'bridges' must be a list of strings", status=400)
if not bridges_input:
return api_response(success=False, message="Bridge list cannot be empty", status=400)
# Validate and clean bridge names
bridges = []
for b in bridges_input:
if not isinstance(b, str):
return api_response(success=False, message="All bridges must be strings", status=400)
b_clean = b.strip()
if not b_clean or len(b_clean) > 255:
return api_response(success=False, message="Each bridge name must be a non-empty string with max 255 characters", status=400)
bridges.append(b_clean)
# Fetch the workload host
workload_host = WorkloadHost.query.filter_by(id=workload_host_id, deleted=False).first_or_404()
# Get current bridges (non-deleted)
current_bridges = db.session.query(WorkloadHostOVSBridge).filter_by(
workload_host_id=workload_host_id, deleted=False
).all()
current_bridge_names = {b.ovs_bridge_name for b in current_bridges}
# Identify bridges to delete (existing but not in new list)
to_delete_names = current_bridge_names - set(bridges)
for name in to_delete_names:
bridge_to_delete = db.session.query(WorkloadHostOVSBridge).filter_by(
workload_host_id=workload_host_id, ovs_bridge_name=name, deleted=False
).first()
if bridge_to_delete:
bridge_to_delete.soft_delete()
# Identify bridges to add (new but not existing)
to_add_names = set(bridges) - current_bridge_names
for name in to_add_names:
new_bridge = WorkloadHostOVSBridge(
workload_host_id=workload_host_id,
ovs_bridge_name=name
)
db.session.add(new_bridge)
# Update timestamps for remaining existing bridges
remaining_names = current_bridge_names & set(bridges)
for name in remaining_names:
bridge_to_update = db.session.query(WorkloadHostOVSBridge).filter_by(
workload_host_id=workload_host_id, ovs_bridge_name=name, deleted=False
).first()
if bridge_to_update:
bridge_to_update.updated_at = datetime.utcnow()
try:
db.session.commit()
logger.info(f"OVS bridges updated for workload host {workload_host_id}: {bridges}")
return api_response(
data={
"workload_host_id": workload_host_id,
"bridges": bridges,
"message": "OVS bridges updated successfully"
},
status=200
)
except Exception as e:
db.session.rollback()
logger.error(f"Database error updating OVS bridges for {workload_host_id}: {str(e)}")
return api_response(success=False, message="Internal server error", status=500)
@api_bp.route("/workload_hosts/<workload_host_id>/pci_devices", methods=['POST'])
def register_pci_devices(workload_host_id):
"""
Called by the worker on join to report its discovered PCI devices (GPUs, etc.).
Upserts WorkloadHostFixedResource rows and updates gpu-capable / gpu-count labels
so the scheduler can select this host for GPU workloads.
"""
workload_host = WorkloadHost.query.filter_by(id=workload_host_id, deleted=False).first_or_404()
data = request.json
if not data or 'devices' not in data:
return api_response(success=False, message="Missing 'devices' list in request body", status=400)
devices = data['devices']
if not isinstance(devices, list):
return api_response(success=False, message="'devices' must be a list", status=400)
try:
# --- Upsert WorkloadHostFixedResource rows ---
existing_rows = WorkloadHostFixedResource.query.filter_by(
workload_host_id=workload_host_id, deleted=False
).all()
existing_by_addr = {r.physical_address: r for r in existing_rows}
incoming_addrs = {d.get("pci_address") for d in devices}
for addr, row in existing_by_addr.items():
if addr not in incoming_addrs:
row.soft_delete()
gpu_count = 0
for device in devices:
device_type = device.get("matched_filter_type", "accelerator")
pci_address = device.get("pci_address")
if pci_address in existing_by_addr:
row = existing_by_addr[pci_address]
row.type = device_type
row.model = device.get("device_name")
row.manufacturer = device.get("vendor_name")
row.name = f"{device.get('vendor_name', 'Unknown')} {device.get('device_name', 'PCI device')}"
row.status = "active"
else:
resource = WorkloadHostFixedResource(
workload_host_id=workload_host_id,
type=device_type,
model=device.get("device_name"),
manufacturer=device.get("vendor_name"),
physical_address=pci_address,
name=f"{device.get('vendor_name', 'Unknown')} {device.get('device_name', 'PCI device')}",
status="active",
)
db.session.add(resource)
if device_type == "gpu":
gpu_count += 1
db.session.flush()
# --- Update gpu-capable / gpu-count labels ---
# Remove any existing gpu labels first
Label.query.filter_by(
target_object_type="workload_hosts",
target_object_id=workload_host_id,
).filter(Label.label_key.in_(["gpu-capable", "gpu-count"])).update(
{"deleted": True}, synchronize_session=False
)
if gpu_count > 0:
workload_host.addLabel("gpu-capable", "true")
workload_host.addLabel("gpu-count", str(gpu_count))
logger.info(f"Host {workload_host.name} ({workload_host_id}): registered {gpu_count} GPU(s)")
else:
logger.info(f"Host {workload_host.name} ({workload_host_id}): no GPUs in reported devices")
db.session.commit()
return api_response(
data={
"workload_host_id": workload_host_id,
"devices_registered": len(devices),
"gpu_count": gpu_count,
},
status=201
)
except Exception as e:
db.session.rollback()
logger.error(f"Error registering PCI devices for host {workload_host_id}: {e}")
return api_response(success=False, message="Internal server error", status=500)
@api_bp.route("/workload_hosts/<workload_host_id>/active_workloads", methods=['GET'])
def get_all_active_workloads(workload_host_id):
# Get a list of all active workloads on the given host, will eventually be used
# to allow the host to track and manage it's own state
workload_list=Workload.query.filter_by(workload_host_id=(workload_host_id),deleted=0).all()
enriched_workloads = []
for item in workload_list:
data = item.to_json()
data["resource_usage"] = item.get_resource_usage()
enriched_workloads.append(data)
return api_response(data=enriched_workloads)
@api_bp.route('/workload_hosts/enroll', methods=['POST'])
def enroll_workload_host():
"""Handles workload host enrollment, creating a WorkloadHost"""
data = request.json
logger.debug(f"Enrollment data {data}")
# Validate required fields
required_fields = ['region_id', 'physical_identifier', 'region_enrollment_key']
for field in required_fields:
if field not in data:
logger.error("Enrollment failed")
return api_response(success=False, message=f'Missing required field: {field}', status=400)
try:
# Validate region_id
region = Region.query.get((data['region_id']))
if not region:
logger.error("Invalid region_id")
return api_response(success=False, message='Invalid region ID', status=400)
if not str(region.enrollment_key)==data['region_enrollment_key']:
logger.error(f"Supplied enrollment key {data['region_enrollment_key']} vs key in DB {region.enrollment_key}")
return api_response(success=False, message='Invalid region enrollment key', status=400)
# Process IP addresses
ip_address_eastwest = None
ip_address_northsouth = None
# Parse the all_ip_addresses field from data
all_ip_addresses = data.get('all_ip_addresses', [])
# Process IP range checking only if the region has defined ranges
if region.ip_address_range_eastwest or region.ip_address_range_northsouth:
# Check each IP address against the region's IP ranges
if all_ip_addresses and isinstance(all_ip_addresses, list):
for ip_entry in all_ip_addresses:
if not isinstance(ip_entry, dict) or 'ip' not in ip_entry:
continue
ip_addr = ip_entry.get('ip')
# Skip non-IPv4 addresses (simple check)
if not isinstance(ip_addr, str) or ':' in ip_addr:
continue
# Check for eastwest network match
if region.ip_address_range_eastwest and is_ip_in_cidr(ip_addr, region.ip_address_range_eastwest):
ip_address_eastwest = ip_addr
# Check for northsouth network match
if region.ip_address_range_northsouth and is_ip_in_cidr(ip_addr, region.ip_address_range_northsouth):
ip_address_northsouth = ip_addr
# Create WorkloadHost entry
workload_host = WorkloadHost(
name=data.get('hostname'),
region_id=(data['region_id']),
system_manufacturer=data.get('system_manufacturer'),
system_model=data.get('system_model'),
physical_identifier=data['physical_identifier'],
dcim_identifier=data.get('dcim_identifier'),
installed_status=data.get('installed_status', 'active'),
hostname=data.get('hostname'),
ip_address_all=str(data.get('all_ip_addresses')),
ip_address_eastwest=ip_address_eastwest,
ip_address_northsouth=ip_address_northsouth
)
db.session.add(workload_host)
db.session.flush() # Flush to get workload_host.id without committing yet
# Create CPU resource entry
cpu_count = data.get('cpu_count', 1) # Default to 1 if not provided
cpu_resource = WorkloadHostPooledResource(
name=f"CPU resource for {workload_host.hostname}",
workload_host_id=workload_host.id,
resource_type="cpu",
total_quantity=float(cpu_count),
quantity_in_use=0.0,
quantity_available=float(cpu_count)
)
db.session.add(cpu_resource)
# Create memory resource entry
memory_mb = data.get('memory_mb', 1024) # Default to 1GB if not provided
memory_resource = WorkloadHostPooledResource(
name=f"Memory resource for {workload_host.hostname}",
workload_host_id=workload_host.id,
resource_type="ram",
total_quantity=float(memory_mb),
quantity_in_use=0.0,
quantity_available=float(memory_mb)
)
db.session.add(memory_resource)
# Now commit all changes
db.session.commit()
response_object={
"worker_id": workload_host.id,
"worker_secret": workload_host.secret_key
}
logger.info(f"WorkloadHost enrolled successfully: {workload_host.id}")
return api_response(data=response_object, status=201)
except Exception as e:
db.session.rollback()
logger.error(f"Enrollment failed: {str(e)}")
return api_response(success=False, message='Enrollment failed', error_details={'details': str(e)}, status=500)
@api_bp.route('/workload_hosts/<host_id>/labels', methods=['GET'])
def get_workload_host_labels(host_id):
# Validate host_id as UUID
try:
host_uuid = (host_id)
except (ValueError, TypeError) as e:
error_message = "Invalid UUID format for host_id."
logger.error(f"{error_message} Error: {str(e)}")
return api_response(success=False, message=error_message, status=400)
# Fetch the labels
try:
labels = Label.query.filter_by(
target_object_id=host_uuid,
target_object_type="workload_hosts",
deleted=False,
).all()
return api_response(data=[label.to_json() for label in labels], status=200)
except Exception as e:
error_message = f"Failed to fetch labels for host {host_id}."
logger.error(f"{error_message} Error: {str(e)}")
return api_response(success=False, message=error_message, status=500)
@api_bp.route('/workload_hosts/<host_id>/labels', methods=['POST', 'DELETE'])
def manage_workload_host_labels(host_id):
# Validate host_id as UUID
try:
host_uuid = (host_id)
except (ValueError, TypeError) as e:
error_message = "Invalid UUID format for host_id."
logger.error(f"{error_message} Error: {str(e)}")
return api_response(success=False, message=error_message, status=400)
if request.method == 'POST':
# Add or update label
data = request.json
if not data:
return api_response(success=False, message="No data provided", status=400)
required_fields = ['label_key', 'label_value']
if not all(field in data for field in required_fields):
return api_response(success=False, message=f"Missing required fields. Required: {required_fields}", status=400)
try:
# Check if label already exists
label = Label.query.filter_by(
target_object_id=host_uuid,
target_object_type="workload_hosts",
label_key=data['label_key']
).first()
if label:
# Update existing label
label.label_value = data['label_value']
label.admin_only_view = data.get('admin_only_view', False)
label.admin_only_set = data.get('admin_only_set', False)
message = "Label updated"
else:
# Create new label
label = Label(
target_object_id=host_uuid,
target_object_type="workload_hosts",
label_key=data['label_key'],
label_value=data['label_value'],
admin_only_view=data.get('admin_only_view', False),
admin_only_set=data.get('admin_only_set', False),
name=f"Label {data['label_key']}={data['label_value']}",
description=f"Label for host {host_id}"
)
db.session.add(label)
message = "Label added"
db.session.commit()
return api_response(data={"label": label.to_json()}, message=message, status=200)
except Exception as e:
db.session.rollback()
error_message = f"Failed to {'update' if label else 'add'} label"
logger.error(f"{error_message} Error: {str(e)}")
return api_response(success=False, message=error_message, status=500)
elif request.method == 'DELETE':
# Remove label
data = request.json
if not data or 'label_key' not in data:
return api_response(success=False, message="label_key must be specified in request body", status=400)
try:
label = Label.query.filter_by(
target_object_id=host_uuid,
target_object_type="workload_hosts",
label_key=data['label_key']
).first()
if not label:
return api_response(success=False, message=f"Label {data['label_key']} not found", status=404)
db.session.delete(label)
db.session.commit()
return api_response(data={"deleted_label": label.to_json()}, message="Label deleted", status=200)
except Exception as e:
db.session.rollback()
error_message = f"Failed to delete label {data['label_key']}"
logger.error(f"{error_message} Error: {str(e)}")
return api_response(success=False, message=error_message, status=500)
@api_bp.route('/workload_hosts/<workload_host_id>/utilization', methods=['GET'])
def get_workload_host_utilization(workload_host_id):
"""
Get resource utilization for a specific workload host.
Fetches the WorkloadHost and computes utilization using the helper method.
"""
# Validate host_id as UUID
try:
host_uuid = uuid.UUID(workload_host_id)
except (ValueError, TypeError) as e:
error_message = "Invalid UUID format for workload_host_id."
logger.error(f"{error_message} Error: {str(e)}")
return api_response(success=False, message=error_message, status=400)
# Fetch the workload host
workload_host = WorkloadHost.query.filter_by(
id=workload_host_id,
deleted=False
).first_or_404()
try:
# Compute utilization using the helper method
utilization_data = workload_host.get_resource_utilization()
return api_response(data=utilization_data, status=200)
except Exception as e:
error_message = f"Failed to compute resource utilization for host {workload_host_id}."
logger.error(f"{error_message} Error: {str(e)}")
return api_response(success=False, message=error_message, status=500)
def is_ip_in_cidr(ip_address, cidr_range):
"""
Check if an IP address is within a CIDR range
Args:
ip_address (str): IP address to check
cidr_range (str): CIDR range (e.g. '192.168.0.0/24')
Returns:
bool: True if IP is in the CIDR range, False otherwise
"""
try:
network = ipaddress.ip_network(cidr_range, strict=False)
ip = ipaddress.ip_address(ip_address)
logger.debug(f"Checking IP {ip} for network {network}")
return ip in network
except (ValueError, TypeError):
logger.error(f"Invalid IP address or CIDR range: {ip_address}, {cidr_range}")
return False
# WorkloadHostAvailablePort Routes
@api_bp.route('/workload_hosts/<workload_host_id>/available_ports', methods=['GET'])
def get_workload_host_available_ports(workload_host_id):
"""Get all available ports for a workload host"""
from app.models.models import WorkloadHostAvailablePort
available_ports = WorkloadHostAvailablePort.query.filter_by(
workload_host_id=workload_host_id, deleted=False
).all()
return api_response(data=[port.to_json() for port in available_ports], status=200)
@api_bp.route('/workload_hosts/<workload_host_id>/available_ports', methods=['POST'])
def add_workload_host_available_port(workload_host_id):
"""Add available ports for a workload host"""
from app.models.models import WorkloadHostAvailablePort
data = request.json
# Validate required fields
required_fields = ['start_port', 'end_port']
for field in required_fields:
if field not in data:
return api_response(success=False, message=f'Missing required field: {field}', status=400)
# Validate port range
start_port = data['start_port']
end_port = data['end_port']
protocol = data.get('protocol', 'tcp')
if not isinstance(start_port, int) or not isinstance(end_port, int):
return api_response(success=False, message='start_port and end_port must be integers', status=400)
if start_port < 1 or end_port > 65535 or start_port > end_port:
return api_response(success=False, message='Invalid port range. Ports must be between 1 and 65535', status=400)
if protocol not in ['tcp', 'udp']:
return api_response(success=False, message='Invalid protocol. Must be "tcp" or "udp"', status=400)
# Create available port entry
available_port = WorkloadHostAvailablePort(
workload_host_id=workload_host_id,
start_port=start_port,
end_port=end_port,
protocol=protocol,
name=f"Available ports {start_port}-{end_port} ({protocol.upper()})"
)
db.session.add(available_port)
db.session.commit()
return api_response(data=available_port.to_json(), status=201)
@api_bp.route('/workload_hosts/<workload_host_id>/available_ports/<available_port_id>', methods=['DELETE'])
def remove_workload_host_available_port(workload_host_id, available_port_id):
"""Remove available ports for a workload host"""
from app.models.models import WorkloadHostAvailablePort
available_port = WorkloadHostAvailablePort.query.filter_by(
id=available_port_id, workload_host_id=workload_host_id, deleted=False
).first_or_404()
available_port.soft_delete()
db.session.commit()
return api_response(message='Available port range deleted successfully', status=200)
@api_bp.route("/workload_hosts/<workload_host_id>/container_workloads", methods=['GET'])
def get_container_workloads_for_host(workload_host_id):
"""
Get all container workloads for a specific host in pod-update format.
By default, only non-deleted containers are returned. To include deleted containers,
pass the query parameter 'include_deleted=true'.
Query Parameters:
include_deleted (str): Set to 'true' to include deleted containers in the response.
Defaults to 'false'.
This endpoint returns all containers assigned to this host in the same format
used by the pod-update task, allowing the worker to reconcile its state.
"""
try:
# Validate host_id as UUID
host_uuid = (workload_host_id)
except (ValueError, TypeError) as e:
error_message = "Invalid UUID format for host_id."
logger.error(f"{error_message} Error: {str(e)}")
return api_response(success=False, message=error_message, status=400)
# Verify host exists
host = WorkloadHost.query.filter_by(id=host_uuid, deleted=False).first()
if not host:
return api_response(success=False, message="Host not found", status=404)
# Get all container workloads for this host
# We need to get pods assigned to this host and then get all containers in those pods
from app.models.models import ContainerPod
from app.utils.create_workload_container import build_pod_payload
# Check for include_deleted query parameter
include_deleted = request.args.get('include_deleted', 'false').lower() == 'true'
# Get all pods assigned to this host
pods = ContainerPod.query.filter_by(workload_host_id=host_uuid, deleted=False).all()
# Build a list of all containers for this host in pod-update format
all_containers = []
for pod in pods:
try:
# Use the existing build_pod_payload function to get the container data
pod_payload = build_pod_payload(pod, use_db_state=True, include_deleted=include_deleted)
# Extract containers from the pod payload and add pod_id to each
containers = pod_payload.get("job_details", {}).get("containers", [])
for container in containers:
container["pod_id"] = str(pod.id)
all_containers.extend(containers)
except Exception as e:
logger.error(f"Error building pod payload for pod {pod.id}: {str(e)}")
# Continue with other pods even if one fails
# Include standalone NSControllers (not attached to any pod).
# Without this, reconcile_and_delete receives an incomplete expected set and
# may delete healthy NSController containers as "unexpected".
standalone_nscontrollers = Workload.query.filter_by(
workload_host_id=host_uuid,
workload_type="NSController",
deleted=False,
).all()
for nsc in standalone_nscontrollers:
# Skip pod-backed NSControllers; those are already covered by build_pod_payload.
if nsc.pod_id:
continue
try:
# Determine network context from the NSController's port (if present).
ports = NetworkPort.query.filter_by(workload_id=nsc.id).order_by(
NetworkPort.deleted.asc(),
NetworkPort.updated_at.desc(),
NetworkPort.created_at.desc(),
).all()
port = next((p for p in ports if not p.deleted), None)
context_port = port or (ports[0] if ports else None)
network_id = context_port.network_id if context_port else None
nsc_config = build_nscontroller_config(
vdc_id=nsc.vdc_id,
container_name=nsc.name,
pod_id=None,
include_ports=False,
network_id=network_id,
)
nsc_container = {
"container_id": str(nsc.id),
"docker_image": nsc_config.get("docker_image", "xcloudify-nscontroller:latest"),
"container_name": nsc.name,
"workload_type": "NSController",
"desired_state": "running",
"use_dns": True,
"vdc_id": nsc.vdc_id,
"dns_config": nsc_config.get("dns_config", {}),
"env": nsc_config.get("env", {}),
"cpu": nsc_config.get("cpu", 4),
"mem_limit": nsc_config.get("mem_limit", 128),
"pod_id": f"standalone-nsc-{nsc.id}",
"network_ports": [], # Initialize as empty; will populate if port exists
"enable_host_nat": nsc_config.get("enable_host_nat", False),
}
if port:
bridge = port.network.ovs_bridge if port.network else None
if not bridge and host.ovs_bridges:
bridge = host.ovs_bridges[0].ovs_bridge_name
nsc_container["network_ports"] = [{
"network_id": str(port.network_id),
"name": port.name,
"ip_address": port.ip_address,
"mac_address": port.mac_address,
"subnet_mask": port.subnet_mask,
"dns_servers": port.dns_servers.split(",") if port.dns_servers else [],
"vni": port.network.vni if port.network else None,
"ovs_bridge": bridge,
"gateway": port.network.ipv4_gateway if port.network else None,
}]
all_containers.append(nsc_container)
except Exception as e:
logger.error(f"Error building standalone NSController payload for workload {nsc.id}: {str(e)}")
# Continue with other workloads even if one fails
# Return the containers in a format compatible with pod-update
response_payload = {
"worker_id": str(host_uuid),
"task_type": "pod-update",
"job_details": {
"pod_id": None, # No specific pod since this is all containers for the host
"containers": all_containers,
},
}
return api_response(data=response_payload)