From 4c0082d92ce4256beaa58052b71127738df3b30d Mon Sep 17 00:00:00 2001 From: Ramgopal Nagaboina Date: Tue, 6 Oct 2026 16:44:48 -0400 Subject: [PATCH 1/7] drs: cluster power management with predictive demand and active evacuation Extend the cluster DRS poll to power manage hosts. When a cluster's load falls below a low threshold, DRS releases an empty, out-of-band capable host: it disables the host first so no new workload is scheduled onto it, then powers it off through the out-of-band management driver, rolling the resource state back if the power operation fails. When load rises above a high threshold, a host that DRS previously powered off is woken through the same driver and re-enabled. Add an opt-in predictive mode that forecasts short-term cluster utilisation from a rolling history with a least-squares trend, so a host can be woken ahead of rising demand rather than after it. Add an opt-in active evacuation mode that migrates running VMs off a releasable host, capped by the existing per-run migration limit and aware of reserved capacity so it never pushes a destination host past its headroom. All of this is off by default and gated by ConfigKeys. Unit tests cover the release and wakeup classification, the disable-then-power-off path with rollback, the utilisation forecast and the capped evacuation planner. --- .../cloudstack/cluster/ClusterDrsService.java | 40 ++ .../cluster/ClusterDrsServiceImpl.java | 394 +++++++++++++++++- .../ClusterDrsPowerManagementTest.java | 150 +++++++ .../ClusterDrsPowerOrchestrationTest.java | 128 ++++++ 4 files changed, 711 insertions(+), 1 deletion(-) create mode 100644 server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerManagementTest.java create mode 100644 server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java diff --git a/api/src/main/java/org/apache/cloudstack/cluster/ClusterDrsService.java b/api/src/main/java/org/apache/cloudstack/cluster/ClusterDrsService.java index ba6a6464fc20..12ab333ab6ab 100644 --- a/api/src/main/java/org/apache/cloudstack/cluster/ClusterDrsService.java +++ b/api/src/main/java/org/apache/cloudstack/cluster/ClusterDrsService.java @@ -88,6 +88,46 @@ public interface ClusterDrsService extends Manager, Configurable, Scheduler { true, ConfigKey.Scope.Cluster, null, "DRS imbalance skip threshold for Condensed algorithm", null, null, null); + ConfigKey ClusterDrsPowerManagementEnabled = new ConfigKey<>(Boolean.class, "drs.power.management.enable", + ConfigKey.CATEGORY_ADVANCED, "false", + "Enable distributed power management on the cluster. When the cluster is under-utilized, a host that has " + + "been emptied by DRS is powered off through its out-of-band management; when the cluster is over-utilized, " + + "a powered-off host is powered back on. Requires automatic DRS and out-of-band management to be configured.", + true, ConfigKey.Scope.Cluster, null, "Enable DRS power management", null, null, null); + + ConfigKey ClusterDrsPowerManagementLowThreshold = new ConfigKey<>(Float.class, "drs.power.management.low.threshold", + ConfigKey.CATEGORY_ADVANCED, "0.3", + "Cluster utilization (0.0 to 1.0, on the configured DRS metric) below which an idle host becomes a candidate " + + "to power off, provided the remaining hosts can carry the load without exceeding the high threshold.", + true, ConfigKey.Scope.Cluster, null, "DRS power management low threshold", null, null, null); + + ConfigKey ClusterDrsPowerManagementHighThreshold = new ConfigKey<>(Float.class, "drs.power.management.high.threshold", + ConfigKey.CATEGORY_ADVANCED, "0.75", + "Cluster utilization (0.0 to 1.0, on the configured DRS metric) above which a powered-off host is powered " + + "back on. Also the ceiling the remaining hosts must stay under before a host is powered off.", + true, ConfigKey.Scope.Cluster, null, "DRS power management high threshold", null, null, null); + + ConfigKey ClusterDrsPowerManagementEvacuate = new ConfigKey<>(Boolean.class, "drs.power.management.evacuate.enable", + ConfigKey.CATEGORY_ADVANCED, "false", + "When DRS power management decides a host can be released but the host still runs VMs, migrate those VMs " + + "onto the remaining hosts (only when every VM can be placed) so the host can then be powered off. " + + "Off by default: power management otherwise waits for DRS to empty a host on its own.", + true, ConfigKey.Scope.Cluster, null, "Actively evacuate before power-off", null, null, null); + + ConfigKey ClusterDrsPredictiveEnabled = new ConfigKey<>(Boolean.class, "drs.predictive.enable", + ConfigKey.CATEGORY_ADVANCED, "false", + "Use the recent trend of cluster utilization, not only the instantaneous value, when deciding power " + + "management actions. A rising trend powers a host back on earlier and holds off powering one off; a falling " + + "trend is ignored for power-off so a brief dip does not churn hosts. Requires DRS power management. The trend " + + "window is kept in memory by the management server that runs the poll, so after a restart or in a multi-management-server " + + "deployment it rebuilds from the samples seen since that server last took the poll.", + true, ConfigKey.Scope.Cluster, null, "Enable predictive DRS", null, null, null); + + ConfigKey ClusterDrsPredictiveWindow = new ConfigKey<>(Integer.class, "drs.predictive.window", + ConfigKey.CATEGORY_ADVANCED, "5", + "Number of recent DRS poll samples of cluster utilization used to compute the trend for predictive DRS.", + true, ConfigKey.Scope.Cluster, null, "Predictive DRS window", null, null, null); + /** * Generate a DRS plan for a cluster and save it as per the parameters diff --git a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java index 62075aae596e..adc744667de0 100644 --- a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java +++ b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java @@ -32,9 +32,13 @@ import com.cloud.event.EventVO; import com.cloud.event.dao.EventDao; import com.cloud.exception.InvalidParameterValueException; +import com.cloud.host.DetailVO; import com.cloud.host.Host; import com.cloud.host.HostVO; +import com.cloud.host.Status; import com.cloud.host.dao.HostDao; +import com.cloud.host.dao.HostDetailsDao; +import com.cloud.resource.ResourceState; import com.cloud.offering.ServiceOffering; import com.cloud.org.Cluster; import com.cloud.server.ManagementServer; @@ -54,6 +58,8 @@ import com.cloud.vm.VMInstanceDetailVO; import com.cloud.vm.VMInstanceVO; import com.cloud.vm.VirtualMachine; +import org.apache.cloudstack.outofbandmanagement.OutOfBandManagement; +import org.apache.cloudstack.outofbandmanagement.OutOfBandManagementService; import com.cloud.vm.VirtualMachineProfile; import com.cloud.vm.VirtualMachineProfileImpl; import com.cloud.vm.VmDetailConstants; @@ -89,8 +95,11 @@ import java.util.Collections; import java.util.Date; import java.util.HashMap; +import java.util.LinkedList; +import java.util.concurrent.ConcurrentHashMap; import java.util.HashSet; import java.util.List; +import java.util.TreeMap; import java.util.Map; import java.util.Set; import java.util.Timer; @@ -117,9 +126,24 @@ public class ClusterDrsServiceImpl extends ManagerBase implements ClusterDrsServ @Inject HostDao hostDao; + @Inject + HostDetailsDao hostDetailsDao; + + @Inject + OutOfBandManagementService outOfBandManagementService; + @Inject EventDao eventDao; + // Host detail marking a host that DRS power management powered off, so only those are powered back on + // (never a host that is down for another reason). + protected static final String DRS_POWER_STATE_DETAIL = "drs.power.state"; + protected static final String DRS_POWER_STATE_OFF = "off"; + + // Recent cluster utilization samples (ratio on the DRS metric) per cluster, newest last, used by predictive DRS. + // Written only from the single-threaded poll under the clusterDRS.poll lock. + protected final Map> clusterUtilizationHistory = new ConcurrentHashMap<>(); + @Inject HostJoinDao hostJoinDao; @@ -199,6 +223,7 @@ public void poll(Date timestamp) { processPlans(); generateDrsPlanForAllClusters(); processPlans(); + managePowerForAllClusters(); } finally { lock.unlock(); } @@ -851,11 +876,378 @@ public String getConfigComponentName() { return ClusterDrsService.class.getSimpleName(); } + protected void managePowerForAllClusters() { + for (ClusterVO cluster : clusterDao.listAll()) { + try { + managePowerForCluster(cluster); + } catch (Exception e) { + logger.warn("DRS power management skipped cluster [{}] due to [{}].", cluster.getId(), e.getMessage(), e); + } + } + } + + protected void managePowerForCluster(ClusterVO cluster) { + if (cluster.getAllocationState() == Disabled + || Boolean.FALSE.equals(ClusterDrsEnabled.valueIn(cluster.getId())) + || Boolean.FALSE.equals(ClusterDrsPowerManagementEnabled.valueIn(cluster.getId()))) { + return; + } + final float lowThreshold = ClusterDrsPowerManagementLowThreshold.valueIn(cluster.getId()); + final float highThreshold = ClusterDrsPowerManagementHighThreshold.valueIn(cluster.getId()); + final boolean useCpu = "cpu".equals(getClusterDrsMetric(cluster.getId())); + + List routingHosts = hostDao.findByClusterId(cluster.getId(), Host.Type.Routing); + List upHosts = new ArrayList<>(); + List poweredOffByDrs = new ArrayList<>(); + for (HostVO host : routingHosts) { + if (host.getStatus() == Status.Up && host.getResourceState() == ResourceState.Enabled) { + // A host that is Up+Enabled is in service; clear any stale DRS power marker left by a power-off + // that failed or by a host that came back by another path, so it is not excluded forever. + clearStalePowerMarker(host); + upHosts.add(host); + } else if (host.getStatus() != Status.Up && isPoweredOffByDrs(host)) { + // Only a host that is actually down (disconnected) and was powered off by DRS is a wake candidate; + // a host still Up while shutting down is in neither list. + poweredOffByDrs.add(host); + } + } + if (upHosts.isEmpty()) { + return; + } + + Map> capacityMap = getHostCapacityMap(upHosts, useCpu); + double clusterUsed = 0d; + double clusterTotal = 0d; + for (Ternary capacity : capacityMap.values()) { + // count used + reserved as the committed load, so a host holding HA/allocation reservations is not + // powered off just because its live usage is low. + clusterUsed += capacity.first() + capacity.second(); + clusterTotal += capacity.third(); + } + + // predictive DRS: fold the recent trend in. Use the more conservative of the instantaneous and forecast + // utilization, so a rising trend wakes a host sooner and holds off power-off, while a transient dip never + // triggers an aggressive power-off. + double effectiveUsed = clusterUsed; + double instantaneousRatio = clusterTotal > 0d ? clusterUsed / clusterTotal : 0d; + if (Boolean.TRUE.equals(ClusterDrsPredictiveEnabled.valueIn(cluster.getId()))) { + int window = ClusterDrsPredictiveWindow.valueIn(cluster.getId()); + double forecastRatio = forecastUtilization(recordAndGetUtilizationHistory(cluster.getId(), instantaneousRatio, window)); + effectiveUsed = Math.max(instantaneousRatio, forecastRatio) * clusterTotal; + } + + // prefer restoring capacity over saving power: if the cluster is hot and we have a host we powered + // off earlier that we can still reach over out-of-band management, bring it back before any power-off. + if (clusterNeedsWakeup(effectiveUsed, clusterTotal, highThreshold, poweredOffByDrs.size())) { + for (HostVO poweredOff : poweredOffByDrs) { + if (isPowerManageable(poweredOff)) { + powerOnHost(poweredOff, cluster); + return; + } + } + } + + // power off at most one empty, out-of-band-manageable host per poll, and only if the rest can carry the load. + for (HostVO host : upHosts) { + Ternary capacity = capacityMap.get(host.getId()); + if (capacity == null) { + continue; + } + if (hostHasNoRunningVms(host) && isPowerManageable(host) && !isPoweredOffByDrs(host) + && clusterCanReleaseHost(effectiveUsed, clusterTotal, capacity.third(), upHosts.size(), lowThreshold, highThreshold)) { + powerOffHost(host, cluster); + return; + } + } + + // nothing empty to power off; if the operator opted in, drain a releasable host so a later poll can + // power it off once empty. + if (Boolean.TRUE.equals(ClusterDrsPowerManagementEvacuate.valueIn(cluster.getId()))) { + evacuateReleasableHost(cluster, upHosts, capacityMap, effectiveUsed, clusterTotal, lowThreshold, highThreshold, useCpu); + } + } + + /** + * Picks the least-loaded host the cluster can give up, and if every VM on it can be placed on the remaining + * hosts, migrates them away (a later poll powers the now-empty host off). Does nothing if no host qualifies or + * if the VMs cannot all be placed, so a host is never left partially drained. + */ + protected void evacuateReleasableHost(ClusterVO cluster, List upHosts, Map> capacityMap, + double effectiveUsed, double clusterTotal, float lowThreshold, float highThreshold, boolean useCpu) { + HostVO candidate = null; + double candidateUsed = Double.MAX_VALUE; + for (HostVO host : upHosts) { + Ternary cap = capacityMap.get(host.getId()); + if (cap == null || hostHasNoRunningVms(host) || !isPowerManageable(host) || !hostEvacuatable(host) + || !clusterCanReleaseHost(effectiveUsed, clusterTotal, cap.third(), upHosts.size(), lowThreshold, highThreshold)) { + continue; + } + if (cap.first() < candidateUsed) { + candidate = host; + candidateUsed = cap.first(); + } + } + if (candidate == null) { + return; + } + + Map vmNeeds = new HashMap<>(); + for (VMInstanceVO vm : vmInstanceDao.listByHostId(candidate.getId())) { + vmNeeds.put(vm.getId(), vmResourceNeed(vm, useCpu)); + } + Map hostFree = new HashMap<>(); + for (HostVO host : upHosts) { + if (host.getId() == candidate.getId()) { + continue; + } + Ternary cap = capacityMap.get(host.getId()); + if (cap != null) { + // Free = total - (used + reserved), consistent with the release decision, so evacuated VMs are not + // placed onto capacity another host is holding for HA/allocation reservations. + hostFree.put(host.getId(), (double) cap.third() - (cap.first() + cap.second())); + } + } + + Map plan = planHostEvacuation(vmNeeds, hostFree); + if (plan.isEmpty()) { + logger.debug("DRS power management: host [{}] in cluster [{}] cannot be evacuated; not all VMs fit on the remaining hosts.", + candidate.getId(), cluster.getId()); + return; + } + // respect the cluster's DRS migration budget: drain at most ClusterDrsMaxMigrations VMs per poll, the host + // empties over subsequent polls and is powered off once truly empty. + int maxMigrations = ClusterDrsMaxMigrations.valueIn(cluster.getId()); + logger.info("DRS power management: cluster [{}] is under-utilized; draining host [{}] ({} VMs, up to {} per poll) to power it off.", + cluster.getId(), candidate.getId(), plan.size(), maxMigrations); + long eventId = ActionEventUtils.onStartedActionEvent(User.UID_SYSTEM, Account.ACCOUNT_ID_SYSTEM, + EventTypes.EVENT_VM_MIGRATE, + String.format("DRS power management draining host %d in cluster %s", candidate.getId(), cluster.getUuid()), + candidate.getId(), ApiCommandResourceType.Host.toString(), true, 0); + int submitted = 0; + for (Map.Entry entry : plan.entrySet()) { + if (submitted >= maxMigrations) { + break; + } + VirtualMachine vm = vmInstanceDao.findById(entry.getKey()); + HostVO destination = hostDao.findById(entry.getValue()); + if (vm != null && destination != null) { + createMigrateVMAsyncJob(vm, destination, eventId); + submitted++; + } + } + } + + protected double vmResourceNeed(VMInstanceVO vm, boolean useCpu) { + ServiceOffering offering = serviceOfferingDao.findByIdIncludingRemoved(vm.getId(), vm.getServiceOfferingId()); + if (offering == null) { + return 0d; + } + if (useCpu) { + return (double) offering.getCpu() * offering.getSpeed(); + } + return (double) offering.getRamSize() * 1024L * 1024L; + } + + /** + * First-fit-decreasing placement of the given VMs (id to resource need) onto hosts (id to free capacity), + * returning a VM-to-host assignment or an empty map if any VM does not fit on capacity alone. This is a + * capacity feasibility pre-check only: it does NOT account for affinity/anti-affinity, host tags, storage + * access or dedication. Each migration is still validated authoritatively by the migration job, which may + * reject a placement this map proposed; in that case the host is drained over later polls rather than at once. + * Pure function, unit-tested. + */ + protected Map planHostEvacuation(Map vmNeeds, Map hostFreeCapacity) { + Map assignment = new HashMap<>(); + // TreeMap so host iteration is by id and placement is deterministic run-to-run; secondary sort by vm id + // breaks ties among equal-sized VMs for the same reason. + Map remaining = new TreeMap<>(hostFreeCapacity); + List> vms = new ArrayList<>(vmNeeds.entrySet()); + vms.sort((a, b) -> b.getValue().equals(a.getValue()) ? Long.compare(a.getKey(), b.getKey()) : Double.compare(b.getValue(), a.getValue())); + for (Map.Entry vm : vms) { + Long target = null; + for (Map.Entry host : remaining.entrySet()) { + if (host.getValue() >= vm.getValue()) { + target = host.getKey(); + break; + } + } + if (target == null) { + return Collections.emptyMap(); + } + assignment.put(vm.getKey(), target); + remaining.put(target, remaining.get(target) - vm.getValue()); + } + return assignment; + } + + /** + * Appends the latest cluster utilization ratio to the cluster's rolling history (trimmed to {@code window} + * samples) and returns a snapshot of it, newest last. + */ + protected List recordAndGetUtilizationHistory(long clusterId, double ratio, int window) { + int cap = Math.max(1, window); + LinkedList history = clusterUtilizationHistory.computeIfAbsent(clusterId, k -> new LinkedList<>()); + synchronized (history) { + history.addLast(ratio); + while (history.size() > cap) { + history.removeFirst(); + } + return new ArrayList<>(history); + } + } + + /** + * Projects one poll interval ahead from the recent utilization samples using the least-squares trend, clamped + * to [0, 1]. With fewer than two samples it returns the latest sample (no trend to project). + */ + protected double forecastUtilization(List history) { + if (history == null || history.isEmpty()) { + return 0d; + } + int n = history.size(); + if (n == 1) { + return history.get(0); + } + double sumX = 0d; + double sumY = 0d; + double sumXY = 0d; + double sumXX = 0d; + for (int i = 0; i < n; i++) { + double y = history.get(i); + sumX += i; + sumY += y; + sumXY += i * y; + sumXX += (double) i * i; + } + double denominator = n * sumXX - sumX * sumX; + double slope = denominator == 0d ? 0d : (n * sumXY - sumX * sumY) / denominator; + double forecast = history.get(n - 1) + slope; + return Math.max(0d, Math.min(1d, forecast)); + } + + protected Map> getHostCapacityMap(List hosts, boolean useCpu) { + List joins = hostJoinDao.searchByIds(hosts.stream().map(HostVO::getId).toArray(Long[]::new)); + Map> map = new HashMap<>(); + for (HostJoinVO join : joins) { + if (useCpu) { + Integer cpus = join.getCpus(); + Long speed = join.getSpeed(); + long cpuTotal = (cpus == null || speed == null) ? 0L : (long) cpus * speed; + map.put(join.getId(), new Ternary<>(join.getCpuUsedCapacity(), join.getCpuReservedCapacity(), cpuTotal)); + } else { + map.put(join.getId(), new Ternary<>(join.getMemUsedCapacity(), join.getMemReservedCapacity(), join.getTotalMemory())); + } + } + return map; + } + + /** + * True when the cluster can give up the candidate host: the cluster is below the low utilization threshold, + * there is more than one host up, and the remaining hosts can carry the current load without crossing the + * high threshold. + */ + protected boolean clusterCanReleaseHost(double clusterUsed, double clusterTotal, double candidateHostTotal, + int upHostCount, float lowThreshold, float highThreshold) { + if (upHostCount <= 1 || clusterTotal <= 0d) { + return false; + } + if ((clusterUsed / clusterTotal) >= lowThreshold) { + return false; + } + double remainingTotal = clusterTotal - candidateHostTotal; + if (remainingTotal <= 0d) { + return false; + } + return (clusterUsed / remainingTotal) <= highThreshold; + } + + /** + * True when the cluster is above the high utilization threshold and there is a host that DRS powered off + * earlier which can be powered back on. + */ + protected boolean clusterNeedsWakeup(double clusterUsed, double clusterTotal, float highThreshold, int poweredOffHostCount) { + if (poweredOffHostCount <= 0 || clusterTotal <= 0d) { + return false; + } + return (clusterUsed / clusterTotal) > highThreshold; + } + + protected boolean hostHasNoRunningVms(HostVO host) { + List vms = vmInstanceDao.listByHostId(host.getId()); + return vms == null || vms.isEmpty(); + } + + protected boolean isPowerManageable(HostVO host) { + try { + return outOfBandManagementService.isOutOfBandManagementEnabled(host); + } catch (Exception e) { + logger.debug("Could not determine out-of-band management status for host [{}]: {}", host.getId(), e.getMessage()); + return false; + } + } + + /** + * A host can be drained by power management only if every VM on it is a running user VM: a system VM cannot be + * moved with a user-VM migration, and a VM in a transitional state must not be forced, so such a host is never + * chosen (it could not be fully emptied). + */ + protected boolean hostEvacuatable(HostVO host) { + List vms = vmInstanceDao.listByHostId(host.getId()); + if (vms == null || vms.isEmpty()) { + return false; + } + for (VMInstanceVO vm : vms) { + if (vm.getType().isUsedBySystem() || vm.getState() != VirtualMachine.State.Running) { + return false; + } + } + return true; + } + + protected boolean isPoweredOffByDrs(HostVO host) { + DetailVO detail = hostDetailsDao.findDetail(host.getId(), DRS_POWER_STATE_DETAIL); + return detail != null && DRS_POWER_STATE_OFF.equals(detail.getValue()); + } + + protected void clearStalePowerMarker(HostVO host) { + DetailVO detail = hostDetailsDao.findDetail(host.getId(), DRS_POWER_STATE_DETAIL); + if (detail != null) { + hostDetailsDao.remove(detail.getId()); + } + } + + protected void powerOffHost(HostVO host, ClusterVO cluster) { + logger.info("DRS power management: cluster [{}] is under-utilized; disabling and powering off empty host [{}].", cluster.getId(), host.getId()); + // Disable first so CloudStack stops scheduling to the host and does not treat the imminent agent + // disconnect as a failure (host monitor / HA). The host also drops out of the capacity accounting at once. + hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, host); + try { + outOfBandManagementService.executePowerOperation(host, OutOfBandManagement.PowerOperation.OFF, null); + } catch (Exception e) { + // Power-off failed: undo the disable so the host stays usable, and do not mark it. + hostDao.updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, host); + throw e; + } + hostDetailsDao.persist(new DetailVO(host.getId(), DRS_POWER_STATE_DETAIL, DRS_POWER_STATE_OFF)); + } + + protected void powerOnHost(HostVO host, ClusterVO cluster) { + logger.info("DRS power management: cluster [{}] is over-utilized; powering on and re-enabling host [{}].", cluster.getId(), host.getId()); + outOfBandManagementService.executePowerOperation(host, OutOfBandManagement.PowerOperation.ON, null); + hostDao.updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, host); + DetailVO detail = hostDetailsDao.findDetail(host.getId(), DRS_POWER_STATE_DETAIL); + if (detail != null) { + hostDetailsDao.remove(detail.getId()); + } + } + @Override public ConfigKey[] getConfigKeys() { return new ConfigKey[]{ClusterDrsPlanExpireInterval, ClusterDrsEnabled, ClusterDrsInterval, ClusterDrsMaxMigrations, ClusterDrsAlgorithm, ClusterDrsImbalanceThreshold, ClusterDrsMetric, ClusterDrsMetricType, ClusterDrsMetricUseRatio, - ClusterDrsImbalanceSkipThreshold}; + ClusterDrsImbalanceSkipThreshold, ClusterDrsPowerManagementEnabled, ClusterDrsPowerManagementLowThreshold, + ClusterDrsPowerManagementHighThreshold, ClusterDrsPredictiveEnabled, ClusterDrsPredictiveWindow, + ClusterDrsPowerManagementEvacuate}; } @Override diff --git a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerManagementTest.java b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerManagementTest.java new file mode 100644 index 000000000000..41a6b26fd400 --- /dev/null +++ b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerManagementTest.java @@ -0,0 +1,150 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. +package org.apache.cloudstack.cluster; + +import java.util.Arrays; +import java.util.Collections; + +import org.junit.Assert; +import org.junit.Test; + +public class ClusterDrsPowerManagementTest { + + private final ClusterDrsServiceImpl drs = new ClusterDrsServiceImpl(); + + @Test + public void releasesHostWhenUnderUtilizedAndRemainingHostsCanCarryLoad() { + // 4 hosts of 100 each, 90 used total (22.5%): below low 0.30, and after dropping one host 90/300 = 30% <= high 0.75. + Assert.assertTrue(drs.clusterCanReleaseHost(90d, 400d, 100d, 4, 0.30f, 0.75f)); + } + + @Test + public void keepsHostWhenClusterIsNotUnderUtilized() { + // 250/400 = 62.5% is above the low threshold, so nothing is powered off. + Assert.assertFalse(drs.clusterCanReleaseHost(250d, 400d, 100d, 4, 0.30f, 0.75f)); + } + + @Test + public void keepsHostWhenRemainingHostsWouldExceedHighThreshold() { + // 40/200 = 20% used (below the low threshold), but the candidate is a large host (170 of 200): dropping it + // leaves only 30 of capacity, so 40/30 = 133% would blow past the high threshold. Keep it. + Assert.assertFalse(drs.clusterCanReleaseHost(40d, 200d, 170d, 2, 0.30f, 0.75f)); + } + + @Test + public void neverReleasesTheLastHost() { + Assert.assertFalse(drs.clusterCanReleaseHost(10d, 100d, 100d, 1, 0.30f, 0.75f)); + } + + @Test + public void wakesHostWhenOverUtilizedAndOneWasPoweredOff() { + // 320/400 = 80% is above high 0.75 and one host is available to wake. + Assert.assertTrue(drs.clusterNeedsWakeup(320d, 400d, 0.75f, 1)); + } + + @Test + public void doesNotWakeWhenNoHostWasPoweredOff() { + Assert.assertFalse(drs.clusterNeedsWakeup(320d, 400d, 0.75f, 0)); + } + + @Test + public void doesNotWakeWhenBelowHighThreshold() { + Assert.assertFalse(drs.clusterNeedsWakeup(200d, 400d, 0.75f, 1)); + } + + @Test + public void forecastProjectsARisingTrendAboveTheLastSample() { + double forecast = drs.forecastUtilization(Arrays.asList(0.40d, 0.50d, 0.60d, 0.70d)); + Assert.assertTrue("rising trend should forecast above the last sample", forecast > 0.70d); + } + + @Test + public void forecastOnFlatSeriesEqualsTheLastSample() { + Assert.assertEquals(0.50d, drs.forecastUtilization(Arrays.asList(0.50d, 0.50d, 0.50d)), 0.0001d); + } + + @Test + public void forecastOnFallingTrendIsBelowTheLastSample() { + double forecast = drs.forecastUtilization(Arrays.asList(0.80d, 0.70d, 0.60d, 0.50d)); + Assert.assertTrue("falling trend should forecast below the last sample", forecast < 0.50d); + } + + @Test + public void forecastClampsToOne() { + double forecast = drs.forecastUtilization(Arrays.asList(0.70d, 0.85d, 0.99d)); + Assert.assertTrue("forecast must never exceed 1.0", forecast <= 1.0d); + } + + @Test + public void forecastOfSingleSampleIsThatSample() { + Assert.assertEquals(0.42d, drs.forecastUtilization(Collections.singletonList(0.42d)), 0.0001d); + } + + @Test + public void forecastOfEmptyHistoryIsZero() { + Assert.assertEquals(0d, drs.forecastUtilization(Collections.emptyList()), 0.0001d); + } + + @Test + public void rollingHistoryIsTrimmedToWindowNewestLast() { + drs.recordAndGetUtilizationHistory(99L, 0.10d, 3); + drs.recordAndGetUtilizationHistory(99L, 0.20d, 3); + drs.recordAndGetUtilizationHistory(99L, 0.30d, 3); + java.util.List history = drs.recordAndGetUtilizationHistory(99L, 0.40d, 3); + Assert.assertEquals(Arrays.asList(0.20d, 0.30d, 0.40d), history); + } + + @Test + public void evacuationPlanPlacesEveryVmWhenTheyFit() { + java.util.Map vmNeeds = new java.util.HashMap<>(); + vmNeeds.put(1L, 30d); + vmNeeds.put(2L, 20d); + java.util.Map hostFree = new java.util.HashMap<>(); + hostFree.put(10L, 50d); + hostFree.put(11L, 40d); + + java.util.Map plan = drs.planHostEvacuation(vmNeeds, hostFree); + + Assert.assertEquals(2, plan.size()); + Assert.assertTrue(plan.containsKey(1L) && plan.containsKey(2L)); + } + + @Test + public void evacuationPlanIsEmptyWhenAVmCannotBePlaced() { + java.util.Map vmNeeds = new java.util.HashMap<>(); + vmNeeds.put(1L, 100d); + java.util.Map hostFree = new java.util.HashMap<>(); + hostFree.put(10L, 50d); + + Assert.assertTrue(drs.planHostEvacuation(vmNeeds, hostFree).isEmpty()); + } + + @Test + public void evacuationPlanPacksLargestFirst() { + java.util.Map vmNeeds = new java.util.HashMap<>(); + vmNeeds.put(1L, 60d); + vmNeeds.put(2L, 40d); + vmNeeds.put(3L, 40d); + java.util.Map hostFree = new java.util.HashMap<>(); + hostFree.put(10L, 60d); + hostFree.put(11L, 80d); + + java.util.Map plan = drs.planHostEvacuation(vmNeeds, hostFree); + + Assert.assertEquals("all three VMs placed", 3, plan.size()); + } +} diff --git a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java new file mode 100644 index 000000000000..ea79ce2d966d --- /dev/null +++ b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java @@ -0,0 +1,128 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. +package org.apache.cloudstack.cluster; + +import java.util.Arrays; +import java.util.Collections; + +import org.apache.cloudstack.outofbandmanagement.OutOfBandManagementService; +import org.junit.Assert; +import org.junit.Test; +import org.junit.runner.RunWith; +import org.mockito.InjectMocks; +import org.mockito.Mock; +import org.mockito.Mockito; +import org.mockito.junit.MockitoJUnitRunner; + +import com.cloud.host.HostVO; +import com.cloud.service.ServiceOfferingVO; +import com.cloud.service.dao.ServiceOfferingDao; +import com.cloud.vm.VMInstanceVO; +import com.cloud.vm.VirtualMachine; +import com.cloud.vm.dao.VMInstanceDao; + +@RunWith(MockitoJUnitRunner.class) +public class ClusterDrsPowerOrchestrationTest { + + @Mock + private VMInstanceDao vmInstanceDao; + @Mock + private ServiceOfferingDao serviceOfferingDao; + @Mock + private OutOfBandManagementService outOfBandManagementService; + + @InjectMocks + private ClusterDrsServiceImpl drs = new ClusterDrsServiceImpl(); + + private VMInstanceVO vm(long id, VirtualMachine.Type type, VirtualMachine.State state) { + VMInstanceVO vm = Mockito.mock(VMInstanceVO.class); + Mockito.lenient().when(vm.getId()).thenReturn(id); + Mockito.lenient().when(vm.getType()).thenReturn(type); + Mockito.lenient().when(vm.getState()).thenReturn(state); + return vm; + } + + private HostVO host(long id) { + HostVO host = Mockito.mock(HostVO.class); + Mockito.lenient().when(host.getId()).thenReturn(id); + return host; + } + + @Test + public void hostEvacuatableTrueForRunningUserVmsOnly() { + VMInstanceVO a = vm(1L, VirtualMachine.Type.User, VirtualMachine.State.Running); + VMInstanceVO b = vm(2L, VirtualMachine.Type.User, VirtualMachine.State.Running); + Mockito.when(vmInstanceDao.listByHostId(10L)).thenReturn(Arrays.asList(a, b)); + Assert.assertTrue(drs.hostEvacuatable(host(10L))); + } + + @Test + public void hostEvacuatableFalseWhenASystemVmIsPresent() { + VMInstanceVO user = vm(1L, VirtualMachine.Type.User, VirtualMachine.State.Running); + VMInstanceVO router = vm(3L, VirtualMachine.Type.DomainRouter, VirtualMachine.State.Running); + Mockito.when(vmInstanceDao.listByHostId(11L)).thenReturn(Arrays.asList(user, router)); + Assert.assertFalse(drs.hostEvacuatable(host(11L))); + } + + @Test + public void hostEvacuatableFalseWhenAVmIsNotRunning() { + VMInstanceVO stopping = vm(4L, VirtualMachine.Type.User, VirtualMachine.State.Stopping); + Mockito.when(vmInstanceDao.listByHostId(12L)).thenReturn(Collections.singletonList(stopping)); + Assert.assertFalse(drs.hostEvacuatable(host(12L))); + } + + @Test + public void hostEvacuatableFalseWhenEmpty() { + Mockito.when(vmInstanceDao.listByHostId(13L)).thenReturn(Collections.emptyList()); + Assert.assertFalse(drs.hostEvacuatable(host(13L))); + } + + @Test + public void vmResourceNeedZeroWhenOfferingMissing() { + VMInstanceVO vm = vm(5L, VirtualMachine.Type.User, VirtualMachine.State.Running); + Mockito.when(serviceOfferingDao.findByIdIncludingRemoved(Mockito.eq(5L), Mockito.anyLong())).thenReturn(null); + Assert.assertEquals(0d, drs.vmResourceNeed(vm, true), 0.0001d); + } + + @Test + public void vmResourceNeedComputesCpuAndMemory() { + VMInstanceVO vm = vm(6L, VirtualMachine.Type.User, VirtualMachine.State.Running); + Mockito.when(vm.getServiceOfferingId()).thenReturn(99L); + ServiceOfferingVO offering = Mockito.mock(ServiceOfferingVO.class); + Mockito.when(offering.getCpu()).thenReturn(4); + Mockito.when(offering.getSpeed()).thenReturn(2000); + Mockito.when(offering.getRamSize()).thenReturn(2048); + Mockito.when(serviceOfferingDao.findByIdIncludingRemoved(6L, 99L)).thenReturn(offering); + + Assert.assertEquals(8000d, drs.vmResourceNeed(vm, true), 0.0001d); + Assert.assertEquals(2048d * 1024L * 1024L, drs.vmResourceNeed(vm, false), 0.0001d); + } + + @Test + public void isPowerManageableReflectsOobmEnabled() { + HostVO h = host(20L); + Mockito.when(outOfBandManagementService.isOutOfBandManagementEnabled(h)).thenReturn(true); + Assert.assertTrue(drs.isPowerManageable(h)); + } + + @Test + public void isPowerManageableFalseWhenOobmLookupThrows() { + HostVO h = host(21L); + Mockito.when(outOfBandManagementService.isOutOfBandManagementEnabled(h)).thenThrow(new RuntimeException("boom")); + Assert.assertFalse(drs.isPowerManageable(h)); + } +} From 53a13c628c2d5575ef579c495c55d999f83fcdcd Mon Sep 17 00:00:00 2001 From: Ramgopal Nagaboina Date: Wed, 7 Oct 2026 17:49:36 -0400 Subject: [PATCH 2/7] drs: harden cluster power management lifecycle Fixes found by auditing the power orchestration: - persist the wake marker before the irreversible power-off and remove it on rollback, so a crash between the power-off and the marker write no longer strands a host powered off and unrecognised. - abort the power-off when the host could not be disabled, instead of powering it off while CloudStack still believes it is schedulable. - keep the marker and leave the host disabled after issuing power-on, and reclaim it (re-enable, clear the marker) only once the host is actually Up, so an accepted but unconfirmed power-on cannot strand the host, and a host returned by any path is never left disabled and marked forever. - skip power management for a cluster with a DRS migration plan in flight, so a migration source or destination is never powered off under it. - cap each evacuation destination at the high utilization threshold so draining a host does not push another host past it. Adds unit tests for the power-off ordering, disable gating and rollback, the power-on marker handling, and the in-flight-plan guard. --- .../cluster/ClusterDrsServiceImpl.java | 75 ++++++++++----- .../ClusterDrsPowerOrchestrationTest.java | 94 +++++++++++++++++++ 2 files changed, 148 insertions(+), 21 deletions(-) diff --git a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java index adc744667de0..a83e57956857 100644 --- a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java +++ b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java @@ -892,6 +892,13 @@ protected void managePowerForCluster(ClusterVO cluster) { || Boolean.FALSE.equals(ClusterDrsPowerManagementEnabled.valueIn(cluster.getId()))) { return; } + // Do not power-manage a cluster with a DRS migration plan still in flight: a host that is the source or + // destination of a pending or running migration must not be disabled or powered off underneath it. + if (hasInFlightDrsPlan(cluster.getId())) { + logger.debug("DRS power management: skipping cluster [{}] while a DRS migration plan is in flight.", cluster.getId()); + return; + } + final float lowThreshold = ClusterDrsPowerManagementLowThreshold.valueIn(cluster.getId()); final float highThreshold = ClusterDrsPowerManagementHighThreshold.valueIn(cluster.getId()); final boolean useCpu = "cpu".equals(getClusterDrsMetric(cluster.getId())); @@ -900,14 +907,24 @@ protected void managePowerForCluster(ClusterVO cluster) { List upHosts = new ArrayList<>(); List poweredOffByDrs = new ArrayList<>(); for (HostVO host : routingHosts) { - if (host.getStatus() == Status.Up && host.getResourceState() == ResourceState.Enabled) { - // A host that is Up+Enabled is in service; clear any stale DRS power marker left by a power-off - // that failed or by a host that came back by another path, so it is not excluded forever. - clearStalePowerMarker(host); - upHosts.add(host); - } else if (host.getStatus() != Status.Up && isPoweredOffByDrs(host)) { - // Only a host that is actually down (disconnected) and was powered off by DRS is a wake candidate; - // a host still Up while shutting down is in neither list. + if (host.getStatus() == Status.Up) { + if (isPoweredOffByDrs(host)) { + // Our host is back in service: either a wake completed, or a power-off we issued never took + // effect (command accepted but the host stayed up). Re-enable it if we had disabled it, drop + // the marker, and return it to the capacity pool. Keying only on Up avoids leaving such a host + // stranded Disabled and marked, in neither list, forever. + if (host.getResourceState() == ResourceState.Disabled) { + hostDao.updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, host); + } + clearStalePowerMarker(host); + upHosts.add(host); + } else if (host.getResourceState() == ResourceState.Enabled) { + upHosts.add(host); + } + // Up but Disabled by someone other than DRS: leave it alone, it is not ours to schedule onto. + } else if (isPoweredOffByDrs(host)) { + // Down and marked by DRS: a wake candidate. The marker is kept across the wake (power-on is + // idempotent) and cleared only once the host is actually Up again, above. poweredOffByDrs.add(host); } } @@ -1002,9 +1019,11 @@ protected void evacuateReleasableHost(ClusterVO cluster, List upHosts, M } Ternary cap = capacityMap.get(host.getId()); if (cap != null) { - // Free = total - (used + reserved), consistent with the release decision, so evacuated VMs are not - // placed onto capacity another host is holding for HA/allocation reservations. - hostFree.put(host.getId(), (double) cap.third() - (cap.first() + cap.second())); + // Placeable room = (high threshold of total) - (used + reserved), floored at zero. Capping each + // destination at the high threshold keeps per-host placement consistent with the aggregate release + // decision, so draining a host never pushes another one past the threshold it is meant to respect. + double placeable = (double) highThreshold * cap.third() - (cap.first() + cap.second()); + hostFree.put(host.getId(), Math.max(0d, placeable)); } } @@ -1216,29 +1235,43 @@ protected void clearStalePowerMarker(HostVO host) { } } + protected boolean hasInFlightDrsPlan(long clusterId) { + return !drsPlanDao.listByClusterIdAndStatus(clusterId, ClusterDrsPlan.Status.UNDER_REVIEW).isEmpty() + || !drsPlanDao.listByClusterIdAndStatus(clusterId, ClusterDrsPlan.Status.READY).isEmpty() + || !drsPlanDao.listByClusterIdAndStatus(clusterId, ClusterDrsPlan.Status.IN_PROGRESS).isEmpty(); + } + protected void powerOffHost(HostVO host, ClusterVO cluster) { logger.info("DRS power management: cluster [{}] is under-utilized; disabling and powering off empty host [{}].", cluster.getId(), host.getId()); // Disable first so CloudStack stops scheduling to the host and does not treat the imminent agent - // disconnect as a failure (host monitor / HA). The host also drops out of the capacity accounting at once. - hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, host); + // disconnect as a failure (host monitor / HA). Abort if the transition did not take effect, so the + // host is never powered off while CloudStack still believes it is schedulable. + if (!hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, host)) { + logger.warn("DRS power management: could not disable host [{}]; skipping power-off.", host.getId()); + return; + } + // Record the durable wake intent BEFORE the irreversible power-off. If the management server dies between + // the power-off and here, the host is still recognised on the next poll and powered back on, rather than + // left off forever with no marker. + DetailVO marker = new DetailVO(host.getId(), DRS_POWER_STATE_DETAIL, DRS_POWER_STATE_OFF); + hostDetailsDao.persist(marker); try { outOfBandManagementService.executePowerOperation(host, OutOfBandManagement.PowerOperation.OFF, null); } catch (Exception e) { - // Power-off failed: undo the disable so the host stays usable, and do not mark it. + // Power-off failed: undo the marker and the disable so the host stays in service. + hostDetailsDao.remove(marker.getId()); hostDao.updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, host); throw e; } - hostDetailsDao.persist(new DetailVO(host.getId(), DRS_POWER_STATE_DETAIL, DRS_POWER_STATE_OFF)); } protected void powerOnHost(HostVO host, ClusterVO cluster) { - logger.info("DRS power management: cluster [{}] is over-utilized; powering on and re-enabling host [{}].", cluster.getId(), host.getId()); + logger.info("DRS power management: cluster [{}] is over-utilized; powering on host [{}].", cluster.getId(), host.getId()); + // Issue the power-on but keep the marker and leave the host Disabled: the host is still down until its + // agent reconnects, and an accepted-but-unconfirmed power-on must not look done. The classification loop + // re-enables the host and clears the marker only once it is actually Up. Power-on is idempotent, so a host + // still booting is simply re-issued the command on a later poll until it connects. outOfBandManagementService.executePowerOperation(host, OutOfBandManagement.PowerOperation.ON, null); - hostDao.updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, host); - DetailVO detail = hostDetailsDao.findDetail(host.getId(), DRS_POWER_STATE_DETAIL); - if (detail != null) { - hostDetailsDao.remove(detail.getId()); - } } @Override diff --git a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java index ea79ce2d966d..82786daada0a 100644 --- a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java +++ b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java @@ -19,16 +19,24 @@ import java.util.Arrays; import java.util.Collections; +import org.apache.cloudstack.outofbandmanagement.OutOfBandManagement; import org.apache.cloudstack.outofbandmanagement.OutOfBandManagementService; +import org.apache.cloudstack.cluster.dao.ClusterDrsPlanDao; import org.junit.Assert; import org.junit.Test; import org.junit.runner.RunWith; import org.mockito.InjectMocks; +import org.mockito.InOrder; import org.mockito.Mock; import org.mockito.Mockito; import org.mockito.junit.MockitoJUnitRunner; +import com.cloud.host.DetailVO; import com.cloud.host.HostVO; +import com.cloud.host.dao.HostDao; +import com.cloud.host.dao.HostDetailsDao; +import com.cloud.dc.ClusterVO; +import com.cloud.resource.ResourceState; import com.cloud.service.ServiceOfferingVO; import com.cloud.service.dao.ServiceOfferingDao; import com.cloud.vm.VMInstanceVO; @@ -44,10 +52,22 @@ public class ClusterDrsPowerOrchestrationTest { private ServiceOfferingDao serviceOfferingDao; @Mock private OutOfBandManagementService outOfBandManagementService; + @Mock + private HostDao hostDao; + @Mock + private HostDetailsDao hostDetailsDao; + @Mock + private ClusterDrsPlanDao drsPlanDao; @InjectMocks private ClusterDrsServiceImpl drs = new ClusterDrsServiceImpl(); + private ClusterVO cluster(long id) { + ClusterVO cluster = Mockito.mock(ClusterVO.class); + Mockito.lenient().when(cluster.getId()).thenReturn(id); + return cluster; + } + private VMInstanceVO vm(long id, VirtualMachine.Type type, VirtualMachine.State state) { VMInstanceVO vm = Mockito.mock(VMInstanceVO.class); Mockito.lenient().when(vm.getId()).thenReturn(id); @@ -125,4 +145,78 @@ public void isPowerManageableFalseWhenOobmLookupThrows() { Mockito.when(outOfBandManagementService.isOutOfBandManagementEnabled(h)).thenThrow(new RuntimeException("boom")); Assert.assertFalse(drs.isPowerManageable(h)); } + + @Test + public void powerOffPersistsWakeMarkerBeforeThePowerOff() { + HostVO h = host(30L); + Mockito.when(hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, h)).thenReturn(true); + + drs.powerOffHost(h, cluster(1L)); + + // the durable wake marker must be written BEFORE the irreversible power-off, so a crash in between + // still leaves a host that can be recognised and powered back on. + InOrder inOrder = Mockito.inOrder(hostDao, hostDetailsDao, outOfBandManagementService); + inOrder.verify(hostDao).updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, h); + inOrder.verify(hostDetailsDao).persist(Mockito.any(DetailVO.class)); + inOrder.verify(outOfBandManagementService).executePowerOperation(Mockito.eq(h), Mockito.eq(OutOfBandManagement.PowerOperation.OFF), Mockito.any()); + } + + @Test + public void powerOffAbortsWhenTheHostCannotBeDisabled() { + HostVO h = host(31L); + Mockito.when(hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, h)).thenReturn(false); + + drs.powerOffHost(h, cluster(1L)); + + // a host that could not be disabled must never be powered off, and must not be marked. + Mockito.verify(outOfBandManagementService, Mockito.never()).executePowerOperation(Mockito.any(), Mockito.any(), Mockito.any()); + Mockito.verify(hostDetailsDao, Mockito.never()).persist(Mockito.any(DetailVO.class)); + } + + @Test + public void powerOffRollsBackMarkerAndDisableWhenThePowerOffFails() { + HostVO h = host(32L); + Mockito.when(hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, h)).thenReturn(true); + Mockito.doThrow(new RuntimeException("oobm down")).when(outOfBandManagementService) + .executePowerOperation(Mockito.eq(h), Mockito.eq(OutOfBandManagement.PowerOperation.OFF), Mockito.any()); + + try { + drs.powerOffHost(h, cluster(1L)); + Assert.fail("expected the power-off failure to propagate"); + } catch (RuntimeException expected) { + // expected + } + + // on failure the marker is removed and the host is re-enabled, so it is not left disabled-and-marked. + Mockito.verify(hostDetailsDao).remove(Mockito.anyLong()); + Mockito.verify(hostDao).updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, h); + } + + @Test + public void powerOnKeepsTheMarkerAndDoesNotEnableUntilTheHostIsUp() { + HostVO h = host(33L); + + drs.powerOnHost(h, cluster(1L)); + + // the power-on is issued, but the marker is kept and the host is not re-enabled: an unconfirmed power-on + // must not look done, or a host that never actually boots is stranded out of the wake set. + Mockito.verify(outOfBandManagementService).executePowerOperation(Mockito.eq(h), Mockito.eq(OutOfBandManagement.PowerOperation.ON), Mockito.any()); + Mockito.verify(hostDetailsDao, Mockito.never()).remove(Mockito.anyLong()); + Mockito.verify(hostDao, Mockito.never()).updateResourceState(Mockito.any(), Mockito.any(), Mockito.any(), Mockito.any()); + } + + @Test + public void hasInFlightDrsPlanTrueWhenAPlanIsInProgress() { + Mockito.when(drsPlanDao.listByClusterIdAndStatus(5L, ClusterDrsPlan.Status.UNDER_REVIEW)).thenReturn(Collections.emptyList()); + Mockito.when(drsPlanDao.listByClusterIdAndStatus(5L, ClusterDrsPlan.Status.READY)).thenReturn(Collections.emptyList()); + Mockito.when(drsPlanDao.listByClusterIdAndStatus(5L, ClusterDrsPlan.Status.IN_PROGRESS)) + .thenReturn(Collections.singletonList(Mockito.mock(ClusterDrsPlanVO.class))); + Assert.assertTrue(drs.hasInFlightDrsPlan(5L)); + } + + @Test + public void hasInFlightDrsPlanFalseWhenNoPlansArePending() { + Mockito.when(drsPlanDao.listByClusterIdAndStatus(Mockito.eq(5L), Mockito.any())).thenReturn(Collections.emptyList()); + Assert.assertFalse(drs.hasInFlightDrsPlan(5L)); + } } From cfad78797eb218eaad0c488973fa0d2bb59ed917 Mon Sep 17 00:00:00 2001 From: Ramgopal Nagaboina Date: Wed, 7 Oct 2026 18:08:30 -0400 Subject: [PATCH 3/7] drs: back off and abandon a power-management drain that makes no progress The evacuation path drained the least-loaded releasable host by submitting migrations and re-picking it each poll until empty. A VM that is unplaceable for affinity, host tags or storage passes the capacity pre-check but is rejected by the migration itself, so the host was re-submitted every poll and left half-drained. Track a drain with a distinct draining host marker and the host's VM count at the last batch. Continue a drain already in progress before starting another, wait while a batch is still in flight, and abandon the drain (clearing the marker) once it makes no progress across a poll or the remaining VMs no longer fit, instead of re-submitting forever. Power-off replaces the draining marker with the off marker once the host is empty. --- .../cluster/ClusterDrsServiceImpl.java | 126 ++++++++++++++++-- .../ClusterDrsPowerOrchestrationTest.java | 68 ++++++++++ 2 files changed, 180 insertions(+), 14 deletions(-) diff --git a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java index a83e57956857..27b2a3fa30d3 100644 --- a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java +++ b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java @@ -139,11 +139,18 @@ public class ClusterDrsServiceImpl extends ManagerBase implements ClusterDrsServ // (never a host that is down for another reason). protected static final String DRS_POWER_STATE_DETAIL = "drs.power.state"; protected static final String DRS_POWER_STATE_OFF = "off"; + // Marks a host DRS is actively draining so it can be powered off once empty; distinct from OFF so a draining + // host is never treated as a wake candidate and is picked up across polls to continue or abandon the drain. + protected static final String DRS_POWER_STATE_DRAINING = "draining"; // Recent cluster utilization samples (ratio on the DRS metric) per cluster, newest last, used by predictive DRS. // Written only from the single-threaded poll under the clusterDRS.poll lock. protected final Map> clusterUtilizationHistory = new ConcurrentHashMap<>(); + // VM count on each host at the last drain batch, so a drain that stops making progress (migrations rejected for + // affinity/tags/storage) is abandoned instead of re-submitted forever. Written only under the poll lock. + protected final Map drainingVmCountByHost = new ConcurrentHashMap<>(); + @Inject HostJoinDao hostJoinDao; @@ -984,14 +991,69 @@ && clusterCanReleaseHost(effectiveUsed, clusterTotal, capacity.third(), upHosts. } } + protected boolean isDrainingByDrs(HostVO host) { + DetailVO detail = hostDetailsDao.findDetail(host.getId(), DRS_POWER_STATE_DETAIL); + return detail != null && DRS_POWER_STATE_DRAINING.equals(detail.getValue()); + } + + protected boolean hasMigratingVm(long hostId) { + List vms = vmInstanceDao.listByHostId(hostId); + if (vms == null) { + return false; + } + for (VMInstanceVO vm : vms) { + if (vm.getState() == VirtualMachine.State.Migrating) { + return true; + } + } + return false; + } + + protected void setPowerMarker(long hostId, String value) { + DetailVO existing = hostDetailsDao.findDetail(hostId, DRS_POWER_STATE_DETAIL); + if (existing != null) { + hostDetailsDao.remove(existing.getId()); + } + hostDetailsDao.persist(new DetailVO(hostId, DRS_POWER_STATE_DETAIL, value)); + } + + protected void abandonDrain(HostVO host, String reason) { + logger.info("DRS power management: abandoning drain of host [{}]: {}", host.getId(), reason); + clearStalePowerMarker(host); + drainingVmCountByHost.remove(host.getId()); + } + /** - * Picks the least-loaded host the cluster can give up, and if every VM on it can be placed on the remaining - * hosts, migrates them away (a later poll powers the now-empty host off). Does nothing if no host qualifies or - * if the VMs cannot all be placed, so a host is never left partially drained. + * Drives the eviction side of power management when no host is empty yet. Continues a drain already in progress + * before starting another, otherwise starts draining the least-loaded releasable host. A drain that stops making + * progress across polls (its migrations keep being rejected for affinity, host tags or storage) is abandoned + * instead of re-submitted forever, so a host is never left indefinitely half-drained. */ protected void evacuateReleasableHost(ClusterVO cluster, List upHosts, Map> capacityMap, double effectiveUsed, double clusterTotal, float lowThreshold, float highThreshold, boolean useCpu) { HostVO candidate = null; + for (HostVO host : upHosts) { + if (isDrainingByDrs(host)) { + candidate = host; + break; + } + } + if (candidate == null) { + candidate = selectDrainCandidate(upHosts, capacityMap, effectiveUsed, clusterTotal, lowThreshold, highThreshold); + } + if (candidate == null) { + return; + } + drainHostBatch(cluster, candidate, upHosts, capacityMap, highThreshold, useCpu); + } + + /** + * The least-loaded host the cluster can give up: powerable, holding only running user VMs, and such that the + * remaining hosts can still carry the load. Null if no host qualifies. + */ + protected HostVO selectDrainCandidate(List upHosts, Map> capacityMap, + double effectiveUsed, double clusterTotal, float lowThreshold, float highThreshold) { + HostVO candidate = null; double candidateUsed = Double.MAX_VALUE; for (HostVO host : upHosts) { Ternary cap = capacityMap.get(host.getId()); @@ -1004,12 +1066,36 @@ protected void evacuateReleasableHost(ClusterVO cluster, List upHosts, M candidateUsed = cap.first(); } } - if (candidate == null) { + return candidate; + } + + /** + * Submits one batch of migrations off {@code candidate}. Waits while a previous batch is still in flight; if the + * host is quiescent and still holds as many VMs as at the previous batch the drain made no progress and is + * abandoned; otherwise the host is marked draining and up to ClusterDrsMaxMigrations of its remaining VMs are + * placed by capacity and migrated away. The host empties over successive polls and is powered off once empty. + */ + protected void drainHostBatch(ClusterVO cluster, HostVO candidate, List upHosts, + Map> capacityMap, float highThreshold, boolean useCpu) { + List vms = vmInstanceDao.listByHostId(candidate.getId()); + if (vms == null || vms.isEmpty()) { + // fully drained; the empty-host power-off path takes it from here. + abandonDrain(candidate, "host is empty"); + return; + } + if (hasMigratingVm(candidate.getId())) { + // a batch is still in flight; wait for it before judging progress or submitting more. + return; + } + int current = vms.size(); + Integer previous = drainingVmCountByHost.get(candidate.getId()); + if (previous != null && current >= previous) { + abandonDrain(candidate, "no progress since the last batch; remaining VMs cannot be migrated off"); return; } Map vmNeeds = new HashMap<>(); - for (VMInstanceVO vm : vmInstanceDao.listByHostId(candidate.getId())) { + for (VMInstanceVO vm : vms) { vmNeeds.put(vm.getId(), vmResourceNeed(vm, useCpu)); } Map hostFree = new HashMap<>(); @@ -1019,9 +1105,8 @@ protected void evacuateReleasableHost(ClusterVO cluster, List upHosts, M } Ternary cap = capacityMap.get(host.getId()); if (cap != null) { - // Placeable room = (high threshold of total) - (used + reserved), floored at zero. Capping each - // destination at the high threshold keeps per-host placement consistent with the aggregate release - // decision, so draining a host never pushes another one past the threshold it is meant to respect. + // Placeable room = (high threshold of total) - (used + reserved), floored at zero, so draining + // never pushes a destination past the high threshold it is meant to respect. double placeable = (double) highThreshold * cap.third() - (cap.first() + cap.second()); hostFree.put(host.getId(), Math.max(0d, placeable)); } @@ -1029,14 +1114,17 @@ protected void evacuateReleasableHost(ClusterVO cluster, List upHosts, M Map plan = planHostEvacuation(vmNeeds, hostFree); if (plan.isEmpty()) { - logger.debug("DRS power management: host [{}] in cluster [{}] cannot be evacuated; not all VMs fit on the remaining hosts.", - candidate.getId(), cluster.getId()); + if (previous != null) { + abandonDrain(candidate, "remaining VMs no longer fit on the other hosts"); + } else { + logger.debug("DRS power management: host [{}] in cluster [{}] cannot be evacuated; not all VMs fit on the remaining hosts.", + candidate.getId(), cluster.getId()); + } return; } - // respect the cluster's DRS migration budget: drain at most ClusterDrsMaxMigrations VMs per poll, the host - // empties over subsequent polls and is powered off once truly empty. + int maxMigrations = ClusterDrsMaxMigrations.valueIn(cluster.getId()); - logger.info("DRS power management: cluster [{}] is under-utilized; draining host [{}] ({} VMs, up to {} per poll) to power it off.", + logger.info("DRS power management: cluster [{}] is under-utilized; draining host [{}] ({} VMs left, up to {} per poll) to power it off.", cluster.getId(), candidate.getId(), plan.size(), maxMigrations); long eventId = ActionEventUtils.onStartedActionEvent(User.UID_SYSTEM, Account.ACCOUNT_ID_SYSTEM, EventTypes.EVENT_VM_MIGRATE, @@ -1054,6 +1142,10 @@ protected void evacuateReleasableHost(ClusterVO cluster, List upHosts, M submitted++; } } + if (submitted > 0) { + setPowerMarker(candidate.getId(), DRS_POWER_STATE_DRAINING); + drainingVmCountByHost.put(candidate.getId(), current); + } } protected double vmResourceNeed(VMInstanceVO vm, boolean useCpu) { @@ -1252,7 +1344,12 @@ protected void powerOffHost(HostVO host, ClusterVO cluster) { } // Record the durable wake intent BEFORE the irreversible power-off. If the management server dies between // the power-off and here, the host is still recognised on the next poll and powered back on, rather than - // left off forever with no marker. + // left off forever with no marker. Replace any existing marker (e.g. a draining marker when powering off a + // host that has just finished draining) so only the off marker remains. + DetailVO existing = hostDetailsDao.findDetail(host.getId(), DRS_POWER_STATE_DETAIL); + if (existing != null) { + hostDetailsDao.remove(existing.getId()); + } DetailVO marker = new DetailVO(host.getId(), DRS_POWER_STATE_DETAIL, DRS_POWER_STATE_OFF); hostDetailsDao.persist(marker); try { @@ -1263,6 +1360,7 @@ protected void powerOffHost(HostVO host, ClusterVO cluster) { hostDao.updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, host); throw e; } + drainingVmCountByHost.remove(host.getId()); } protected void powerOnHost(HostVO host, ClusterVO cluster) { diff --git a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java index 82786daada0a..63913f28e534 100644 --- a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java +++ b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java @@ -219,4 +219,72 @@ public void hasInFlightDrsPlanFalseWhenNoPlansArePending() { Mockito.when(drsPlanDao.listByClusterIdAndStatus(Mockito.eq(5L), Mockito.any())).thenReturn(Collections.emptyList()); Assert.assertFalse(drs.hasInFlightDrsPlan(5L)); } + + @Test + public void hasMigratingVmDetectsAnInFlightMigration() { + VMInstanceVO running = vm(1L, VirtualMachine.Type.User, VirtualMachine.State.Running); + VMInstanceVO migrating = vm(2L, VirtualMachine.Type.User, VirtualMachine.State.Migrating); + Mockito.when(vmInstanceDao.listByHostId(40L)).thenReturn(Arrays.asList(running, migrating)); + Assert.assertTrue(drs.hasMigratingVm(40L)); + } + + @Test + public void hasMigratingVmFalseWhenNothingIsMigrating() { + VMInstanceVO running = vm(1L, VirtualMachine.Type.User, VirtualMachine.State.Running); + Mockito.when(vmInstanceDao.listByHostId(40L)).thenReturn(Collections.singletonList(running)); + Assert.assertFalse(drs.hasMigratingVm(40L)); + } + + @Test + public void isDrainingByDrsTrueOnlyForTheDrainingMarker() { + HostVO h = host(41L); + Mockito.when(hostDetailsDao.findDetail(41L, "drs.power.state")) + .thenReturn(new DetailVO(41L, "drs.power.state", "draining")); + Assert.assertTrue(drs.isDrainingByDrs(h)); + } + + @Test + public void drainBatchWaitsWhileAMigrationIsInFlight() { + HostVO candidate = host(42L); + VMInstanceVO migrating = vm(1L, VirtualMachine.Type.User, VirtualMachine.State.Migrating); + Mockito.when(vmInstanceDao.listByHostId(42L)).thenReturn(Collections.singletonList(migrating)); + drs.drainingVmCountByHost.put(42L, 3); + + drs.drainHostBatch(cluster(1L), candidate, Collections.emptyList(), Collections.emptyMap(), 0.75f, true); + + // a batch is still running: do not touch the marker or the recorded progress, just wait. + Mockito.verify(hostDetailsDao, Mockito.never()).remove(Mockito.anyLong()); + Assert.assertEquals(Integer.valueOf(3), drs.drainingVmCountByHost.get(42L)); + } + + @Test + public void drainBatchAbandonsWhenNoProgressSinceLastBatch() { + HostVO candidate = host(43L); + VMInstanceVO a = vm(1L, VirtualMachine.Type.User, VirtualMachine.State.Running); + VMInstanceVO b = vm(2L, VirtualMachine.Type.User, VirtualMachine.State.Running); + Mockito.when(vmInstanceDao.listByHostId(43L)).thenReturn(Arrays.asList(a, b)); + Mockito.when(hostDetailsDao.findDetail(43L, "drs.power.state")) + .thenReturn(new DetailVO(43L, "drs.power.state", "draining")); + drs.drainingVmCountByHost.put(43L, 2); // same count as now: the last batch moved nothing + + drs.drainHostBatch(cluster(1L), candidate, Collections.emptyList(), Collections.emptyMap(), 0.75f, true); + + // no progress: the drain is abandoned (marker cleared, progress forgotten) instead of re-submitted forever. + Mockito.verify(hostDetailsDao).remove(Mockito.anyLong()); + Assert.assertFalse(drs.drainingVmCountByHost.containsKey(43L)); + } + + @Test + public void drainBatchAbandonsWhenTheHostIsAlreadyEmpty() { + HostVO candidate = host(44L); + Mockito.when(vmInstanceDao.listByHostId(44L)).thenReturn(Collections.emptyList()); + Mockito.when(hostDetailsDao.findDetail(44L, "drs.power.state")) + .thenReturn(new DetailVO(44L, "drs.power.state", "draining")); + drs.drainingVmCountByHost.put(44L, 1); + + drs.drainHostBatch(cluster(1L), candidate, Collections.emptyList(), Collections.emptyMap(), 0.75f, true); + + Mockito.verify(hostDetailsDao).remove(Mockito.anyLong()); + Assert.assertFalse(drs.drainingVmCountByHost.containsKey(44L)); + } } From 05f483a827b04eb12923207c37c9990821239a8e Mon Sep 17 00:00:00 2001 From: Ramgopal Nagaboina Date: Wed, 7 Oct 2026 18:19:04 -0400 Subject: [PATCH 4/7] drs: suppress the host-down alert on power-off and avoid an orphaned drain event - detach the agent without investigation before cutting power, so the link drop from a DRS power-off is not reported as a host-down failure. The host is already disabled and empty and is going away intentionally. - resolve the concrete VM and destination pairs before opening the drain action event, and open it only when the batch is non-empty, so a poll that finds nothing actually migratable no longer leaves a start event with no matching terminal event. --- .../cluster/ClusterDrsServiceImpl.java | 41 +++++++++++++------ .../ClusterDrsPowerOrchestrationTest.java | 10 ++++- 2 files changed, 36 insertions(+), 15 deletions(-) diff --git a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java index 27b2a3fa30d3..3905f528f42c 100644 --- a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java +++ b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java @@ -19,6 +19,7 @@ package org.apache.cloudstack.cluster; +import com.cloud.agent.AgentManager; import com.cloud.api.ApiGsonHelper; import com.cloud.api.query.dao.HostJoinDao; import com.cloud.api.query.vo.HostJoinVO; @@ -95,6 +96,7 @@ import java.util.Collections; import java.util.Date; import java.util.HashMap; +import java.util.LinkedHashMap; import java.util.LinkedList; import java.util.concurrent.ConcurrentHashMap; import java.util.HashSet; @@ -154,6 +156,9 @@ public class ClusterDrsServiceImpl extends ManagerBase implements ClusterDrsServ @Inject HostJoinDao hostJoinDao; + @Inject + AgentManager agentManager; + @Inject VMInstanceDao vmInstanceDao; @@ -1123,29 +1128,36 @@ protected void drainHostBatch(ClusterVO cluster, HostVO candidate, List return; } + // Resolve the concrete VM/destination pairs to migrate this poll (bounded by the DRS migration budget) + // before starting any event, so the start event is only opened when there is real work to do. int maxMigrations = ClusterDrsMaxMigrations.valueIn(cluster.getId()); - logger.info("DRS power management: cluster [{}] is under-utilized; draining host [{}] ({} VMs left, up to {} per poll) to power it off.", - cluster.getId(), candidate.getId(), plan.size(), maxMigrations); - long eventId = ActionEventUtils.onStartedActionEvent(User.UID_SYSTEM, Account.ACCOUNT_ID_SYSTEM, - EventTypes.EVENT_VM_MIGRATE, - String.format("DRS power management draining host %d in cluster %s", candidate.getId(), cluster.getUuid()), - candidate.getId(), ApiCommandResourceType.Host.toString(), true, 0); - int submitted = 0; + Map batch = new LinkedHashMap<>(); for (Map.Entry entry : plan.entrySet()) { - if (submitted >= maxMigrations) { + if (batch.size() >= maxMigrations) { break; } VirtualMachine vm = vmInstanceDao.findById(entry.getKey()); HostVO destination = hostDao.findById(entry.getValue()); if (vm != null && destination != null) { - createMigrateVMAsyncJob(vm, destination, eventId); - submitted++; + batch.put(vm, destination); } } - if (submitted > 0) { - setPowerMarker(candidate.getId(), DRS_POWER_STATE_DRAINING); - drainingVmCountByHost.put(candidate.getId(), current); + if (batch.isEmpty()) { + return; + } + logger.info("DRS power management: cluster [{}] is under-utilized; draining host [{}] ({} VMs left, {} this poll) to power it off.", + cluster.getId(), candidate.getId(), plan.size(), batch.size()); + // One start event for the batch. Each migration job carries it as its start event id and completes it, as in + // executeDrsPlan, so the event is never left open; opening it only when the batch is non-empty avoids an orphan. + long eventId = ActionEventUtils.onStartedActionEvent(User.UID_SYSTEM, Account.ACCOUNT_ID_SYSTEM, + EventTypes.EVENT_VM_MIGRATE, + String.format("DRS power management draining host %d in cluster %s", candidate.getId(), cluster.getUuid()), + candidate.getId(), ApiCommandResourceType.Host.toString(), true, 0); + for (Map.Entry migration : batch.entrySet()) { + createMigrateVMAsyncJob(migration.getKey(), migration.getValue(), eventId); } + setPowerMarker(candidate.getId(), DRS_POWER_STATE_DRAINING); + drainingVmCountByHost.put(candidate.getId(), current); } protected double vmResourceNeed(VMInstanceVO vm, boolean useCpu) { @@ -1353,6 +1365,9 @@ protected void powerOffHost(HostVO host, ClusterVO cluster) { DetailVO marker = new DetailVO(host.getId(), DRS_POWER_STATE_DETAIL, DRS_POWER_STATE_OFF); hostDetailsDao.persist(marker); try { + // Detach the agent without investigation first so the imminent link drop from cutting power is not + // reported as a host-down failure (the host is intentionally going away, and it is already empty). + agentManager.disconnectWithoutInvestigation(host.getId(), Status.Event.ShutdownRequested); outOfBandManagementService.executePowerOperation(host, OutOfBandManagement.PowerOperation.OFF, null); } catch (Exception e) { // Power-off failed: undo the marker and the disable so the host stays in service. diff --git a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java index 63913f28e534..07bae34073ba 100644 --- a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java +++ b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java @@ -22,6 +22,8 @@ import org.apache.cloudstack.outofbandmanagement.OutOfBandManagement; import org.apache.cloudstack.outofbandmanagement.OutOfBandManagementService; import org.apache.cloudstack.cluster.dao.ClusterDrsPlanDao; +import com.cloud.agent.AgentManager; +import com.cloud.host.Status; import org.junit.Assert; import org.junit.Test; import org.junit.runner.RunWith; @@ -58,6 +60,8 @@ public class ClusterDrsPowerOrchestrationTest { private HostDetailsDao hostDetailsDao; @Mock private ClusterDrsPlanDao drsPlanDao; + @Mock + private AgentManager agentManager; @InjectMocks private ClusterDrsServiceImpl drs = new ClusterDrsServiceImpl(); @@ -154,10 +158,12 @@ public void powerOffPersistsWakeMarkerBeforeThePowerOff() { drs.powerOffHost(h, cluster(1L)); // the durable wake marker must be written BEFORE the irreversible power-off, so a crash in between - // still leaves a host that can be recognised and powered back on. - InOrder inOrder = Mockito.inOrder(hostDao, hostDetailsDao, outOfBandManagementService); + // still leaves a host that can be recognised and powered back on; and the agent is detached without + // investigation before the power is cut so the link drop is not reported as a host-down failure. + InOrder inOrder = Mockito.inOrder(hostDao, hostDetailsDao, agentManager, outOfBandManagementService); inOrder.verify(hostDao).updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, h); inOrder.verify(hostDetailsDao).persist(Mockito.any(DetailVO.class)); + inOrder.verify(agentManager).disconnectWithoutInvestigation(30L, Status.Event.ShutdownRequested); inOrder.verify(outOfBandManagementService).executePowerOperation(Mockito.eq(h), Mockito.eq(OutOfBandManagement.PowerOperation.OFF), Mockito.any()); } From 41a200432044797bc466a928442f96a3eba6c6c9 Mon Sep 17 00:00:00 2001 From: Ramgopal Nagaboina Date: Wed, 7 Oct 2026 18:37:16 -0400 Subject: [PATCH 5/7] drs: do not roll back a failed power-off that may have taken effect powerOffHost removed the wake marker and re-enabled the host whenever the power-off command threw. A failure response does not prove the host is still running: a chassis power-off can take effect and still report a timeout or error, which left the host powered off, enabled and unmarked, so DRS never woke it and the scheduler kept targeting a dead host. Leave the host disabled and marked on failure; the next poll re-enables and unmarks it if it is still up, or treats it as a DRS-powered-off host if it is actually down. --- .../cluster/ClusterDrsServiceImpl.java | 10 ++++++---- .../ClusterDrsPowerOrchestrationTest.java | 16 ++++++---------- 2 files changed, 12 insertions(+), 14 deletions(-) diff --git a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java index 3905f528f42c..82e4f7cceb47 100644 --- a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java +++ b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java @@ -1370,10 +1370,12 @@ protected void powerOffHost(HostVO host, ClusterVO cluster) { agentManager.disconnectWithoutInvestigation(host.getId(), Status.Event.ShutdownRequested); outOfBandManagementService.executePowerOperation(host, OutOfBandManagement.PowerOperation.OFF, null); } catch (Exception e) { - // Power-off failed: undo the marker and the disable so the host stays in service. - hostDetailsDao.remove(marker.getId()); - hostDao.updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, host); - throw e; + // A failure response does not prove the host is still running: a chassis power-off can take effect and + // still report a timeout or error. Rather than roll back (which would strand a host that did power off + // as enabled and unmarked), leave it disabled and marked. The next poll reclaims it (re-enable, clear + // marker) if it is still up, or treats it as a DRS-powered-off host if it is actually down. + logger.warn("DRS power management: power-off command for host [{}] failed; leaving it disabled and marked for reconciliation on the next poll.", + host.getId(), e); } drainingVmCountByHost.remove(host.getId()); } diff --git a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java index 07bae34073ba..16418df8e3dd 100644 --- a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java +++ b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java @@ -180,22 +180,18 @@ public void powerOffAbortsWhenTheHostCannotBeDisabled() { } @Test - public void powerOffRollsBackMarkerAndDisableWhenThePowerOffFails() { + public void powerOffLeavesHostDisabledAndMarkedWhenThePowerOffFails() { HostVO h = host(32L); Mockito.when(hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, h)).thenReturn(true); Mockito.doThrow(new RuntimeException("oobm down")).when(outOfBandManagementService) .executePowerOperation(Mockito.eq(h), Mockito.eq(OutOfBandManagement.PowerOperation.OFF), Mockito.any()); - try { - drs.powerOffHost(h, cluster(1L)); - Assert.fail("expected the power-off failure to propagate"); - } catch (RuntimeException expected) { - // expected - } + drs.powerOffHost(h, cluster(1L)); - // on failure the marker is removed and the host is re-enabled, so it is not left disabled-and-marked. - Mockito.verify(hostDetailsDao).remove(Mockito.anyLong()); - Mockito.verify(hostDao).updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, h); + // a failed power-off must not roll back: the host stays disabled and marked so the next poll reconciles it, + // because the failure does not prove the chassis is still on. No marker removal, no re-enable. + Mockito.verify(hostDetailsDao, Mockito.never()).remove(Mockito.anyLong()); + Mockito.verify(hostDao, Mockito.never()).updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, h); } @Test From a15098134bac863c9e2b32604937d4631774e7b9 Mon Sep 17 00:00:00 2001 From: Ramgopal Nagaboina Date: Wed, 7 Oct 2026 19:03:57 -0400 Subject: [PATCH 6/7] drs: disable the drain candidate and treat draining as a first-class state A host being drained stayed Enabled, so the allocator could keep placing new VMs on it (it is the least-loaded host, which the allocator prefers), fighting the drain and tripping the no-progress abandon. Draining hosts were also only reconciled while in upHosts, so one that went Down or was disabled mid-drain leaked its marker and progress entry. Disable the candidate when a drain starts so the scheduler leaves it alone, and re-enable it if the drain is abandoned. Classify draining hosts in their own bucket: continue the drain while the host is up (regardless of whether evacuation is still enabled, so toggling it off does not strand a host) and abandon it if the host is no longer up. Count a draining host's load in cluster utilization, power off a host that has finished draining directly (powerOffHost now tolerates an already-disabled host), and only start a new drain when no host is empty to power off. --- .../cluster/ClusterDrsServiceImpl.java | 85 +++++++++++-------- .../ClusterDrsPowerOrchestrationTest.java | 9 +- 2 files changed, 56 insertions(+), 38 deletions(-) diff --git a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java index 82e4f7cceb47..0a6d818eee78 100644 --- a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java +++ b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java @@ -918,7 +918,19 @@ protected void managePowerForCluster(ClusterVO cluster) { List routingHosts = hostDao.findByClusterId(cluster.getId(), Host.Type.Routing); List upHosts = new ArrayList<>(); List poweredOffByDrs = new ArrayList<>(); + List drainingHosts = new ArrayList<>(); for (HostVO host : routingHosts) { + if (isDrainingByDrs(host)) { + if (host.getStatus() == Status.Up) { + // still reachable: continue (or finish) the drain below. + drainingHosts.add(host); + } else { + // went down or away mid-drain: DRS did not power it off, so stop tracking it and release the + // marker rather than leaking it and blocking a later drain. + abandonDrain(host, "host is no longer up"); + } + continue; + } if (host.getStatus() == Status.Up) { if (isPoweredOffByDrs(host)) { // Our host is back in service: either a wake completed, or a power-off we issued never took @@ -940,11 +952,15 @@ protected void managePowerForCluster(ClusterVO cluster) { poweredOffByDrs.add(host); } } - if (upHosts.isEmpty()) { + if (upHosts.isEmpty() && poweredOffByDrs.isEmpty() && drainingHosts.isEmpty()) { return; } - Map> capacityMap = getHostCapacityMap(upHosts, useCpu); + // Capacity reflects every host still carrying load: the schedulable up hosts plus any host being drained + // (its VMs still run until it empties), so utilization is not understated while a drain is in progress. + List loadHosts = new ArrayList<>(upHosts); + loadHosts.addAll(drainingHosts); + Map> capacityMap = getHostCapacityMap(loadHosts, useCpu); double clusterUsed = 0d; double clusterTotal = 0d; for (Ternary capacity : capacityMap.values()) { @@ -989,10 +1005,15 @@ && clusterCanReleaseHost(effectiveUsed, clusterTotal, capacity.third(), upHosts. } } - // nothing empty to power off; if the operator opted in, drain a releasable host so a later poll can - // power it off once empty. - if (Boolean.TRUE.equals(ClusterDrsPowerManagementEvacuate.valueIn(cluster.getId()))) { - evacuateReleasableHost(cluster, upHosts, capacityMap, effectiveUsed, clusterTotal, lowThreshold, highThreshold, useCpu); + // Continue a drain already in progress even if evacuation was since turned off, so a host is never stranded + // mid-drain; only START a new drain when the operator has opted in and no host is empty to power off. + if (!drainingHosts.isEmpty()) { + drainHostBatch(cluster, drainingHosts.get(0), upHosts, capacityMap, highThreshold, useCpu); + } else if (Boolean.TRUE.equals(ClusterDrsPowerManagementEvacuate.valueIn(cluster.getId()))) { + HostVO candidate = selectDrainCandidate(upHosts, capacityMap, effectiveUsed, clusterTotal, lowThreshold, highThreshold); + if (candidate != null) { + drainHostBatch(cluster, candidate, upHosts, capacityMap, highThreshold, useCpu); + } } } @@ -1024,34 +1045,15 @@ protected void setPowerMarker(long hostId, String value) { protected void abandonDrain(HostVO host, String reason) { logger.info("DRS power management: abandoning drain of host [{}]: {}", host.getId(), reason); + // the host was disabled when the drain started so the scheduler would leave it alone while it emptied; + // put it back in service now that the drain is being given up. + if (host.getResourceState() == ResourceState.Disabled) { + hostDao.updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, host); + } clearStalePowerMarker(host); drainingVmCountByHost.remove(host.getId()); } - /** - * Drives the eviction side of power management when no host is empty yet. Continues a drain already in progress - * before starting another, otherwise starts draining the least-loaded releasable host. A drain that stops making - * progress across polls (its migrations keep being rejected for affinity, host tags or storage) is abandoned - * instead of re-submitted forever, so a host is never left indefinitely half-drained. - */ - protected void evacuateReleasableHost(ClusterVO cluster, List upHosts, Map> capacityMap, - double effectiveUsed, double clusterTotal, float lowThreshold, float highThreshold, boolean useCpu) { - HostVO candidate = null; - for (HostVO host : upHosts) { - if (isDrainingByDrs(host)) { - candidate = host; - break; - } - } - if (candidate == null) { - candidate = selectDrainCandidate(upHosts, capacityMap, effectiveUsed, clusterTotal, lowThreshold, highThreshold); - } - if (candidate == null) { - return; - } - drainHostBatch(cluster, candidate, upHosts, capacityMap, highThreshold, useCpu); - } - /** * The least-loaded host the cluster can give up: powerable, holding only running user VMs, and such that the * remaining hosts can still carry the load. Null if no host qualifies. @@ -1084,8 +1086,9 @@ protected void drainHostBatch(ClusterVO cluster, HostVO candidate, List Map> capacityMap, float highThreshold, boolean useCpu) { List vms = vmInstanceDao.listByHostId(candidate.getId()); if (vms == null || vms.isEmpty()) { - // fully drained; the empty-host power-off path takes it from here. - abandonDrain(candidate, "host is empty"); + // fully drained: power it off now. It was disabled when the drain started, so the empty-host power-off + // loop (which only scans enabled up hosts) does not see it; powerOffHost handles the already-disabled case. + powerOffHost(candidate, cluster); return; } if (hasMigratingVm(candidate.getId())) { @@ -1145,6 +1148,14 @@ protected void drainHostBatch(ClusterVO cluster, HostVO candidate, List if (batch.isEmpty()) { return; } + // Starting a new drain: disable the host first so the allocator stops placing new VMs on it (it is the + // least-loaded host, which the allocator would otherwise prefer, and that would fight the drain). A drain + // already in progress is left as is. If the host cannot be disabled, do not start. + if (previous == null && candidate.getResourceState() == ResourceState.Enabled + && !hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, candidate)) { + logger.warn("DRS power management: could not disable host [{}] to begin draining it; skipping.", candidate.getId()); + return; + } logger.info("DRS power management: cluster [{}] is under-utilized; draining host [{}] ({} VMs left, {} this poll) to power it off.", cluster.getId(), candidate.getId(), plan.size(), batch.size()); // One start event for the batch. Each migration job carries it as its start event id and completes it, as in @@ -1347,10 +1358,12 @@ protected boolean hasInFlightDrsPlan(long clusterId) { protected void powerOffHost(HostVO host, ClusterVO cluster) { logger.info("DRS power management: cluster [{}] is under-utilized; disabling and powering off empty host [{}].", cluster.getId(), host.getId()); - // Disable first so CloudStack stops scheduling to the host and does not treat the imminent agent - // disconnect as a failure (host monitor / HA). Abort if the transition did not take effect, so the - // host is never powered off while CloudStack still believes it is schedulable. - if (!hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, host)) { + // Disable first so CloudStack stops scheduling to the host and does not treat the imminent agent disconnect + // as a failure (host monitor / HA). A host that was being drained is already Disabled, so only transition an + // Enabled host, and abort only when that transition does not take effect, so the host is never powered off + // while CloudStack still believes it is schedulable. + if (host.getResourceState() == ResourceState.Enabled + && !hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, host)) { logger.warn("DRS power management: could not disable host [{}]; skipping power-off.", host.getId()); return; } diff --git a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java index 16418df8e3dd..90ece7bd5e47 100644 --- a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java +++ b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java @@ -83,6 +83,7 @@ private VMInstanceVO vm(long id, VirtualMachine.Type type, VirtualMachine.State private HostVO host(long id) { HostVO host = Mockito.mock(HostVO.class); Mockito.lenient().when(host.getId()).thenReturn(id); + Mockito.lenient().when(host.getResourceState()).thenReturn(ResourceState.Enabled); return host; } @@ -277,8 +278,10 @@ public void drainBatchAbandonsWhenNoProgressSinceLastBatch() { } @Test - public void drainBatchAbandonsWhenTheHostIsAlreadyEmpty() { + public void drainBatchPowersOffAnEmptiedDrainingHost() { HostVO candidate = host(44L); + // a drained host is already disabled, so powerOffHost must proceed without re-disabling it. + Mockito.when(candidate.getResourceState()).thenReturn(ResourceState.Disabled); Mockito.when(vmInstanceDao.listByHostId(44L)).thenReturn(Collections.emptyList()); Mockito.when(hostDetailsDao.findDetail(44L, "drs.power.state")) .thenReturn(new DetailVO(44L, "drs.power.state", "draining")); @@ -286,7 +289,9 @@ public void drainBatchAbandonsWhenTheHostIsAlreadyEmpty() { drs.drainHostBatch(cluster(1L), candidate, Collections.emptyList(), Collections.emptyMap(), 0.75f, true); - Mockito.verify(hostDetailsDao).remove(Mockito.anyLong()); + // the emptied host is powered off (the empty-host loop never sees it because it is disabled), the draining + // marker is replaced by the off marker, and the progress entry is cleared. + Mockito.verify(outOfBandManagementService).executePowerOperation(Mockito.eq(candidate), Mockito.eq(OutOfBandManagement.PowerOperation.OFF), Mockito.any()); Assert.assertFalse(drs.drainingVmCountByHost.containsKey(44L)); } } From 974cdec522fab1327bf7621d4344284070625a2d Mon Sep 17 00:00:00 2001 From: Ramgopal Nagaboina Date: Thu, 8 Oct 2026 14:33:51 -0400 Subject: [PATCH 7/7] drs: mark before disabling a drain candidate and re-check guards on completion - Persist the draining marker before disabling the host when a drain starts, mirroring the marker-before-action ordering powerOffHost uses. If job submission or the management server fails in the window after the disable, the host is still recognised as draining and reconciled on the next poll instead of stranded disabled with no marker. Undo the marker if the disable itself fails. - When a drain completes, power the emptied host off only if it is still out-of-band manageable and the cluster can still release it; otherwise return it to service, so a host is not flapped off when management went away or load rose during the drain. --- .../cluster/ClusterDrsServiceImpl.java | 43 ++++++++++++------ .../ClusterDrsPowerOrchestrationTest.java | 44 +++++++++++++++---- 2 files changed, 65 insertions(+), 22 deletions(-) diff --git a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java index 0a6d818eee78..1535a5a5e7b6 100644 --- a/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java +++ b/server/src/main/java/org/apache/cloudstack/cluster/ClusterDrsServiceImpl.java @@ -1008,11 +1008,11 @@ && clusterCanReleaseHost(effectiveUsed, clusterTotal, capacity.third(), upHosts. // Continue a drain already in progress even if evacuation was since turned off, so a host is never stranded // mid-drain; only START a new drain when the operator has opted in and no host is empty to power off. if (!drainingHosts.isEmpty()) { - drainHostBatch(cluster, drainingHosts.get(0), upHosts, capacityMap, highThreshold, useCpu); + drainHostBatch(cluster, drainingHosts.get(0), upHosts, capacityMap, effectiveUsed, clusterTotal, lowThreshold, highThreshold, useCpu); } else if (Boolean.TRUE.equals(ClusterDrsPowerManagementEvacuate.valueIn(cluster.getId()))) { HostVO candidate = selectDrainCandidate(upHosts, capacityMap, effectiveUsed, clusterTotal, lowThreshold, highThreshold); if (candidate != null) { - drainHostBatch(cluster, candidate, upHosts, capacityMap, highThreshold, useCpu); + drainHostBatch(cluster, candidate, upHosts, capacityMap, effectiveUsed, clusterTotal, lowThreshold, highThreshold, useCpu); } } } @@ -1083,12 +1083,21 @@ protected HostVO selectDrainCandidate(List upHosts, Map upHosts, - Map> capacityMap, float highThreshold, boolean useCpu) { + Map> capacityMap, double effectiveUsed, double clusterTotal, + float lowThreshold, float highThreshold, boolean useCpu) { List vms = vmInstanceDao.listByHostId(candidate.getId()); if (vms == null || vms.isEmpty()) { - // fully drained: power it off now. It was disabled when the drain started, so the empty-host power-off - // loop (which only scans enabled up hosts) does not see it; powerOffHost handles the already-disabled case. - powerOffHost(candidate, cluster); + // Fully drained. It was disabled when the drain started, so the empty-host power-off loop (which only + // scans enabled up hosts) does not see it; power it off here, applying the same guards that loop does. + // If it can no longer be powered off (out-of-band management gone, or the cluster is no longer + // under-utilized because load rose during the drain), put it back in service instead of flapping it. + Ternary cap = capacityMap.get(candidate.getId()); + if (isPowerManageable(candidate) && cap != null + && clusterCanReleaseHost(effectiveUsed, clusterTotal, cap.third(), upHosts.size() + 1, lowThreshold, highThreshold)) { + powerOffHost(candidate, cluster); + } else { + abandonDrain(candidate, "host is drained but can no longer be powered off; returning it to service"); + } return; } if (hasMigratingVm(candidate.getId())) { @@ -1148,13 +1157,20 @@ protected void drainHostBatch(ClusterVO cluster, HostVO candidate, List if (batch.isEmpty()) { return; } - // Starting a new drain: disable the host first so the allocator stops placing new VMs on it (it is the - // least-loaded host, which the allocator would otherwise prefer, and that would fight the drain). A drain - // already in progress is left as is. If the host cannot be disabled, do not start. - if (previous == null && candidate.getResourceState() == ResourceState.Enabled - && !hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, candidate)) { - logger.warn("DRS power management: could not disable host [{}] to begin draining it; skipping.", candidate.getId()); - return; + // Starting a new drain: record the durable draining marker BEFORE disabling the host, so a crash or an + // exception in the window between the two never leaves the host disabled with no marker (which would strand + // it out of the scheduling pool with no self-heal). Then disable it so the allocator stops placing new VMs + // on it (it is the least-loaded host, which the allocator would otherwise prefer, and that would fight the + // drain). If the host cannot be disabled, undo the marker and do not start. A drain already in progress + // (previous != null) already has its marker and is already disabled. + if (previous == null) { + setPowerMarker(candidate.getId(), DRS_POWER_STATE_DRAINING); + if (candidate.getResourceState() == ResourceState.Enabled + && !hostDao.updateResourceState(ResourceState.Enabled, ResourceState.Event.Disable, ResourceState.Disabled, candidate)) { + logger.warn("DRS power management: could not disable host [{}] to begin draining it; skipping.", candidate.getId()); + clearStalePowerMarker(candidate); + return; + } } logger.info("DRS power management: cluster [{}] is under-utilized; draining host [{}] ({} VMs left, {} this poll) to power it off.", cluster.getId(), candidate.getId(), plan.size(), batch.size()); @@ -1167,7 +1183,6 @@ protected void drainHostBatch(ClusterVO cluster, HostVO candidate, List for (Map.Entry migration : batch.entrySet()) { createMigrateVMAsyncJob(migration.getKey(), migration.getValue(), eventId); } - setPowerMarker(candidate.getId(), DRS_POWER_STATE_DRAINING); drainingVmCountByHost.put(candidate.getId(), current); } diff --git a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java index 90ece7bd5e47..12705705bff6 100644 --- a/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java +++ b/server/src/test/java/org/apache/cloudstack/cluster/ClusterDrsPowerOrchestrationTest.java @@ -23,6 +23,7 @@ import org.apache.cloudstack.outofbandmanagement.OutOfBandManagementService; import org.apache.cloudstack.cluster.dao.ClusterDrsPlanDao; import com.cloud.agent.AgentManager; +import com.cloud.utils.Ternary; import com.cloud.host.Status; import org.junit.Assert; import org.junit.Test; @@ -253,7 +254,7 @@ public void drainBatchWaitsWhileAMigrationIsInFlight() { Mockito.when(vmInstanceDao.listByHostId(42L)).thenReturn(Collections.singletonList(migrating)); drs.drainingVmCountByHost.put(42L, 3); - drs.drainHostBatch(cluster(1L), candidate, Collections.emptyList(), Collections.emptyMap(), 0.75f, true); + drs.drainHostBatch(cluster(1L), candidate, Collections.emptyList(), Collections.emptyMap(), 0d, 100d, 0.30f, 0.75f, true); // a batch is still running: do not touch the marker or the recorded progress, just wait. Mockito.verify(hostDetailsDao, Mockito.never()).remove(Mockito.anyLong()); @@ -270,7 +271,7 @@ public void drainBatchAbandonsWhenNoProgressSinceLastBatch() { .thenReturn(new DetailVO(43L, "drs.power.state", "draining")); drs.drainingVmCountByHost.put(43L, 2); // same count as now: the last batch moved nothing - drs.drainHostBatch(cluster(1L), candidate, Collections.emptyList(), Collections.emptyMap(), 0.75f, true); + drs.drainHostBatch(cluster(1L), candidate, Collections.emptyList(), Collections.emptyMap(), 0d, 100d, 0.30f, 0.75f, true); // no progress: the drain is abandoned (marker cleared, progress forgotten) instead of re-submitted forever. Mockito.verify(hostDetailsDao).remove(Mockito.anyLong()); @@ -278,20 +279,47 @@ public void drainBatchAbandonsWhenNoProgressSinceLastBatch() { } @Test - public void drainBatchPowersOffAnEmptiedDrainingHost() { + public void drainBatchPowersOffAnEmptiedDrainingHostWhenStillReleasable() { HostVO candidate = host(44L); + HostVO other = host(45L); // a drained host is already disabled, so powerOffHost must proceed without re-disabling it. Mockito.when(candidate.getResourceState()).thenReturn(ResourceState.Disabled); Mockito.when(vmInstanceDao.listByHostId(44L)).thenReturn(Collections.emptyList()); - Mockito.when(hostDetailsDao.findDetail(44L, "drs.power.state")) - .thenReturn(new DetailVO(44L, "drs.power.state", "draining")); + Mockito.when(outOfBandManagementService.isOutOfBandManagementEnabled(candidate)).thenReturn(true); + java.util.Map> capacityMap = new java.util.HashMap<>(); + capacityMap.put(44L, new Ternary<>(0L, 0L, 100L)); + capacityMap.put(45L, new Ternary<>(10L, 0L, 100L)); drs.drainingVmCountByHost.put(44L, 1); - drs.drainHostBatch(cluster(1L), candidate, Collections.emptyList(), Collections.emptyMap(), 0.75f, true); + // cluster at 10/200 = 5% (below low 0.30), remaining after release 10/100 = 10% (below high 0.75): releasable. + drs.drainHostBatch(cluster(1L), candidate, Collections.singletonList(other), capacityMap, 10d, 200d, 0.30f, 0.75f, true); - // the emptied host is powered off (the empty-host loop never sees it because it is disabled), the draining - // marker is replaced by the off marker, and the progress entry is cleared. + // the emptied host is powered off (the empty-host loop never sees it because it is disabled) and its + // progress entry is cleared. Mockito.verify(outOfBandManagementService).executePowerOperation(Mockito.eq(candidate), Mockito.eq(OutOfBandManagement.PowerOperation.OFF), Mockito.any()); Assert.assertFalse(drs.drainingVmCountByHost.containsKey(44L)); } + + @Test + public void drainBatchReturnsAnEmptiedHostToServiceWhenNoLongerPowerManageable() { + HostVO candidate = host(46L); + HostVO other = host(47L); + Mockito.when(candidate.getResourceState()).thenReturn(ResourceState.Disabled); + Mockito.when(vmInstanceDao.listByHostId(46L)).thenReturn(Collections.emptyList()); + // out-of-band management is no longer available for the drained host. + Mockito.when(outOfBandManagementService.isOutOfBandManagementEnabled(candidate)).thenReturn(false); + Mockito.when(hostDetailsDao.findDetail(46L, "drs.power.state")) + .thenReturn(new DetailVO(46L, "drs.power.state", "draining")); + java.util.Map> capacityMap = new java.util.HashMap<>(); + capacityMap.put(46L, new Ternary<>(0L, 0L, 100L)); + capacityMap.put(47L, new Ternary<>(10L, 0L, 100L)); + drs.drainingVmCountByHost.put(46L, 1); + + drs.drainHostBatch(cluster(1L), candidate, Collections.singletonList(other), capacityMap, 10d, 200d, 0.30f, 0.75f, true); + + // it is not powered off; instead it is re-enabled and unmarked so it goes back into the scheduling pool. + Mockito.verify(outOfBandManagementService, Mockito.never()).executePowerOperation(Mockito.any(), Mockito.any(), Mockito.any()); + Mockito.verify(hostDao).updateResourceState(ResourceState.Disabled, ResourceState.Event.Enable, ResourceState.Enabled, candidate); + Assert.assertFalse(drs.drainingVmCountByHost.containsKey(46L)); + } }