Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions plugins/storage/volume/linstor/CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,17 @@ All notable changes to Linstor CloudStack plugin will be documented in this file
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).

## [2026-10-07]

### Fixed

- Live migration with storage into Linstor failed with "cannot precreate storage for disk type 'block'"
if the auto-placement didn't put a copy of the new resource on the target host; the resource is
now made available on the target host before the migration starts.
- Live migration with storage into Linstor failed on cgroup v2 hosts with
"shares '<n>' must be in range [1, 10000]" for VMs with more than 10000 cpus * MHz, and smaller
VMs ended up with an unscaled CPU weight; the CPU shares calculated by the target host are now used.

## [2026-06-24]

### Fixed
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
import com.linbit.linstor.api.model.ApiCallRcList;
import com.linbit.linstor.api.model.ResourceDefinition;
import com.linbit.linstor.api.model.ResourceDefinitionModify;
import com.linbit.linstor.api.model.ResourceMakeAvailable;

import javax.inject.Inject;

Expand All @@ -35,8 +36,11 @@

import com.cloud.agent.AgentManager;
import com.cloud.agent.api.Answer;
import com.cloud.agent.api.CheckVirtualMachineAnswer;
import com.cloud.agent.api.CheckVirtualMachineCommand;
import com.cloud.agent.api.MigrateAnswer;
import com.cloud.agent.api.MigrateCommand;
import com.cloud.agent.api.PrepareForMigrationAnswer;
import com.cloud.agent.api.PrepareForMigrationCommand;
import com.cloud.agent.api.to.DataObjectType;
import com.cloud.agent.api.to.VirtualMachineTO;
Expand All @@ -54,6 +58,7 @@
import com.cloud.storage.dao.VolumeDao;
import com.cloud.utils.exception.CloudRuntimeException;
import com.cloud.vm.VMInstanceVO;
import com.cloud.vm.VirtualMachine;
import com.cloud.vm.dao.VMInstanceDao;
import org.apache.cloudstack.engine.subsystem.api.storage.CopyCommandResult;
import org.apache.cloudstack.engine.subsystem.api.storage.DataMotionStrategy;
Expand All @@ -74,7 +79,6 @@
import org.apache.cloudstack.storage.datastore.util.LinstorUtil;
import org.apache.commons.collections.CollectionUtils;
import org.apache.commons.collections.MapUtils;
import org.apache.commons.lang3.ObjectUtils;
import org.apache.logging.log4j.LogManager;
import org.apache.logging.log4j.Logger;
import org.springframework.stereotype.Component;
Expand Down Expand Up @@ -169,11 +173,51 @@ private VolumeVO createNewVolumeVO(Volume volume, StoragePoolVO storagePoolVO) {
return _volumeDao.persist(newVol);
}

private DevelopersApi getLinstorAPI(StoragePoolVO storagePool) {
return LinstorUtil.getLinstorAPI(storagePool.getHostAddress(),
LinstorConfigurationManager.ApiToken.valueIn(storagePool.getId()),
Boolean.TRUE.equals(LinstorConfigurationManager.InsecureSsl.valueIn(storagePool.getId())));
}

/**
* Makes the new resource available on the migration target host and returns its device path there.
*
* <p>The resource is spawned by LINSTOR's auto-placement, which does not know about the target host.
* libvirt can only precreate file disks on the target, so if the block device is missing there the
* migration fails with "cannot precreate storage for disk type 'block'". The source disk is not on
* LINSTOR, so a plain (diskless for DRBD) make-available is enough, no dual-primary needed.</p>
*/
private String makeAvailableOnHost(StoragePoolVO storagePool, String rscName, Host host) {
DevelopersApi api = getLinstorAPI(storagePool);
try {
logger.info("Linstor: make resource {} available on migration target {}", rscName, host.getName());
ApiCallRcList answers = api.resourceMakeAvailableOnNode(rscName, host.getName(), new ResourceMakeAvailable());
LinstorUtil.checkLinstorAnswersThrow(answers);
return LinstorUtil.getDevicePath(api, rscName);
} catch (ApiException apiEx) {
logger.error("Linstor: ApiEx - {}", apiEx.getMessage());
throw new CloudRuntimeException(apiEx.getBestMessage(), apiEx);
}
}

/**
* A paused domain (incoming migration still in progress) is reported as PowerUnknown, so this
* is only true once the VM really runs on the host.
*/
private boolean isVmRunningOnHost(VirtualMachineTO vmTO, Host host) {
try {
Answer answer = _agentManager.send(host.getId(), new CheckVirtualMachineCommand(vmTO.getName()));
return answer instanceof CheckVirtualMachineAnswer && answer.getResult() &&
VirtualMachine.PowerState.PowerOn.equals(((CheckVirtualMachineAnswer) answer).getState());
} catch (AgentUnavailableException | OperationTimedoutException e) {
logger.warn("Unable to check if VM [{}] is running on host [{}]", vmTO, host, e);
return false;
}
}

private void removeExactSizeProperty(VolumeInfo volumeInfo) {
StoragePoolVO destStoragePool = _storagePool.findById(volumeInfo.getDataStore().getId());
DevelopersApi api = LinstorUtil.getLinstorAPI(destStoragePool.getHostAddress(),
LinstorConfigurationManager.ApiToken.valueIn(destStoragePool.getId()),
Boolean.TRUE.equals(LinstorConfigurationManager.InsecureSsl.valueIn(destStoragePool.getId())));
DevelopersApi api = getLinstorAPI(destStoragePool);

ResourceDefinitionModify rdm = new ResourceDefinitionModify();
rdm.setDeleteProps(Collections.singletonList(LinstorUtil.LIN_PROP_DRBDOPT_EXACT_SIZE));
Expand Down Expand Up @@ -266,10 +310,10 @@ private void handlePostMigration(boolean success, Map<VolumeInfo, VolumeInfo> sr
_volumeService.expungeVolumeAsync(destVolumeInfo);

if (destroyFuture.get().isFailed()) {
logger.debug("Failed to clean up dest volume on storage");
logger.warn("Failed to clean up dest volume {} on storage", destVolumeInfo);
}
} catch (Exception e) {
logger.debug("Failed to clean up dest volume on storage", e);
logger.warn("Failed to clean up dest volume {} on storage", destVolumeInfo, e);
}
}
}
Expand All @@ -293,9 +337,7 @@ private void handlePostMigration(boolean success, Map<VolumeInfo, VolumeInfo> sr
private boolean needsExactSizeProp(VolumeInfo srcVolumeInfo) {
StoragePoolVO srcStoragePool = _storagePool.findById(srcVolumeInfo.getDataStore().getId());
if (srcStoragePool.getPoolType() == Storage.StoragePoolType.Linstor) {
DevelopersApi api = LinstorUtil.getLinstorAPI(srcStoragePool.getHostAddress(),
LinstorConfigurationManager.ApiToken.valueIn(srcStoragePool.getId()),
Boolean.TRUE.equals(LinstorConfigurationManager.InsecureSsl.valueIn(srcStoragePool.getId())));
DevelopersApi api = getLinstorAPI(srcStoragePool);

String rscName = LinstorUtil.RSC_PREFIX + srcVolumeInfo.getPath();
try {
Expand Down Expand Up @@ -335,6 +377,7 @@ public void copyAsync(Map<VolumeInfo, DataStore> volumeDataStoreMap, VirtualMach

Map<String, MigrateCommand.MigrateDiskInfo> migrateStorage = new HashMap<>();
Map<VolumeInfo, VolumeInfo> srcVolumeInfoToDestVolumeInfo = new HashMap<>();
boolean migrateCommandSent = false;

try {
for (Map.Entry<VolumeInfo, DataStore> entry : volumeDataStoreMap.entrySet()) {
Expand All @@ -351,20 +394,25 @@ public void copyAsync(Map<VolumeInfo, DataStore> volumeDataStoreMap, VirtualMach
VolumeVO destVolume = createNewVolumeVO(srcVolume, destStoragePool);

VolumeInfo destVolumeInfo = _volumeDataFactory.getVolume(destVolume.getId(), destDataStore);
// registered right away, so a failure from here on cleans the new volume up again
srcVolumeInfoToDestVolumeInfo.put(srcVolumeInfo, destVolumeInfo);

destVolumeInfo.processEvent(ObjectInDataStoreStateMachine.Event.MigrationCopyRequested);
destVolumeInfo.processEvent(ObjectInDataStoreStateMachine.Event.MigrationCopySucceeded);
destVolumeInfo.processEvent(ObjectInDataStoreStateMachine.Event.MigrationRequested);

boolean exactSize = needsExactSizeProp(srcVolumeInfo);

String devPath = LinstorUtil.createResource(
destVolumeInfo, destStoragePool, _storagePoolDao, exactSize);
LinstorUtil.createResource(destVolumeInfo, destStoragePool, _storagePoolDao, exactSize);

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

if creating the linstor resource fails here, does the new volume row still get left behind? it only goes into the cleanup list a few lines later

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Also thanks, it is added to the cleanup list now after the DB object was created

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

that covers it, thanks


_volumeDao.update(destVolume.getId(), destVolume);
destVolume = _volumeDao.findById(destVolume.getId());

destVolumeInfo = _volumeDataFactory.getVolume(destVolume.getId(), destDataStore);
srcVolumeInfoToDestVolumeInfo.put(srcVolumeInfo, destVolumeInfo);

String devPath = makeAvailableOnHost(
destStoragePool, LinstorUtil.RSC_PREFIX + destVolumeInfo.getUuid(), destHost);

MigrateCommand.MigrateDiskInfo migrateDiskInfo = new MigrateCommand.MigrateDiskInfo(
srcVolumeInfo.getPath(),
Expand All @@ -375,13 +423,12 @@ public void copyAsync(Map<VolumeInfo, DataStore> volumeDataStoreMap, VirtualMach
migrateDiskInfoList.add(migrateDiskInfo);

migrateStorage.put(srcVolumeInfo.getPath(), migrateDiskInfo);

srcVolumeInfoToDestVolumeInfo.put(srcVolumeInfo, destVolumeInfo);
}

PrepareForMigrationCommand pfmc = new PrepareForMigrationCommand(vmTO);
Answer pfma;
try {
Answer pfma = _agentManager.send(destHost.getId(), pfmc);
pfma = _agentManager.send(destHost.getId(), pfmc);

if (pfma == null || !pfma.getResult()) {
String details = pfma != null ? pfma.getDetails() : "null answer returned";
Expand All @@ -403,33 +450,60 @@ public void copyAsync(Map<VolumeInfo, DataStore> volumeDataStoreMap, VirtualMach
migrateCommand.setWait(StorageManager.KvmStorageOnlineMigrationWait.value());
migrateCommand.setMigrateStorage(migrateStorage);
migrateCommand.setMigrateStorageManaged(true);
migrateCommand.setNewVmCpuShares(
vmTO.getCpus() * ObjectUtils.defaultIfNull(vmTO.getMinSpeed(), vmTO.getSpeed()));
// the target host scales the shares to its cgroup version (v2 only accepts 1-10000),
// the raw cpus * speed value is only valid on cgroup v1
Integer newVmCpuShares = ((PrepareForMigrationAnswer) pfma).getNewVmCpuShares();
if (newVmCpuShares != null) {
migrateCommand.setNewVmCpuShares(newVmCpuShares);
}
migrateCommand.setMigrateDiskInfoList(migrateDiskInfoList);

boolean kvmAutoConvergence = StorageManager.KvmAutoConvergence.value();
migrateCommand.setAutoConvergence(kvmAutoConvergence);

MigrateAnswer migrateAnswer = (MigrateAnswer) _agentManager.send(srcHost.getId(), migrateCommand);
boolean success = migrateAnswer != null && migrateAnswer.getResult();
// once the migrate command is sent, the VM may end up running on the new volumes,
// so they are only removed if the source host reports the migration as failed
migrateCommandSent = true;
MigrateAnswer migrateAnswer = null;
boolean success;
try {
migrateAnswer = (MigrateAnswer) _agentManager.send(srcHost.getId(), migrateCommand);
success = migrateAnswer != null && migrateAnswer.getResult();
} catch (OperationTimedoutException ex) {
// no answer from the source host, but the migration may still have finished
if (!isVmRunningOnHost(vmTO, destHost)) {
throw ex;
}
logger.info("VM [{}] is running on the destination host [{}], migration was successful", vmTO, destHost);
success = true;
}

handlePostMigration(success, srcVolumeInfoToDestVolumeInfo, vmTO, destHost);

if (migrateAnswer == null) {
throw new CloudRuntimeException("Unable to get an answer to the migrate command");
}
if (!success) {
if (migrateAnswer == null) {
throw new CloudRuntimeException("Unable to get an answer to the migrate command");
}

if (!migrateAnswer.getResult()) {
errMsg = migrateAnswer.getDetails();

throw new CloudRuntimeException(errMsg);
}
} catch (AgentUnavailableException | OperationTimedoutException | CloudRuntimeException ex) {
errMsg = String.format(
"Copy volume(s) of VM [%s] to storage(s) [%s] and VM to host [%s] failed in LinstorDataMotionStrategy.copyAsync. Error message: [%s].",
vmTO, srcHost, destHost, ex.getMessage());
vmTO, volumeDataStoreMap.values(), destHost, ex.getMessage());
logger.error(errMsg, ex);

if (!migrateCommandSent && !srcVolumeInfoToDestVolumeInfo.isEmpty()) {
// failed before the migration was started: remove the already created destination volumes
try {
handlePostMigration(false, srcVolumeInfoToDestVolumeInfo, vmTO, destHost);
} catch (Exception e) {
logger.warn("Failed to clean up the destination volume(s) of VM [{}]", vmTO, e);
}
}

throw new CloudRuntimeException(errMsg);
} finally {
CopyCmdAnswer copyCmdAnswer = new CopyCmdAnswer(errMsg);
Expand Down
Loading
Loading