From 40a53495e686418d28996d124437cc16d3c882dc Mon Sep 17 00:00:00 2001 From: Claudiu Belu Date: Tue, 18 Aug 2026 20:17:30 +0000 Subject: [PATCH 1/4] integration: Adds coverage for minion pool power cycling Exercises PowerOnMinionMachineTask / PowerOffMinionMachineTask and _set_minion_machine_power_status. These were previously uncovered since existing pool tests never scale a pool beyond its minimum size. The new tests configures pools with the "poweroff" retention strategy and a tiny idle time, then runs two separate transfers concurrently. The second execution allocates a new minion, scaling the pool beyond its minimum. Once both minions go idle, refresh_minion_pool powers the excess one off, then re-running both transfers concurrently reallocates and powers it back on before reuse. _write_systemd will now also enable the systemd unit, so the given service will actually start on reboot (e.g.: minion pool machine reused after being powered off). --- coriolis/tests/integration/base.py | 33 ++++- .../tests/integration/test_minion_pools.py | 138 ++++++++++++++++++ .../tests/integration/test_provider/common.py | 3 + coriolis/tests/test_utils.py | 18 ++- coriolis/utils.py | 5 + 5 files changed, 189 insertions(+), 8 deletions(-) diff --git a/coriolis/tests/integration/base.py b/coriolis/tests/integration/base.py index 34750c7df..6aa278e80 100644 --- a/coriolis/tests/integration/base.py +++ b/coriolis/tests/integration/base.py @@ -172,6 +172,12 @@ def _create_pool( skip_allocation=True, wait_for_allocation=False, platform=constants.PROVIDER_PLATFORM_DESTINATION, + minimum_minions=1, + maximum_minions=1, + minion_max_idle_time=3600, + minion_retention_strategy=( + constants.MINION_POOL_MACHINE_RETENTION_STRATEGY_DELETE + ), ): env_options = ( cls._imp_pool_env @@ -184,12 +190,10 @@ def _create_pool( platform=platform, os_type=constants.OS_TYPE_LINUX, environment_options=env_options, - minimum_minions=1, - maximum_minions=1, - minion_max_idle_time=3600, - minion_retention_strategy=( - constants.MINION_POOL_MACHINE_RETENTION_STRATEGY_DELETE - ), + minimum_minions=minimum_minions, + maximum_minions=maximum_minions, + minion_max_idle_time=minion_max_idle_time, + minion_retention_strategy=minion_retention_strategy, skip_allocation=skip_allocation, ) cls.addClassCleanup(cls._safe_delete_pool, pool.id) @@ -305,6 +309,15 @@ class ReplicaIntegrationTestBase(CoriolisIntegrationTestBase): # source_environment. _EXTRA_SOURCE_ENVIRONMENT = {} + # Overridable params for the pool(s) created when _CREATE_DST_MINION_POOL / + # _CREATE_SRC_MINION_POOL is set. + _POOL_MINIMUM_MINIONS = 1 + _POOL_MAXIMUM_MINIONS = 1 + _POOL_MINION_MAX_IDLE_TIME = 3600 + _POOL_MINION_RETENTION_STRATEGY = ( + constants.MINION_POOL_MACHINE_RETENTION_STRATEGY_DELETE + ) + @classmethod def setUpClass(cls): super().setUpClass() @@ -331,6 +344,10 @@ def setUpClass(cls): "dst-transfer-pool", skip_allocation=False, wait_for_allocation=True, + minimum_minions=cls._POOL_MINIMUM_MINIONS, + maximum_minions=cls._POOL_MAXIMUM_MINIONS, + minion_max_idle_time=cls._POOL_MINION_MAX_IDLE_TIME, + minion_retention_strategy=cls._POOL_MINION_RETENTION_STRATEGY, ) cls._dst_pool_id = pool.id @@ -343,6 +360,10 @@ def setUpClass(cls): skip_allocation=False, wait_for_allocation=True, platform=constants.PROVIDER_PLATFORM_SOURCE, + minimum_minions=cls._POOL_MINIMUM_MINIONS, + maximum_minions=cls._POOL_MAXIMUM_MINIONS, + minion_max_idle_time=cls._POOL_MINION_MAX_IDLE_TIME, + minion_retention_strategy=cls._POOL_MINION_RETENTION_STRATEGY, ) cls._src_pool_id = pool.id diff --git a/coriolis/tests/integration/test_minion_pools.py b/coriolis/tests/integration/test_minion_pools.py index 77ad6b116..7d89fc60f 100644 --- a/coriolis/tests/integration/test_minion_pools.py +++ b/coriolis/tests/integration/test_minion_pools.py @@ -194,3 +194,141 @@ def _create_pool(self, endpoint_id, **kwargs): return super()._create_pool( endpoint_id, platform=constants.PROVIDER_PLATFORM_SOURCE, **kwargs ) + + +class _MinionPoolPowerCycleTestMixin: + """Transfer that reuses pool machines across a power cycle. + + The pool allows up to 2 machines (minimum 1) with a tiny idle time and the + "poweroff" retention strategy. Two separate transfers are executed concurrently; + the second execution finds the pre-existing minimum machine already reserved by the + first and allocates a brand new one instead. Once both go idle, refreshing the pool + powers the excess one off. Re-running both transfers concurrently then reuses both + machines, powering the idled-off one back on before healthchecking and reusing it. + + Subclasses select which side's pool gets exercised by overriding the ``_pool_id`` + property. + """ + + _POOL_MAXIMUM_MINIONS = 2 + _POOL_MINION_MAX_IDLE_TIME = 1 + _POOL_MINION_RETENTION_STRATEGY = ( + constants.MINION_POOL_MACHINE_RETENTION_STRATEGY_POWEROFF + ) + + @property + def _pool_id(self): + raise NotImplementedError + + def setUp(self): + super().setUp() + + # A second transfer, independent from self._transfer (created by + # ReplicaIntegrationTestBase.setUp). Running it concurrently with self._transfer + # forces the pool to allocate a second machine, since the first is already + # reserved by self._transfer's execution. + self._pool_transfer_b = self._create_transfer( + self._src_endpoint.id, + self._dst_endpoint.id, + instances=[self._instance_name], + source_environment=self._transfer._info["source_environment"], + destination_minion_pool_id=self._dst_pool_id, + origin_minion_pool_id=self._src_pool_id, + ) + + def _execute_concurrently_and_wait(self, transfer_ids, timeout=600): + """Start one execution per transfer id before waiting on any.""" + executions = [ + self._client.transfer_executions.create( + transfer_id, shutdown_instances=False + ) + for transfer_id in transfer_ids + ] + for execution in executions: + self.assertExecutionCompleted(execution.id, timeout=timeout) + + def _wait_for_power_status(self, status, timeout=120): + """Poll until one of the pool's machines reaches *status*.""" + ctxt = self._get_db_context() + deadline = time.monotonic() + timeout + machines = [] + + while time.monotonic() < deadline: + pool = db_api.get_minion_pool(ctxt, self._pool_id, include_machines=True) + machines = pool.minion_machines + if any(m.power_status == status for m in machines): + return machines + time.sleep(1) + + self.fail( + "No minion machine of pool '%s' reached power status '%s' within %ds " + "(last statuses: %s)" + % ( + self._pool_id, + status, + timeout, + [m.power_status for m in machines], + ) + ) + + def test_transfer_after_pool_machine_power_cycle(self): + transfer_ids = [self._transfer.id, self._pool_transfer_b.id] + + # Concurrently executing both transfers forces the second one to allocate a new + # machine, since the pre-existing minimum one is already reserved by the first + # (up to the pool's maximum of 2). + self._execute_concurrently_and_wait(transfer_ids) + + pool = db_api.get_minion_pool( + self._get_db_context(), self._pool_id, include_machines=True + ) + self.assertEqual(2, len(pool.minion_machines)) + provider_properties_before = { + machine.id: machine.provider_properties for machine in pool.minion_machines + } + + # Let both machines' idle time expire, then refresh the pool: since their count + # exceeds the pool minimum of 1, the excess one gets powered off. + time.sleep(self._POOL_MINION_MAX_IDLE_TIME + 1) + self._client.minion_pools.refresh_minion_pool(self._pool_id) + self._wait_for_power_status(constants.MINION_MACHINE_POWER_STATUS_POWERED_OFF) + + # Re-running both transfers concurrently reuses both machines, powering the + # idled-off one back on before healthchecking and reusing it. + self._execute_concurrently_and_wait(transfer_ids) + + # The power-cycled machine must genuinely have been reused, not silently deleted + # and recreated from scratch by the healthcheck-failure fallback (which would + # defeat the whole point of the "poweroff" retention strategy) + pool = db_api.get_minion_pool( + self._get_db_context(), self._pool_id, include_machines=True + ) + self.assertEqual(2, len(pool.minion_machines)) + for machine in pool.minion_machines: + self.assertEqual( + provider_properties_before[machine.id], + machine.provider_properties, + "Minion machine '%s' provider properties changed across the power " + "cycle; it was likely deleted and recreated instead of reused." + % machine.id, + ) + + +class MinionPoolPowerCycleTransferTest( + _MinionPoolPowerCycleTestMixin, base.MinionPoolReplicaTestBase +): + """Power-cycle test exercising a destination minion pool.""" + + @property + def _pool_id(self): + return self._dst_pool_id + + +class SourceMinionPoolPowerCycleTransferTest( + _MinionPoolPowerCycleTestMixin, base.SourceMinionPoolReplicaTestBase +): + """Power-cycle test exercising a source minion pool.""" + + @property + def _pool_id(self): + return self._src_pool_id diff --git a/coriolis/tests/integration/test_provider/common.py b/coriolis/tests/integration/test_provider/common.py index 18d600dd7..5bb00bf66 100644 --- a/coriolis/tests/integration/test_provider/common.py +++ b/coriolis/tests/integration/test_provider/common.py @@ -179,5 +179,8 @@ def healthcheck_minion( username = minion_connection_info.get("username", "root") pkey = minion_connection_info.get("pkey") + # A freshly power-cycled minion needs a moment to boot sshd back up. + coriolis_utils.wait_for_port_connectivity(ip, port, max_wait=60) + client = coriolis_utils.connect_ssh(ip, port, username, pkey=pkey) client.close() diff --git a/coriolis/tests/test_utils.py b/coriolis/tests/test_utils.py index 6aece2fa1..aa36e0492 100644 --- a/coriolis/tests/test_utils.py +++ b/coriolis/tests/test_utils.py @@ -1303,6 +1303,9 @@ def test_write_systemd( get_pty=False, ), mock.call(self.mock_ssh, 'sudo systemctl daemon-reload', get_pty=False), + mock.call( + self.mock_ssh, 'sudo systemctl enable svc_name', get_pty=False + ), mock.call( self.mock_ssh, 'sudo systemctl start svc_name', get_pty=False ), @@ -1352,8 +1355,15 @@ def test_write_systemd_service_exists(self, mock_test_ssh, mock_exec_ssh_cmd): mock.call(self.mock_ssh, '/lib/systemd/system/svc_name.service'), ] ) - mock_exec_ssh_cmd.assert_called_once_with( - self.mock_ssh, 'sudo systemctl start svc_name', get_pty=False + mock_exec_ssh_cmd.assert_has_calls( + [ + mock.call( + self.mock_ssh, 'sudo systemctl enable svc_name', get_pty=False + ), + mock.call( + self.mock_ssh, 'sudo systemctl start svc_name', get_pty=False + ), + ] ) @mock.patch('coriolis.utils.exec_ssh_cmd') @@ -1371,6 +1381,7 @@ def test_write_systemd_service_selinux_exception( exception.CoriolisException(), None, None, + None, ] _write_systemd_undecorated = testutils.get_wrapped_function( @@ -1433,6 +1444,9 @@ def test_test_write_systemd_with_run_as( get_pty=False, ), mock.call(self.mock_ssh, 'sudo systemctl daemon-reload', get_pty=False), + mock.call( + self.mock_ssh, 'sudo systemctl enable svc_name', get_pty=False + ), mock.call( self.mock_ssh, 'sudo systemctl start svc_name', get_pty=False ), diff --git a/coriolis/utils.py b/coriolis/utils.py index 9e53e4c5b..e21d6ff8d 100644 --- a/coriolis/utils.py +++ b/coriolis/utils.py @@ -887,12 +887,17 @@ def _write_systemd(ssh, cmdline, svcname, run_as=None, start=True): serviceFilePath = "%s/%s.service" % (systemd_unit_dir, svcname) if test_ssh_path(ssh, serviceFilePath): + exec_ssh_cmd(ssh, "sudo systemctl enable %s" % svcname, get_pty=False) if start: exec_ssh_cmd(ssh, "sudo systemctl start %s" % svcname, get_pty=False) return def _reload_and_start(start=True): exec_ssh_cmd(ssh, "sudo systemctl daemon-reload", get_pty=False) + # NOTE: the service must be enabled so that it comes back up on its own after + # the underlying instance is rebooted or power-cycled (e.g.: minion pool + # machines reused after being powered off). + exec_ssh_cmd(ssh, "sudo systemctl enable %s" % svcname, get_pty=False) if start: exec_ssh_cmd(ssh, "sudo systemctl start %s" % svcname, get_pty=False) From 78c2f8942b4cb330bc232f969850a9f6443bc04f Mon Sep 17 00:00:00 2001 From: Claudiu Belu Date: Wed, 23 Sep 2026 22:46:44 +0000 Subject: [PATCH 2/4] integration: Adds coverage for minion pool machine deletion on refresh Mirrors MinionPoolPowerCycleTransferTest but with the default "delete" retention strategy: once the pool's excess machine (beyond its minimum) goes idle, refreshing the pool deletes it instead of powering it off. As with the power-cycle tests, two separate transfers are executed concurrently to force the pool to allocate a second machine. --- coriolis/tests/integration/base.py | 11 ++ .../tests/integration/test_minion_pools.py | 101 ++++++++++++++++-- 2 files changed, 101 insertions(+), 11 deletions(-) diff --git a/coriolis/tests/integration/base.py b/coriolis/tests/integration/base.py index 6aa278e80..58407efe9 100644 --- a/coriolis/tests/integration/base.py +++ b/coriolis/tests/integration/base.py @@ -451,6 +451,17 @@ def _execute_and_wait(self, transfer_id, timeout=600): ) self.assertExecutionCompleted(execution.id, timeout=timeout) + def _execute_concurrently_and_wait(self, transfer_ids, timeout=600): + """Start one execution per transfer id before waiting on any.""" + executions = [ + self._client.transfer_executions.create( + transfer_id, shutdown_instances=False + ) + for transfer_id in transfer_ids + ] + for execution in executions: + self.assertExecutionCompleted(execution.id, timeout=timeout) + def _execute_transfer_and_deployment(self, deployment_kwargs=None): deployment_kwargs = deployment_kwargs or {} diff --git a/coriolis/tests/integration/test_minion_pools.py b/coriolis/tests/integration/test_minion_pools.py index 7d89fc60f..e1dfbb6c3 100644 --- a/coriolis/tests/integration/test_minion_pools.py +++ b/coriolis/tests/integration/test_minion_pools.py @@ -236,17 +236,6 @@ def setUp(self): origin_minion_pool_id=self._src_pool_id, ) - def _execute_concurrently_and_wait(self, transfer_ids, timeout=600): - """Start one execution per transfer id before waiting on any.""" - executions = [ - self._client.transfer_executions.create( - transfer_id, shutdown_instances=False - ) - for transfer_id in transfer_ids - ] - for execution in executions: - self.assertExecutionCompleted(execution.id, timeout=timeout) - def _wait_for_power_status(self, status, timeout=120): """Poll until one of the pool's machines reaches *status*.""" ctxt = self._get_db_context() @@ -332,3 +321,93 @@ class SourceMinionPoolPowerCycleTransferTest( @property def _pool_id(self): return self._src_pool_id + + +class _MinionPoolRefreshDeallocationTestMixin: + """Excess pool machine gets deleted on refresh. + + Mirrors _MinionPoolPowerCycleTestMixin but with the default "delete" retention + strategy: once the pool's excess machine (beyond its minimum of 1) goes idle, + refreshing the pool deletes it instead of powering it off, exercising + + Subclasses select which side's pool gets exercised by overriding the ``_pool_id`` + property. + """ + + _POOL_MAXIMUM_MINIONS = 2 + _POOL_MINION_MAX_IDLE_TIME = 1 + + @property + def _pool_id(self): + raise NotImplementedError + + def setUp(self): + super().setUp() + + # A second transfer, independent from self._transfer (created by + # ReplicaIntegrationTestBase.setUp). Running it concurrently with self._transfer + # forces the pool to allocate a second machine, since the first is already + # reserved by self._transfer's execution. + self._pool_transfer_b = self._create_transfer( + self._src_endpoint.id, + self._dst_endpoint.id, + instances=[self._instance_name], + source_environment=self._transfer._info["source_environment"], + destination_minion_pool_id=self._dst_pool_id, + origin_minion_pool_id=self._src_pool_id, + ) + + def test_excess_pool_machine_deleted_on_refresh(self): + transfer_ids = [self._transfer.id, self._pool_transfer_b.id] + + # Concurrently executing both transfers forces the second one to allocate a new + # machine, since the pre-existing minimum one is already reserved by the first + # (up to the pool's maximum of 2). + self._execute_concurrently_and_wait(transfer_ids) + + pool = db_api.get_minion_pool( + self._get_db_context(), self._pool_id, include_machines=True + ) + self.assertEqual(2, len(pool.minion_machines)) + + # Let both machines' idle time expire, then refresh the pool: since their count + # exceeds the pool minimum of 1, the excess one gets deleted. + time.sleep(self._POOL_MINION_MAX_IDLE_TIME + 1) + self._client.minion_pools.refresh_minion_pool(self._pool_id) + + ctxt = self._get_db_context() + deadline = time.monotonic() + 120 + pool = None + while time.monotonic() < deadline: + pool = db_api.get_minion_pool(ctxt, self._pool_id, include_machines=True) + if len(pool.minion_machines) == 1: + break + time.sleep(1) + + self.assertEqual( + 1, + len(pool.minion_machines), + "Expected the excess minion machine to be deleted from pool '%s'; " + "machines still present: %s" + % (self._pool_id, [m.id for m in pool.minion_machines]), + ) + + +class MinionPoolRefreshDeallocationTransferTest( + _MinionPoolRefreshDeallocationTestMixin, base.MinionPoolReplicaTestBase +): + """Deletion-on-refresh test exercising a destination minion pool.""" + + @property + def _pool_id(self): + return self._dst_pool_id + + +class SourceMinionPoolRefreshDeallocationTransferTest( + _MinionPoolRefreshDeallocationTestMixin, base.SourceMinionPoolReplicaTestBase +): + """Deletion-on-refresh test exercising a source minion pool.""" + + @property + def _pool_id(self): + return self._src_pool_id From 908e2863518577fa895acc50f9cd12dfc55e6bcd Mon Sep 17 00:00:00 2001 From: Claudiu Belu Date: Tue, 18 Aug 2026 21:07:52 +0000 Subject: [PATCH 3/4] integration: Adds coverage for minion pool refresh cron startup recovery _init_pools_refresh_cron_jobs runs once, when a minion manager service endpoint is constructed, to re-register periodic refresh jobs for any pools that were already ALLOCATED (e.g.: after a service restart). Adds a test that asserts that cron jobs are registered as expected. --- .../tests/integration/test_minion_pools.py | 53 +++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/coriolis/tests/integration/test_minion_pools.py b/coriolis/tests/integration/test_minion_pools.py index e1dfbb6c3..13c003fcd 100644 --- a/coriolis/tests/integration/test_minion_pools.py +++ b/coriolis/tests/integration/test_minion_pools.py @@ -16,6 +16,7 @@ from coriolis import constants from coriolis.db import api as db_api +from coriolis.minion_manager.rpc import server as minion_manager_rpc_server from coriolis.tests.integration import base CONF = cfg.CONF @@ -411,3 +412,55 @@ class SourceMinionPoolRefreshDeallocationTransferTest( @property def _pool_id(self): return self._src_pool_id + + +class MinionPoolRefreshCronStartupTest(base.DestinationMinionPoolTestBase): + """Cron jobs are re-registered for pre-existing pools on startup. + + `_init_pools_refresh_cron_jobs` runs once, when a minion manager service endpoint + is instantiated, and scans the DB for already-ALLOCATED pools to re-register their + periodic refresh jobs (e.g.: after a service restart while pools were still + allocated). + """ + + def setUp(self): + super().setUp() + + self._endpoint = self._create_endpoint( + name="pool-cron-dst", + endpoint_type=self._imp_platform, + connection_info=self._imp_conn_info, + ) + + def test_startup_registers_refresh_jobs_for_existing_pools(self): + # The harness disables automatic refreshing by default (period 0) to avoid + # interference with other tests. Re-enable it so the new endpoint being + # constructed below actually registers jobs. + CONF.set_override( + "minion_pool_default_refresh_period_minutes", 1, group="minion_manager" + ) + self.addCleanup( + CONF.clear_override, + "minion_pool_default_refresh_period_minutes", + group="minion_manager", + ) + + pool = self._create_pool( + self._endpoint.id, skip_allocation=False, wait_for_allocation=True + ) + + new_endpoint = minion_manager_rpc_server.MinionManagerServerEndpoint() + self.addCleanup(new_endpoint._cron.stop) + + job_prefix = ( + minion_manager_rpc_server.MINION_POOL_REFRESH_JOB_PREFIX_FORMAT % pool.id + ) + registered = [ + name for name in new_endpoint._cron._jobs if name.startswith(job_prefix) + ] + self.assertTrue( + registered, + "Expected refresh cron jobs to be registered on startup for pre-existing " + "allocated pool '%s', got jobs: %s" + % (pool.id, list(new_endpoint._cron._jobs)), + ) From f8d57a4dfb65d36b4b6b9498dc7241f5c6439390 Mon Sep 17 00:00:00 2001 From: Claudiu Belu Date: Fri, 7 Aug 2026 09:30:50 +0000 Subject: [PATCH 4/4] integration: Adds coverage for the auto-deploy deployer handoff The integration test test_execution_auto_deploy did have auto_deploy=True, but it never actually waited the deployment to complete. When auto_deploy is True, deployer_manager is supposed to kick off the deployment automatically after the transfer completes. Adds test_execution_auto_deploy_transfer_failure test, which injects a failure into the deployer's transfer execution and waits for the deployment to be moved to the ERROR state. --- .../integration/transfers/test_executions.py | 79 +++++++++++++++++-- 1 file changed, 73 insertions(+), 6 deletions(-) diff --git a/coriolis/tests/integration/transfers/test_executions.py b/coriolis/tests/integration/transfers/test_executions.py index bf96e4b79..7bed60eab 100644 --- a/coriolis/tests/integration/transfers/test_executions.py +++ b/coriolis/tests/integration/transfers/test_executions.py @@ -5,11 +5,20 @@ Integration tests for the transfer executions. """ +from unittest import mock + from coriolis import constants from coriolis.tests.integration import base class TransferExecutionsTests(base.ReplicaIntegrationTestBase): + # Provider method to fail in test_execution_auto_deploy_transfer_failure. + # Plain transfers deploy fresh target resources via deploy_replica_target_resources, + # minion-pool-backed transfers instead attach volumes to a pre-allocated minion via + # attach_volumes_to_minion, so deploy_replica_target_resources is never called and + # would not inject any failure there. + _AUTO_DEPLOY_FAILURE_METHOD = "deploy_replica_target_resources" + def test_executions(self): # We didn't start the execution yet. executions = self._client.transfer_executions.list(self._transfer.id) @@ -55,7 +64,23 @@ def test_shutdown_instances(self): self.assertExecutionCompleted(execution.id) + def _get_transfer_deployment(self): + deployments = self._client.deployments.list() + transfer_deployments = [ + d for d in deployments if d.transfer_id == self._transfer.id + ] + self.assertEqual(1, len(transfer_deployments)) + + return transfer_deployments[0] + def test_execution_auto_deploy(self): + """auto_deploy=True hands the deployment off to the deployer manager. + + Exercises the deployer_manager -> conductor.confirm_deployer_completed handoff: + the deployer_manager service polls the PENDING deployment, notices the + underlying transfer execution (the "deployer") completed, and calls back into + the conductor to kick off the actual deployment. + """ execution = self._client.transfer_executions.create( self._transfer.id, shutdown_instances=False, @@ -69,12 +94,52 @@ def test_execution_auto_deploy(self): self.assertExecutionCompleted(execution.id) - deployments = self._client.deployments.list() - transfer_deployments = [ - d for d in deployments if d.transfer_id == self._transfer.id - ] - self.assertEqual(1, len(transfer_deployments)) - self.addCleanup(self._cleanup_deployment, transfer_deployments[0].id) + deployment = self._get_transfer_deployment() + self.addCleanup(self._cleanup_deployment, deployment.id) + + self.assertDeploymentCompleted(deployment.id) + + def test_execution_auto_deploy_transfer_failure(self): + """A failed "deployer" transfer execution errors out the deployment. + + Exercises the deployer_manager -> conductor.report_deployer_failure path: when + the transfer execution backing an auto-deployed deployment ends up in an error + state instead of COMPLETED, the deployer_manager service must report the failure + back to the conductor, so the PENDING deployment gets moved to ERROR instead of + being stuck forever. + """ + injected_error = Exception("injected auto-deploy transfer failure") + + with mock.patch.object( + self._harness.imp_provider_class, + self._AUTO_DEPLOY_FAILURE_METHOD, + side_effect=injected_error, + ): + execution = self._client.transfer_executions.create( + self._transfer.id, + shutdown_instances=False, + auto_deploy=True, + ) + self.addCleanup( + self._cleanup_execution, + self._transfer.id, + execution.id, + ) + + self.assertExecutionErrored(execution.id) + + deployment = self._get_transfer_deployment() + self.addCleanup(self._cleanup_deployment, deployment.id) + + deployment = self.wait_for_deployment( + deployment.id, desired_statuses=[constants.EXECUTION_STATUS_ERROR] + ) + self.assertEqual( + constants.EXECUTION_STATUS_ERROR, + deployment.last_execution_status, + "Deployment %s ended with status %s" + % (deployment.id, deployment.last_execution_status), + ) def test_cancel_running_execution(self): self._test_cancel_running_execution(False) @@ -128,3 +193,5 @@ class MinionPoolTransferExecutionsTests( base.MinionPoolReplicaTestBase, TransferExecutionsTests ): """Transfer executions that use a pre-allocated destination minion pool.""" + + _AUTO_DEPLOY_FAILURE_METHOD = "attach_volumes_to_minion"