From e4b9f4ae8a6fbc95c9e46b6537c1e8b52f06f645 Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Tue, 29 Sep 2026 21:57:18 +0200 Subject: [PATCH 01/10] Add Hot Aisle bare metal support Co-Authored-By: Claude Opus 5.5 (1M context) --- mkdocs/docs/concepts/backends.md | 12 +- .../_internal/core/backends/base/offers.py | 1 + .../core/backends/hotaisle/api_client.py | 32 ++ .../core/backends/hotaisle/compute.py | 134 +++++--- .../core/backends/hotaisle/__init__.py | 0 .../core/backends/hotaisle/test_api_client.py | 63 ++++ .../core/backends/hotaisle/test_compute.py | 322 ++++++++++++++++++ 7 files changed, 515 insertions(+), 49 deletions(-) create mode 100644 src/tests/_internal/core/backends/hotaisle/__init__.py create mode 100644 src/tests/_internal/core/backends/hotaisle/test_api_client.py create mode 100644 src/tests/_internal/core/backends/hotaisle/test_compute.py diff --git a/mkdocs/docs/concepts/backends.md b/mkdocs/docs/concepts/backends.md index fd23b1ad5f..2dcde07812 100644 --- a/mkdocs/docs/concepts/backends.md +++ b/mkdocs/docs/concepts/backends.md @@ -976,20 +976,18 @@ projects: ??? info "Required permissions" - The API key must have the following roles assigned: + The API key requires the `owner` role for your user and the `operator` and `user` roles for the team specified in `team_handle`. - * **Owner role for the user** - Required for creating and managing SSH keys - * **Operator role for the team** - Required for managing virtual machines within the team +??? info "Instance types" + `dstack` supports Hot Aisle VMs (`vm-mi300x-1`, `vm-mi300x-2`, `vm-mi300x-4`, `vm-mi300x-8`) and bare metal servers (`bm-mi300x-8`). To use a specific type, set [`instance_types`](../reference/dstack.yml/fleet.md#instance_types) in the fleet configuration. -??? info "Pricing" - `dstack` shows the hourly price for Hot Aisle instances. Some instances also require an upfront payment for a minimum reservation period, which is usually a few hours. You will be charged for the full minimum period even if you stop the instance early. - - See the Hot Aisle API for the minimum reservation period for each instance type: + Some instances are charged upfront for a minimum reservation period (8 hours for bare metal servers), so set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) accordingly. To check the period for each instance type:
```shell $ curl -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/virtual_machines/available/ | jq ".[] | {gpus: .Specs.gpus, MinimumReservationMinutes}" + $ curl -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/bare_metal/available/ | jq ".[] | {gpus: .Specs.gpus, MinimumReservationMinutes}" ```
diff --git a/src/dstack/_internal/core/backends/base/offers.py b/src/dstack/_internal/core/backends/base/offers.py index 33d745c1ba..43d6cb484d 100644 --- a/src/dstack/_internal/core/backends/base/offers.py +++ b/src/dstack/_internal/core/backends/base/offers.py @@ -30,6 +30,7 @@ "gcp-dws-calendar-mode", "runpod-cpu", "runpod-cluster", + "hotaisle-bm", ] diff --git a/src/dstack/_internal/core/backends/hotaisle/api_client.py b/src/dstack/_internal/core/backends/hotaisle/api_client.py index a3cc355fcd..7c90d8c1e8 100644 --- a/src/dstack/_internal/core/backends/hotaisle/api_client.py +++ b/src/dstack/_internal/core/backends/hotaisle/api_client.py @@ -3,6 +3,7 @@ import requests from dstack._internal.core.backends.base.configurator import raise_invalid_credentials_error +from dstack._internal.core.errors import NoCapacityError from dstack._internal.utils.logging import get_logger API_URL = "https://admin.hotaisle.app/api" @@ -88,6 +89,37 @@ def terminate_virtual_machine(self, vm_name: str) -> None: return response.raise_for_status() + def reserve_bare_metal_server(self, specs: Dict[str, Any], description: str) -> Dict[str, Any]: + url = f"{API_URL}/teams/{self.team_handle}/bare_metal/" + payload = {"specs": specs, "description": description} + response = self._make_request("POST", url, json=payload) + # 403: the team's bare metal server limit is reached. + # 404: no available server matches the specs, e.g. another team reserved it. + if response.status_code in [403, 404]: + raise NoCapacityError(response.text) + response.raise_for_status() + return response.json() + + def get_bare_metal_server(self, server_id: str) -> Dict[str, Any]: + url = f"{API_URL}/teams/{self.team_handle}/bare_metal/{server_id}/" + response = self._make_request("GET", url) + response.raise_for_status() + return response.json() + + def release_bare_metal_server(self, server_id: str) -> None: + url = f"{API_URL}/teams/{self.team_handle}/bare_metal/{server_id}/" + response = self._make_request( + "DELETE", + url, + params={ + "force": "true", # release even if min reservation time not met + }, + ) + if response.status_code == 404: + logger.debug("Hot Aisle bare metal server %s not found", server_id) + return + response.raise_for_status() + def _make_request( self, method: str, diff --git a/src/dstack/_internal/core/backends/hotaisle/compute.py b/src/dstack/_internal/core/backends/hotaisle/compute.py index 2fbbb37da4..71fcf1c459 100644 --- a/src/dstack/_internal/core/backends/hotaisle/compute.py +++ b/src/dstack/_internal/core/backends/hotaisle/compute.py @@ -1,7 +1,6 @@ import shlex import subprocess import tempfile -from threading import Thread from typing import Any, List, Optional import gpuhunt @@ -18,6 +17,7 @@ from dstack._internal.core.backends.base.offers import get_catalog_offers from dstack._internal.core.backends.hotaisle.api_client import HotAisleAPIClient from dstack._internal.core.backends.hotaisle.models import HotAisleConfig +from dstack._internal.core.errors import ProvisioningError from dstack._internal.core.models.backends.base import BackendType from dstack._internal.core.models.common import ( CoreModel, @@ -32,12 +32,15 @@ ) from dstack._internal.core.models.placement import PlacementGroup from dstack._internal.core.models.runs import JobProvisioningData +from dstack._internal.utils.common import get_or_error from dstack._internal.utils.logging import get_logger logger = get_logger(__name__) SUPPORTED_GPUS = ["MI300X"] +SSH_CONNECT_TIMEOUT_SECONDS = 10 +SSH_LAUNCH_TIMEOUT_SECONDS = 60 class HotAisleCompute( @@ -81,11 +84,24 @@ def create_instance( offer_backend_data = validate_extra_ignore( HotAisleOfferBackendData, instance_offer.backend_data ) - vm_data = self.api_client.create_virtual_machine(offer_backend_data.vm_specs) + if offer_backend_data.bare_metal_specs is not None: + server_data = self.api_client.reserve_bare_metal_server( + specs=offer_backend_data.bare_metal_specs, + description=instance_config.instance_name, + ) + # The deployment ID identifies this reservation, the name identifies the server. + instance_id = server_data["deployment_id"] + ip_address = server_data["ip_address"] + else: + vm_data = self.api_client.create_virtual_machine( + get_or_error(offer_backend_data.vm_specs) + ) + instance_id = vm_data["name"] + ip_address = vm_data["ip_address"] return JobProvisioningData( backend=instance_offer.backend, instance_type=instance_offer.instance, - instance_id=vm_data["name"], + instance_id=instance_id, hostname=None, internal_ip=None, region=instance_offer.region, @@ -95,7 +111,8 @@ def create_instance( dockerized=True, ssh_proxy=None, backend_data=HotAisleInstanceBackendData( - ip_address=vm_data["ip_address"] + ip_address=ip_address, + bare_metal=offer_backend_data.bare_metal_specs is not None, ).model_dump_json(), ) @@ -105,27 +122,31 @@ def update_provisioning_data( project_ssh_public_key: str, project_ssh_private_key: str, ): - vm_state = self.api_client.get_vm_state(provisioning_data.instance_id) - if vm_state == "running": - if provisioning_data.hostname is None and provisioning_data.backend_data: - backend_data = HotAisleInstanceBackendData.load(provisioning_data.backend_data) - provisioning_data.hostname = backend_data.ip_address - commands = get_shim_commands(arch=provisioning_data.instance_type.resources.cpu_arch) - launch_command = "sudo sh -c " + shlex.quote(" && ".join(commands)) - thread = Thread( - target=_start_runner, - kwargs={ - "hostname": provisioning_data.hostname, - "project_ssh_private_key": project_ssh_private_key, - "launch_command": launch_command, - }, - daemon=True, - ) - thread.start() + backend_data = HotAisleInstanceBackendData.load(provisioning_data.backend_data) + if backend_data.bare_metal: + server_data = self.api_client.get_bare_metal_server(provisioning_data.instance_id) + os_install_status = (server_data.get("os_status") or {}).get("os_install_status") + if os_install_status == "failed": + raise ProvisioningError("Hot Aisle bare metal server OS installation failed") + if os_install_status != "installed": + return + elif self.api_client.get_vm_state(provisioning_data.instance_id) != "running": + return + # Retried on the next check until the shim starts. + if not _start_runner( + hostname=backend_data.ip_address, + project_ssh_private_key=project_ssh_private_key, + arch=provisioning_data.instance_type.resources.cpu_arch, + ): + return + provisioning_data.hostname = backend_data.ip_address def terminate_instance( self, instance_id: str, region: str, backend_data: Optional[str] = None ): + if backend_data is not None and HotAisleInstanceBackendData.load(backend_data).bare_metal: + self.api_client.release_bare_metal_server(instance_id) + return vm_name = instance_id self.api_client.terminate_virtual_machine(vm_name) @@ -133,9 +154,11 @@ def terminate_instance( def _start_runner( hostname: str, project_ssh_private_key: str, - launch_command: str, -): - _launch_runner( + arch: Optional[str], +) -> bool: + commands = get_shim_commands(arch=arch) + launch_command = "sudo sh -c " + shlex.quote(" && ".join(commands)) + return _launch_runner( hostname=hostname, ssh_private_key=project_ssh_private_key, launch_command=launch_command, @@ -146,34 +169,59 @@ def _launch_runner( hostname: str, ssh_private_key: str, launch_command: str, -): +) -> bool: daemonized_command = f"{launch_command.rstrip('&')} >/tmp/dstack-shim.log 2>&1 & disown" - _run_ssh_command( + return _run_ssh_command( hostname=hostname, ssh_private_key=ssh_private_key, command=daemonized_command, ) -def _run_ssh_command(hostname: str, ssh_private_key: str, command: str): +def _run_ssh_command(hostname: str, ssh_private_key: str, command: str) -> bool: with tempfile.NamedTemporaryFile("w+", 0o600) as f: f.write(ssh_private_key) f.flush() - subprocess.run( - [ - "ssh", - "-F", - "none", - "-o", - "StrictHostKeyChecking=no", - "-i", - f.name, - f"hotaisle@{hostname}", - command, - ], - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, + try: + proc = subprocess.run( + [ + "ssh", + "-F", + "none", + "-o", + "BatchMode=yes", + "-o", + f"ConnectTimeout={SSH_CONNECT_TIMEOUT_SECONDS}", + "-o", + "ConnectionAttempts=1", + "-o", + "StrictHostKeyChecking=no", + "-o", + "UserKnownHostsFile=/dev/null", + "-o", + "LogLevel=ERROR", + "-i", + f.name, + f"hotaisle@{hostname}", + command, + ], + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + text=True, + timeout=SSH_LAUNCH_TIMEOUT_SECONDS, + ) + except subprocess.TimeoutExpired: + logger.debug("Timed out running SSH command on Hot Aisle instance %s", hostname) + return False + if proc.returncode != 0: + logger.debug( + "SSH command failed on Hot Aisle instance %s: exit_code=%s stderr=%r", + hostname, + proc.returncode, + proc.stderr[-1000:], ) + return False + return True def _supported_instances(offer: InstanceOffer) -> bool: @@ -184,6 +232,7 @@ def _supported_instances(offer: InstanceOffer) -> bool: class HotAisleInstanceBackendData(CoreModel): ip_address: str + bare_metal: bool = False @classmethod def load(cls, raw: Optional[str]) -> "HotAisleInstanceBackendData": @@ -192,4 +241,5 @@ def load(cls, raw: Optional[str]) -> "HotAisleInstanceBackendData": class HotAisleOfferBackendData(CoreModel): - vm_specs: dict[str, Any] + vm_specs: Optional[dict[str, Any]] = None + bare_metal_specs: Optional[dict[str, Any]] = None diff --git a/src/tests/_internal/core/backends/hotaisle/__init__.py b/src/tests/_internal/core/backends/hotaisle/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/src/tests/_internal/core/backends/hotaisle/test_api_client.py b/src/tests/_internal/core/backends/hotaisle/test_api_client.py new file mode 100644 index 0000000000..203d0eada3 --- /dev/null +++ b/src/tests/_internal/core/backends/hotaisle/test_api_client.py @@ -0,0 +1,63 @@ +import pytest +import requests + +from dstack._internal.core.backends.hotaisle.api_client import API_URL, HotAisleAPIClient +from dstack._internal.core.errors import NoCapacityError + +BARE_METAL_URL = f"{API_URL}/teams/test-team/bare_metal/" +SERVER_URL = f"{BARE_METAL_URL}deployment-id/" + + +def _client() -> HotAisleAPIClient: + return HotAisleAPIClient(api_key="test-key", team_handle="test-team") + + +class TestReserveBareMetalServer: + def test_posts_specs_and_description(self, requests_mock): + requests_mock.post(BARE_METAL_URL, json={"deployment_id": "deployment-id"}) + + server_data = _client().reserve_bare_metal_server( + specs={"cpu_cores": 104}, description="test-instance" + ) + + assert server_data == {"deployment_id": "deployment-id"} + assert requests_mock.last_request.json() == { + "specs": {"cpu_cores": 104}, + "description": "test-instance", + } + + @pytest.mark.parametrize( + ("status_code", "text"), + [(403, "tenant limit exceeded"), (404, "no available servers")], + ) + def test_raises_no_capacity(self, requests_mock, status_code, text): + requests_mock.post(BARE_METAL_URL, status_code=status_code, text=text) + + with pytest.raises(NoCapacityError, match=text): + _client().reserve_bare_metal_server(specs={}, description="test-instance") + + def test_raises_on_other_errors(self, requests_mock): + requests_mock.post(BARE_METAL_URL, status_code=402) + + with pytest.raises(requests.HTTPError): + _client().reserve_bare_metal_server(specs={}, description="test-instance") + + +class TestReleaseBareMetalServer: + def test_forces_release(self, requests_mock): + requests_mock.delete(SERVER_URL, status_code=204) + + _client().release_bare_metal_server("deployment-id") + + assert requests_mock.last_request.qs == {"force": ["true"]} + + def test_ignores_missing_server(self, requests_mock): + requests_mock.delete(SERVER_URL, status_code=404) + + _client().release_bare_metal_server("deployment-id") + + def test_raises_on_other_errors(self, requests_mock): + requests_mock.delete(SERVER_URL, status_code=400) + + with pytest.raises(requests.HTTPError): + _client().release_bare_metal_server("deployment-id") diff --git a/src/tests/_internal/core/backends/hotaisle/test_compute.py b/src/tests/_internal/core/backends/hotaisle/test_compute.py new file mode 100644 index 0000000000..edaebbe1c8 --- /dev/null +++ b/src/tests/_internal/core/backends/hotaisle/test_compute.py @@ -0,0 +1,322 @@ +import subprocess +from typing import Optional +from unittest.mock import MagicMock, patch + +import pytest +from gpuhunt.providers.hotaisle import API_URL + +from dstack._internal.core.backends.hotaisle.compute import ( + SSH_CONNECT_TIMEOUT_SECONDS, + SSH_LAUNCH_TIMEOUT_SECONDS, + HotAisleCompute, + HotAisleInstanceBackendData, + _run_ssh_command, +) +from dstack._internal.core.backends.hotaisle.models import HotAisleAPIKeyCreds, HotAisleConfig +from dstack._internal.core.errors import ProvisioningError +from dstack._internal.core.models.backends.base import BackendType +from dstack._internal.core.models.instances import ( + Disk, + Gpu, + InstanceAvailability, + InstanceConfiguration, + InstanceOfferWithAvailability, + InstanceType, + Resources, + SSHKey, +) +from dstack._internal.core.models.runs import JobProvisioningData + +VM_SPECS = { + "cpu_cores": 13, + "ram_capacity": 224 * 1024**3, + "disk_capacity": 12288 * 1024**3, + "cpus": {"count": 1, "manufacturer": "Intel", "model": "Xeon Platinum 8470"}, + "gpus": [{"count": 1, "manufacturer": "AMD", "model": "MI300X"}], +} + +BARE_METAL_SPECS = { + "cpu_cores": 104, + "ram_capacity": 2048 * 1024**3, + "disk_capacity": 123839994396672, + "cpus": [{"count": 2, "manufacturer": "Intel", "model": "Xeon Platinum 8470", "cores": 52}], + "gpus": [{"count": 8, "manufacturer": "AMD", "model": "MI300X"}], +} + + +def _compute() -> HotAisleCompute: + return HotAisleCompute( + HotAisleConfig(team_handle="test-team", creds=HotAisleAPIKeyCreds(api_key="test-key")) + ) + + +def _compute_with_mocked_api_client() -> HotAisleCompute: + compute = _compute() + compute.api_client = MagicMock() + return compute + + +def _instance_type(name: str) -> InstanceType: + return InstanceType( + name=name, + resources=Resources( + cpus=13, + memory_mib=224 * 1024, + gpus=[Gpu(name="MI300X", memory_mib=192 * 1024)], + spot=False, + disk=Disk(size_mib=12288 * 1024), + ), + ) + + +def _offer(name: str, backend_data: dict) -> InstanceOfferWithAvailability: + return InstanceOfferWithAvailability( + backend=BackendType.HOTAISLE, + instance=_instance_type(name), + region="us-michigan-1", + price=1.99, + backend_data=backend_data, + availability=InstanceAvailability.AVAILABLE, + ) + + +def _instance_config() -> InstanceConfiguration: + return InstanceConfiguration( + project_name="test-project", + instance_name="test-instance", + user="test-user", + ssh_keys=[SSHKey(public="ssh-rsa AAAA test")], + ) + + +def _provisioning_data( + backend_data: Optional[HotAisleInstanceBackendData], +) -> JobProvisioningData: + return JobProvisioningData( + backend=BackendType.HOTAISLE, + instance_type=_instance_type("vm-mi300x-1"), + instance_id="instance-id", + hostname=None, + internal_ip=None, + region="us-michigan-1", + price=1.99, + username="hotaisle", + ssh_port=22, + dockerized=True, + ssh_proxy=None, + backend_data=backend_data.model_dump_json() if backend_data is not None else None, + ) + + +class TestGetAllOffersWithAvailability: + def test_returns_vm_and_bare_metal_offers(self, requests_mock): + requests_mock.get( + f"{API_URL}/teams/test-team/virtual_machines/available/", + json=[{"OnDemandPrice": 199, "Specs": VM_SPECS}], + ) + requests_mock.get( + f"{API_URL}/teams/test-team/bare_metal/available/", + json=[{"OnDemandPrice": 2712, "Specs": BARE_METAL_SPECS}], + ) + + offers = _compute().get_all_offers_with_availability(unallocated_resources=False) + + assert [(offer.instance.name, offer.backend_data) for offer in offers] == [ + ("vm-mi300x-1", {"vm_specs": VM_SPECS}), + ("bm-mi300x-8", {"bare_metal_specs": BARE_METAL_SPECS}), + ] + + +class TestCreateInstance: + def test_creates_vm(self): + compute = _compute_with_mocked_api_client() + compute.api_client.create_virtual_machine.return_value = { + "name": "vm-name", + "ip_address": "10.0.0.1", + } + + provisioning_data = compute.create_instance( + _offer("vm-mi300x-1", {"vm_specs": VM_SPECS}), _instance_config(), None + ) + + compute.api_client.upload_ssh_key.assert_called_once_with("ssh-rsa AAAA test") + compute.api_client.create_virtual_machine.assert_called_once_with(VM_SPECS) + compute.api_client.reserve_bare_metal_server.assert_not_called() + assert provisioning_data.instance_id == "vm-name" + assert HotAisleInstanceBackendData.load( + provisioning_data.backend_data + ) == HotAisleInstanceBackendData(ip_address="10.0.0.1", bare_metal=False) + + def test_reserves_bare_metal_server(self): + compute = _compute_with_mocked_api_client() + compute.api_client.reserve_bare_metal_server.return_value = { + "deployment_id": "deployment-id", + "name": "server-01", + "ip_address": "10.0.0.2", + } + + provisioning_data = compute.create_instance( + _offer("bm-mi300x-8", {"bare_metal_specs": BARE_METAL_SPECS}), + _instance_config(), + None, + ) + + compute.api_client.upload_ssh_key.assert_called_once_with("ssh-rsa AAAA test") + compute.api_client.reserve_bare_metal_server.assert_called_once_with( + specs=BARE_METAL_SPECS, description="test-instance" + ) + compute.api_client.create_virtual_machine.assert_not_called() + assert provisioning_data.instance_id == "deployment-id" + assert provisioning_data.username == "hotaisle" + assert HotAisleInstanceBackendData.load( + provisioning_data.backend_data + ) == HotAisleInstanceBackendData(ip_address="10.0.0.2", bare_metal=True) + + +@patch("dstack._internal.core.backends.hotaisle.compute._run_ssh_command", return_value=True) +class TestUpdateProvisioningData: + def test_starts_shim_on_running_vm(self, ssh_mock): + compute = _compute_with_mocked_api_client() + compute.api_client.get_vm_state.return_value = "running" + provisioning_data = _provisioning_data(HotAisleInstanceBackendData(ip_address="10.0.0.1")) + + compute.update_provisioning_data(provisioning_data, "public-key", "private-key") + + compute.api_client.get_vm_state.assert_called_once_with("instance-id") + assert provisioning_data.hostname == "10.0.0.1" + ssh_mock.assert_called_once() + assert ssh_mock.call_args.kwargs["hostname"] == "10.0.0.1" + + def test_retries_if_shim_fails_to_start(self, ssh_mock): + ssh_mock.return_value = False + compute = _compute_with_mocked_api_client() + compute.api_client.get_vm_state.return_value = "running" + provisioning_data = _provisioning_data(HotAisleInstanceBackendData(ip_address="10.0.0.1")) + + compute.update_provisioning_data(provisioning_data, "public-key", "private-key") + + ssh_mock.assert_called_once() + assert provisioning_data.hostname is None + + def test_waits_for_vm_to_run(self, ssh_mock): + compute = _compute_with_mocked_api_client() + compute.api_client.get_vm_state.return_value = "shut off" + provisioning_data = _provisioning_data(HotAisleInstanceBackendData(ip_address="10.0.0.1")) + + compute.update_provisioning_data(provisioning_data, "public-key", "private-key") + + assert provisioning_data.hostname is None + ssh_mock.assert_not_called() + + def test_starts_shim_on_installed_bare_metal_server(self, ssh_mock): + compute = _compute_with_mocked_api_client() + compute.api_client.get_bare_metal_server.return_value = { + "os_status": {"os_install_status": "installed"} + } + provisioning_data = _provisioning_data( + HotAisleInstanceBackendData(ip_address="10.0.0.2", bare_metal=True) + ) + + compute.update_provisioning_data(provisioning_data, "public-key", "private-key") + + compute.api_client.get_bare_metal_server.assert_called_once_with("instance-id") + compute.api_client.get_vm_state.assert_not_called() + assert provisioning_data.hostname == "10.0.0.2" + ssh_mock.assert_called_once() + assert ssh_mock.call_args.kwargs["hostname"] == "10.0.0.2" + + @pytest.mark.parametrize( + "server_data", + [ + {"os_status": {"os_install_status": "installing_os"}}, + {"os_status": {"os_install_status": "first_boot_tasks"}}, + {"os_status": None}, + {}, + ], + ) + def test_waits_for_bare_metal_os_installation(self, ssh_mock, server_data): + compute = _compute_with_mocked_api_client() + compute.api_client.get_bare_metal_server.return_value = server_data + provisioning_data = _provisioning_data( + HotAisleInstanceBackendData(ip_address="10.0.0.2", bare_metal=True) + ) + + compute.update_provisioning_data(provisioning_data, "public-key", "private-key") + + assert provisioning_data.hostname is None + ssh_mock.assert_not_called() + + def test_raises_on_failed_bare_metal_os_installation(self, ssh_mock): + compute = _compute_with_mocked_api_client() + compute.api_client.get_bare_metal_server.return_value = { + "os_status": {"os_install_status": "failed"} + } + provisioning_data = _provisioning_data( + HotAisleInstanceBackendData(ip_address="10.0.0.2", bare_metal=True) + ) + + with pytest.raises(ProvisioningError): + compute.update_provisioning_data(provisioning_data, "public-key", "private-key") + ssh_mock.assert_not_called() + + +class TestTerminateInstance: + def test_terminates_vm(self): + compute = _compute_with_mocked_api_client() + + compute.terminate_instance( + "vm-name", + "us-michigan-1", + HotAisleInstanceBackendData(ip_address="10.0.0.1").model_dump_json(), + ) + + compute.api_client.terminate_virtual_machine.assert_called_once_with("vm-name") + compute.api_client.release_bare_metal_server.assert_not_called() + + def test_terminates_vm_without_backend_data(self): + compute = _compute_with_mocked_api_client() + + compute.terminate_instance("vm-name", "us-michigan-1", None) + + compute.api_client.terminate_virtual_machine.assert_called_once_with("vm-name") + + def test_releases_bare_metal_server(self): + compute = _compute_with_mocked_api_client() + + compute.terminate_instance( + "deployment-id", + "us-michigan-1", + HotAisleInstanceBackendData(ip_address="10.0.0.2", bare_metal=True).model_dump_json(), + ) + + compute.api_client.release_bare_metal_server.assert_called_once_with("deployment-id") + compute.api_client.terminate_virtual_machine.assert_not_called() + + +@patch("dstack._internal.core.backends.hotaisle.compute.subprocess.run") +class TestRunSSHCommand: + def test_returns_true_on_success(self, run_mock): + run_mock.return_value = subprocess.CompletedProcess(args=[], returncode=0, stderr="") + + assert _run_ssh_command("10.0.0.1", "private-key", "true") + + args = run_mock.call_args.args[0] + assert "hotaisle@10.0.0.1" in args + assert f"ConnectTimeout={SSH_CONNECT_TIMEOUT_SECONDS}" in args + assert run_mock.call_args.kwargs["timeout"] == SSH_LAUNCH_TIMEOUT_SECONDS + + @pytest.mark.parametrize( + "run_result", + [ + subprocess.CompletedProcess(args=[], returncode=255, stderr="Connection refused"), + subprocess.TimeoutExpired(cmd="ssh", timeout=SSH_LAUNCH_TIMEOUT_SECONDS), + ], + ids=["ssh-fails", "ssh-times-out"], + ) + def test_returns_false_on_failure(self, run_mock, run_result): + if isinstance(run_result, Exception): + run_mock.side_effect = run_result + else: + run_mock.return_value = run_result + + assert not _run_ssh_command("10.0.0.1", "private-key", "true") From 931a9d8b88da99fbd05da845f3075acf5024ebba Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Wed, 30 Sep 2026 09:25:09 +0200 Subject: [PATCH 02/10] Add a flag to release Hot Aisle bare metal without force Co-Authored-By: Claude Opus 5.5 (1M context) --- .../core/backends/hotaisle/api_client.py | 19 +++++++------- .../core/backends/hotaisle/compute.py | 5 +++- src/dstack/_internal/settings.py | 9 +++++++ .../core/backends/hotaisle/test_api_client.py | 15 ++++++++++- .../core/backends/hotaisle/test_compute.py | 25 +++++++++++++------ 5 files changed, 55 insertions(+), 18 deletions(-) diff --git a/src/dstack/_internal/core/backends/hotaisle/api_client.py b/src/dstack/_internal/core/backends/hotaisle/api_client.py index 7c90d8c1e8..ea09391c86 100644 --- a/src/dstack/_internal/core/backends/hotaisle/api_client.py +++ b/src/dstack/_internal/core/backends/hotaisle/api_client.py @@ -3,7 +3,7 @@ import requests from dstack._internal.core.backends.base.configurator import raise_invalid_credentials_error -from dstack._internal.core.errors import NoCapacityError +from dstack._internal.core.errors import BackendError, NoCapacityError from dstack._internal.utils.logging import get_logger API_URL = "https://admin.hotaisle.app/api" @@ -106,18 +106,19 @@ def get_bare_metal_server(self, server_id: str) -> Dict[str, Any]: response.raise_for_status() return response.json() - def release_bare_metal_server(self, server_id: str) -> None: + def release_bare_metal_server(self, server_id: str, force: bool = True) -> None: url = f"{API_URL}/teams/{self.team_handle}/bare_metal/{server_id}/" - response = self._make_request( - "DELETE", - url, - params={ - "force": "true", # release even if min reservation time not met - }, - ) + # force releases even if min reservation time not met + params = {"force": "true"} if force else None + response = self._make_request("DELETE", url, params=params) if response.status_code == 404: logger.debug("Hot Aisle bare metal server %s not found", server_id) return + if response.status_code == 400 and not force: + raise BackendError( + f"Hot Aisle refused to release bare metal server {server_id}" + f" before its minimum reservation period ends: {response.text}" + ) response.raise_for_status() def _make_request( diff --git a/src/dstack/_internal/core/backends/hotaisle/compute.py b/src/dstack/_internal/core/backends/hotaisle/compute.py index 71fcf1c459..d4fb9ea59f 100644 --- a/src/dstack/_internal/core/backends/hotaisle/compute.py +++ b/src/dstack/_internal/core/backends/hotaisle/compute.py @@ -32,6 +32,7 @@ ) from dstack._internal.core.models.placement import PlacementGroup from dstack._internal.core.models.runs import JobProvisioningData +from dstack._internal.settings import FeatureFlags from dstack._internal.utils.common import get_or_error from dstack._internal.utils.logging import get_logger @@ -145,7 +146,9 @@ def terminate_instance( self, instance_id: str, region: str, backend_data: Optional[str] = None ): if backend_data is not None and HotAisleInstanceBackendData.load(backend_data).bare_metal: - self.api_client.release_bare_metal_server(instance_id) + self.api_client.release_bare_metal_server( + instance_id, force=not FeatureFlags.HOTAISLE_BARE_METAL_NO_FORCE_RELEASE + ) return vm_name = instance_id self.api_client.terminate_virtual_machine(vm_name) diff --git a/src/dstack/_internal/settings.py b/src/dstack/_internal/settings.py index b04b94b856..4e06a20c5b 100644 --- a/src/dstack/_internal/settings.py +++ b/src/dstack/_internal/settings.py @@ -50,3 +50,12 @@ class FeatureFlags: """If DSTACK_FF_CLI_PRINT_JOB_CONNECTION_INFO enabled, `dstack apply` command prints server-provided IDE URL(s) and SSH command(s) before job logs (for dev-environments only). """ + + # TODO: Change the default to "0" before merging https://github.com/dstackai/dstack/pull/4325. + HOTAISLE_BARE_METAL_NO_FORCE_RELEASE = ( + os.getenv("DSTACK_FF_HOTAISLE_BARE_METAL_NO_FORCE_RELEASE", "1") != "0" + ) + """Enabled unless set to `0`. If enabled, Hot Aisle bare metal servers are deleted without `force`, + so a server still within its minimum reservation period isn't deleted and must be released + manually. This prevents accidentally losing a prepaid server. + """ diff --git a/src/tests/_internal/core/backends/hotaisle/test_api_client.py b/src/tests/_internal/core/backends/hotaisle/test_api_client.py index 203d0eada3..c69c969d32 100644 --- a/src/tests/_internal/core/backends/hotaisle/test_api_client.py +++ b/src/tests/_internal/core/backends/hotaisle/test_api_client.py @@ -2,7 +2,7 @@ import requests from dstack._internal.core.backends.hotaisle.api_client import API_URL, HotAisleAPIClient -from dstack._internal.core.errors import NoCapacityError +from dstack._internal.core.errors import BackendError, NoCapacityError BARE_METAL_URL = f"{API_URL}/teams/test-team/bare_metal/" SERVER_URL = f"{BARE_METAL_URL}deployment-id/" @@ -51,6 +51,19 @@ def test_forces_release(self, requests_mock): assert requests_mock.last_request.qs == {"force": ["true"]} + def test_releases_without_force(self, requests_mock): + requests_mock.delete(SERVER_URL, status_code=204) + + _client().release_bare_metal_server("deployment-id", force=False) + + assert requests_mock.last_request.qs == {} + + def test_raises_if_refused_without_force(self, requests_mock): + requests_mock.delete(SERVER_URL, status_code=400, text="minimum usage requirement") + + with pytest.raises(BackendError, match="minimum usage requirement"): + _client().release_bare_metal_server("deployment-id", force=False) + def test_ignores_missing_server(self, requests_mock): requests_mock.delete(SERVER_URL, status_code=404) diff --git a/src/tests/_internal/core/backends/hotaisle/test_compute.py b/src/tests/_internal/core/backends/hotaisle/test_compute.py index edaebbe1c8..27bdef7288 100644 --- a/src/tests/_internal/core/backends/hotaisle/test_compute.py +++ b/src/tests/_internal/core/backends/hotaisle/test_compute.py @@ -26,6 +26,7 @@ SSHKey, ) from dstack._internal.core.models.runs import JobProvisioningData +from dstack._internal.settings import FeatureFlags VM_SPECS = { "cpu_cores": 13, @@ -280,16 +281,26 @@ def test_terminates_vm_without_backend_data(self): compute.api_client.terminate_virtual_machine.assert_called_once_with("vm-name") - def test_releases_bare_metal_server(self): + @pytest.mark.parametrize( + ("no_force_release", "force"), + [(False, True), (True, False)], + ids=["force", "no-force-flag"], + ) + def test_releases_bare_metal_server(self, no_force_release, force): compute = _compute_with_mocked_api_client() - compute.terminate_instance( - "deployment-id", - "us-michigan-1", - HotAisleInstanceBackendData(ip_address="10.0.0.2", bare_metal=True).model_dump_json(), + with patch.object(FeatureFlags, "HOTAISLE_BARE_METAL_NO_FORCE_RELEASE", no_force_release): + compute.terminate_instance( + "deployment-id", + "us-michigan-1", + HotAisleInstanceBackendData( + ip_address="10.0.0.2", bare_metal=True + ).model_dump_json(), + ) + + compute.api_client.release_bare_metal_server.assert_called_once_with( + "deployment-id", force=force ) - - compute.api_client.release_bare_metal_server.assert_called_once_with("deployment-id") compute.api_client.terminate_virtual_machine.assert_not_called() From b2b315479c71a9792e9b9d0552ed984b05d05d0c Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Wed, 30 Sep 2026 09:32:43 +0200 Subject: [PATCH 03/10] Start Hot Aisle shim with nohup instead of disown The shim start is now retried until the SSH command succeeds, and `disown` fails in shells other than bash even though the shim starts. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../core/backends/hotaisle/api_client.py | 2 +- .../core/backends/hotaisle/compute.py | 5 +++- .../core/backends/hotaisle/test_compute.py | 25 ++++++++++++++----- 3 files changed, 24 insertions(+), 8 deletions(-) diff --git a/src/dstack/_internal/core/backends/hotaisle/api_client.py b/src/dstack/_internal/core/backends/hotaisle/api_client.py index ea09391c86..003953d80b 100644 --- a/src/dstack/_internal/core/backends/hotaisle/api_client.py +++ b/src/dstack/_internal/core/backends/hotaisle/api_client.py @@ -93,7 +93,7 @@ def reserve_bare_metal_server(self, specs: Dict[str, Any], description: str) -> url = f"{API_URL}/teams/{self.team_handle}/bare_metal/" payload = {"specs": specs, "description": description} response = self._make_request("POST", url, json=payload) - # 403: the team's bare metal server limit is reached. + # 403: the team's bare metal server limit is reached or the API key lacks permissions. # 404: no available server matches the specs, e.g. another team reserved it. if response.status_code in [403, 404]: raise NoCapacityError(response.text) diff --git a/src/dstack/_internal/core/backends/hotaisle/compute.py b/src/dstack/_internal/core/backends/hotaisle/compute.py index d4fb9ea59f..a3dfd2847c 100644 --- a/src/dstack/_internal/core/backends/hotaisle/compute.py +++ b/src/dstack/_internal/core/backends/hotaisle/compute.py @@ -173,7 +173,10 @@ def _launch_runner( ssh_private_key: str, launch_command: str, ) -> bool: - daemonized_command = f"{launch_command.rstrip('&')} >/tmp/dstack-shim.log 2>&1 & disown" + # nohup instead of disown, which isn't available in all shells and would fail the exit code. + daemonized_command = ( + f"nohup {launch_command.rstrip('&')} >/tmp/dstack-shim.log 2>&1 Date: Wed, 30 Sep 2026 09:41:32 +0200 Subject: [PATCH 04/10] Connect to Hot Aisle bare metal via ssh_access The bare metal server's ip_address is private (100.64.0.0/10); SSH is exposed on ssh_access. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../core/backends/hotaisle/compute.py | 20 ++++++++++-- .../core/backends/hotaisle/test_compute.py | 31 ++++++++++++++----- 2 files changed, 41 insertions(+), 10 deletions(-) diff --git a/src/dstack/_internal/core/backends/hotaisle/compute.py b/src/dstack/_internal/core/backends/hotaisle/compute.py index a3dfd2847c..9a919e0d2a 100644 --- a/src/dstack/_internal/core/backends/hotaisle/compute.py +++ b/src/dstack/_internal/core/backends/hotaisle/compute.py @@ -124,6 +124,8 @@ def update_provisioning_data( project_ssh_private_key: str, ): backend_data = HotAisleInstanceBackendData.load(provisioning_data.backend_data) + hostname = backend_data.ip_address + port = 22 if backend_data.bare_metal: server_data = self.api_client.get_bare_metal_server(provisioning_data.instance_id) os_install_status = (server_data.get("os_status") or {}).get("os_install_status") @@ -131,16 +133,22 @@ def update_provisioning_data( raise ProvisioningError("Hot Aisle bare metal server OS installation failed") if os_install_status != "installed": return + # The bare metal server's ip_address is private, SSH is exposed via ssh_access. + ssh_access = server_data.get("ssh_access") or {} + hostname = ssh_access.get("ip_address") or hostname + port = ssh_access.get("port") or port elif self.api_client.get_vm_state(provisioning_data.instance_id) != "running": return # Retried on the next check until the shim starts. if not _start_runner( - hostname=backend_data.ip_address, + hostname=hostname, + port=port, project_ssh_private_key=project_ssh_private_key, arch=provisioning_data.instance_type.resources.cpu_arch, ): return - provisioning_data.hostname = backend_data.ip_address + provisioning_data.hostname = hostname + provisioning_data.ssh_port = port def terminate_instance( self, instance_id: str, region: str, backend_data: Optional[str] = None @@ -156,6 +164,7 @@ def terminate_instance( def _start_runner( hostname: str, + port: int, project_ssh_private_key: str, arch: Optional[str], ) -> bool: @@ -163,6 +172,7 @@ def _start_runner( launch_command = "sudo sh -c " + shlex.quote(" && ".join(commands)) return _launch_runner( hostname=hostname, + port=port, ssh_private_key=project_ssh_private_key, launch_command=launch_command, ) @@ -170,6 +180,7 @@ def _start_runner( def _launch_runner( hostname: str, + port: int, ssh_private_key: str, launch_command: str, ) -> bool: @@ -179,12 +190,13 @@ def _launch_runner( ) return _run_ssh_command( hostname=hostname, + port=port, ssh_private_key=ssh_private_key, command=daemonized_command, ) -def _run_ssh_command(hostname: str, ssh_private_key: str, command: str) -> bool: +def _run_ssh_command(hostname: str, port: int, ssh_private_key: str, command: str) -> bool: with tempfile.NamedTemporaryFile("w+", 0o600) as f: f.write(ssh_private_key) f.flush() @@ -208,6 +220,8 @@ def _run_ssh_command(hostname: str, ssh_private_key: str, command: str) -> bool: "LogLevel=ERROR", "-i", f.name, + "-p", + str(port), f"hotaisle@{hostname}", command, ], diff --git a/src/tests/_internal/core/backends/hotaisle/test_compute.py b/src/tests/_internal/core/backends/hotaisle/test_compute.py index d4816cb13f..2ad0b61e52 100644 --- a/src/tests/_internal/core/backends/hotaisle/test_compute.py +++ b/src/tests/_internal/core/backends/hotaisle/test_compute.py @@ -186,6 +186,7 @@ def test_starts_shim_on_running_vm(self, ssh_mock): compute.api_client.get_vm_state.assert_called_once_with("instance-id") assert provisioning_data.hostname == "10.0.0.1" + assert provisioning_data.ssh_port == 22 ssh_mock.assert_called_once() assert ssh_mock.call_args.kwargs["hostname"] == "10.0.0.1" @@ -210,10 +211,22 @@ def test_waits_for_vm_to_run(self, ssh_mock): assert provisioning_data.hostname is None ssh_mock.assert_not_called() - def test_starts_shim_on_installed_bare_metal_server(self, ssh_mock): + @pytest.mark.parametrize( + ("ssh_access", "hostname", "port"), + [ + ({"ip_address": "203.0.113.10", "port": 2222}, "203.0.113.10", 2222), + (None, "10.0.0.2", 22), + ], + ids=["ssh-access", "no-ssh-access"], + ) + def test_starts_shim_on_installed_bare_metal_server( + self, ssh_mock, ssh_access, hostname, port + ): compute = _compute_with_mocked_api_client() compute.api_client.get_bare_metal_server.return_value = { - "os_status": {"os_install_status": "installed"} + "ip_address": "10.0.0.2", + "ssh_access": ssh_access, + "os_status": {"os_install_status": "installed"}, } provisioning_data = _provisioning_data( HotAisleInstanceBackendData(ip_address="10.0.0.2", bare_metal=True) @@ -223,9 +236,11 @@ def test_starts_shim_on_installed_bare_metal_server(self, ssh_mock): compute.api_client.get_bare_metal_server.assert_called_once_with("instance-id") compute.api_client.get_vm_state.assert_not_called() - assert provisioning_data.hostname == "10.0.0.2" + assert provisioning_data.hostname == hostname + assert provisioning_data.ssh_port == port ssh_mock.assert_called_once() - assert ssh_mock.call_args.kwargs["hostname"] == "10.0.0.2" + assert ssh_mock.call_args.kwargs["hostname"] == hostname + assert ssh_mock.call_args.kwargs["port"] == port @pytest.mark.parametrize( "server_data", @@ -308,10 +323,11 @@ def test_releases_bare_metal_server(self, no_force_release, force): class TestLaunchRunner: @patch("dstack._internal.core.backends.hotaisle.compute._run_ssh_command", return_value=True) def test_daemonizes_without_disown(self, ssh_mock): - assert _launch_runner("10.0.0.1", "private-key", "sudo sh -c 'dstack-shim'") + assert _launch_runner("10.0.0.1", 22, "private-key", "sudo sh -c 'dstack-shim'") ssh_mock.assert_called_once_with( hostname="10.0.0.1", + port=22, ssh_private_key="private-key", command="nohup sudo sh -c 'dstack-shim' >/tmp/dstack-shim.log 2>&1 Date: Wed, 30 Sep 2026 16:41:09 +0200 Subject: [PATCH 05/10] Don't fail Hot Aisle docs commands when nothing is in stock Co-Authored-By: Claude Opus 5.5 (1M context) --- mkdocs/docs/concepts/backends.md | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mkdocs/docs/concepts/backends.md b/mkdocs/docs/concepts/backends.md index 2dcde07812..5e3b718d46 100644 --- a/mkdocs/docs/concepts/backends.md +++ b/mkdocs/docs/concepts/backends.md @@ -981,13 +981,13 @@ projects: ??? info "Instance types" `dstack` supports Hot Aisle VMs (`vm-mi300x-1`, `vm-mi300x-2`, `vm-mi300x-4`, `vm-mi300x-8`) and bare metal servers (`bm-mi300x-8`). To use a specific type, set [`instance_types`](../reference/dstack.yml/fleet.md#instance_types) in the fleet configuration. - Some instances are charged upfront for a minimum reservation period (8 hours for bare metal servers), so set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) accordingly. To check the period for each instance type: + Some instances are charged upfront for a minimum reservation period (8 hours for bare metal servers), so set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) accordingly. To check the period for instance types currently in stock:
```shell - $ curl -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/virtual_machines/available/ | jq ".[] | {gpus: .Specs.gpus, MinimumReservationMinutes}" - $ curl -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/bare_metal/available/ | jq ".[] | {gpus: .Specs.gpus, MinimumReservationMinutes}" + $ curl -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/virtual_machines/available/ | jq ".[]? | {gpus: .Specs.gpus, MinimumReservationMinutes}" + $ curl -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/bare_metal/available/ | jq ".[]? | {gpus: .Specs.gpus, MinimumReservationMinutes}" ```
From f030ee0726a436795833418f6341845debd29a67 Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Wed, 30 Sep 2026 16:41:58 +0200 Subject: [PATCH 06/10] Tighten Hot Aisle minimum reservation note Co-Authored-By: Claude Opus 5.5 (1M context) --- mkdocs/docs/concepts/backends.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mkdocs/docs/concepts/backends.md b/mkdocs/docs/concepts/backends.md index 5e3b718d46..5a23ee832c 100644 --- a/mkdocs/docs/concepts/backends.md +++ b/mkdocs/docs/concepts/backends.md @@ -981,7 +981,7 @@ projects: ??? info "Instance types" `dstack` supports Hot Aisle VMs (`vm-mi300x-1`, `vm-mi300x-2`, `vm-mi300x-4`, `vm-mi300x-8`) and bare metal servers (`bm-mi300x-8`). To use a specific type, set [`instance_types`](../reference/dstack.yml/fleet.md#instance_types) in the fleet configuration. - Some instances are charged upfront for a minimum reservation period (8 hours for bare metal servers), so set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) accordingly. To check the period for instance types currently in stock: + Some instances are prepaid for a minimum period (8 hours for bare metal), so set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) accordingly. To check it for instance types in stock:
From da7c7cc98cfb22fb521bf314b11de29a412c35b3 Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Wed, 30 Sep 2026 16:42:32 +0200 Subject: [PATCH 07/10] Silence curl progress in Hot Aisle docs commands Co-Authored-By: Claude Opus 5.5 (1M context) --- mkdocs/docs/concepts/backends.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mkdocs/docs/concepts/backends.md b/mkdocs/docs/concepts/backends.md index 5a23ee832c..f5bf65215d 100644 --- a/mkdocs/docs/concepts/backends.md +++ b/mkdocs/docs/concepts/backends.md @@ -986,8 +986,8 @@ projects:
```shell - $ curl -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/virtual_machines/available/ | jq ".[]? | {gpus: .Specs.gpus, MinimumReservationMinutes}" - $ curl -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/bare_metal/available/ | jq ".[]? | {gpus: .Specs.gpus, MinimumReservationMinutes}" + $ curl -sS -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/virtual_machines/available/ | jq ".[]? | {gpus: .Specs.gpus, MinimumReservationMinutes}" + $ curl -sS -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/bare_metal/available/ | jq ".[]? | {gpus: .Specs.gpus, MinimumReservationMinutes}" ```
From 46f3e09b400610f4f684da83ef137dedad7cea76 Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Wed, 30 Sep 2026 16:45:14 +0200 Subject: [PATCH 08/10] Recommend fixed fleet nodes for prepaid Hot Aisle instances Co-Authored-By: Claude Opus 5.5 (1M context) --- mkdocs/docs/concepts/backends.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mkdocs/docs/concepts/backends.md b/mkdocs/docs/concepts/backends.md index f5bf65215d..f909ded2c2 100644 --- a/mkdocs/docs/concepts/backends.md +++ b/mkdocs/docs/concepts/backends.md @@ -981,7 +981,7 @@ projects: ??? info "Instance types" `dstack` supports Hot Aisle VMs (`vm-mi300x-1`, `vm-mi300x-2`, `vm-mi300x-4`, `vm-mi300x-8`) and bare metal servers (`bm-mi300x-8`). To use a specific type, set [`instance_types`](../reference/dstack.yml/fleet.md#instance_types) in the fleet configuration. - Some instances are prepaid for a minimum period (8 hours for bare metal), so set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) accordingly. To check it for instance types in stock: + Some instances are prepaid for a minimum period (8 hours for bare metal). To avoid releasing them early, use a fleet with a fixed number of [`nodes`](../concepts/fleets.md#nodes), or at least set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) to cover the period. To check it for instance types in stock:
From 25eb30c3974f2c4e3273c38ab4d6466d7b4116cb Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Thu, 1 Oct 2026 09:50:41 +0200 Subject: [PATCH 09/10] Mark Hot Aisle bare metal as experimental Keep the no-force flag enabled; drop it once offers carry the minimum reservation period and instances stay idle until it ends. Co-Authored-By: Claude Opus 5.5 (1M context) --- mkdocs/docs/concepts/backends.md | 4 +++- src/dstack/_internal/settings.py | 3 ++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/mkdocs/docs/concepts/backends.md b/mkdocs/docs/concepts/backends.md index f909ded2c2..b34de5781c 100644 --- a/mkdocs/docs/concepts/backends.md +++ b/mkdocs/docs/concepts/backends.md @@ -981,7 +981,7 @@ projects: ??? info "Instance types" `dstack` supports Hot Aisle VMs (`vm-mi300x-1`, `vm-mi300x-2`, `vm-mi300x-4`, `vm-mi300x-8`) and bare metal servers (`bm-mi300x-8`). To use a specific type, set [`instance_types`](../reference/dstack.yml/fleet.md#instance_types) in the fleet configuration. - Some instances are prepaid for a minimum period (8 hours for bare metal). To avoid releasing them early, use a fleet with a fixed number of [`nodes`](../concepts/fleets.md#nodes), or at least set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) to cover the period. To check it for instance types in stock: + Some instances are prepaid for a minimum period (8 hours for bare metal). To avoid terminating them early, use a fleet with a fixed number of [`nodes`](../concepts/fleets.md#nodes), or at least set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) to cover the period. To check it for instance types in stock:
@@ -992,6 +992,8 @@ projects:
+ > Bare metal support is experimental. Within the minimum period, `dstack` doesn't terminate bare metal servers but stops tracking them — terminate them manually in Hot Aisle. After the period, `dstack` terminates them automatically. + ### JarvisLabs Log into your [JarvisLabs](https://cloud.jarvislabs.ai/) account and create an API key. diff --git a/src/dstack/_internal/settings.py b/src/dstack/_internal/settings.py index 4e06a20c5b..1d890f1643 100644 --- a/src/dstack/_internal/settings.py +++ b/src/dstack/_internal/settings.py @@ -51,7 +51,8 @@ class FeatureFlags: IDE URL(s) and SSH command(s) before job logs (for dev-environments only). """ - # TODO: Change the default to "0" before merging https://github.com/dstackai/dstack/pull/4325. + # TODO: Drop once offers carry the minimum reservation period and `dstack` keeps such instances + # idle until it ends. HOTAISLE_BARE_METAL_NO_FORCE_RELEASE = ( os.getenv("DSTACK_FF_HOTAISLE_BARE_METAL_NO_FORCE_RELEASE", "1") != "0" ) From b17471b568fbd962d9bf5973f19208eff183adde Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Thu, 1 Oct 2026 11:53:52 +0200 Subject: [PATCH 10/10] Make Hot Aisle bare metal opt-in via `bare_metal: true` Bare metal offers are suggested only if the backend sets `bare_metal: true`, like RunPod's `community_cloud`. Exclude the field from client requests when unset to keep older servers working. Co-Authored-By: Claude Opus 5.5 (1M context) --- mkdocs/docs/concepts/backends.md | 23 +++++---- .../core/backends/hotaisle/compute.py | 9 +++- .../core/backends/hotaisle/models.py | 17 +++++++ .../_internal/core/compatibility/backends.py | 15 ++++++ src/dstack/api/server/_backends.py | 7 ++- .../core/backends/hotaisle/test_compute.py | 47 ++++++++++++++++--- .../_internal/core/compatibility/__init__.py | 0 .../core/compatibility/test_backends.py | 27 +++++++++++ 8 files changed, 126 insertions(+), 19 deletions(-) create mode 100644 src/dstack/_internal/core/compatibility/backends.py create mode 100644 src/tests/_internal/core/compatibility/__init__.py create mode 100644 src/tests/_internal/core/compatibility/test_backends.py diff --git a/mkdocs/docs/concepts/backends.md b/mkdocs/docs/concepts/backends.md index b34de5781c..58cff86371 100644 --- a/mkdocs/docs/concepts/backends.md +++ b/mkdocs/docs/concepts/backends.md @@ -978,21 +978,26 @@ projects: ??? info "Required permissions" The API key requires the `owner` role for your user and the `operator` and `user` roles for the team specified in `team_handle`. -??? info "Instance types" - `dstack` supports Hot Aisle VMs (`vm-mi300x-1`, `vm-mi300x-2`, `vm-mi300x-4`, `vm-mi300x-8`) and bare metal servers (`bm-mi300x-8`). To use a specific type, set [`instance_types`](../reference/dstack.yml/fleet.md#instance_types) in the fleet configuration. +??? info "Bare metal" + Bare metal servers (`bm-mi300x-8`) are experimental and disabled by default. To enable them, set `bare_metal: true` in the backend settings: - Some instances are prepaid for a minimum period (8 hours for bare metal). To avoid terminating them early, use a fleet with a fixed number of [`nodes`](../concepts/fleets.md#nodes), or at least set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) to cover the period. To check it for instance types in stock: - -
+
- ```shell - $ curl -sS -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/virtual_machines/available/ | jq ".[]? | {gpus: .Specs.gpus, MinimumReservationMinutes}" - $ curl -sS -H "Authorization: Token $API_KEY" https://admin.hotaisle.app/api/teams/$TEAM_HANDLE/bare_metal/available/ | jq ".[]? | {gpus: .Specs.gpus, MinimumReservationMinutes}" + ```yaml + projects: + - name: main + backends: + - type: hotaisle + team_handle: hotaisle-team-handle + creds: + type: api_key + api_key: 9c27a4bb7a8e472fae12ab34.3f2e3c1db75b9a0187fd2196c6b3e56d2b912e1c439ba08d89e7b6fcd4ef1d3f + bare_metal: true ```
- > Bare metal support is experimental. Within the minimum period, `dstack` doesn't terminate bare metal servers but stops tracking them — terminate them manually in Hot Aisle. After the period, `dstack` terminates them automatically. + Bare metal servers are prepaid for 8 hours. To use the full period, use a fleet with a fixed number of [`nodes`](../concepts/fleets.md#nodes), or at least set [`idle_duration`](../reference/dstack.yml/fleet.md#idle_duration) to cover it. Within the period, `dstack` doesn't terminate them but stops tracking them — terminate them manually in Hot Aisle. After the period, `dstack` terminates them automatically. ### JarvisLabs diff --git a/src/dstack/_internal/core/backends/hotaisle/compute.py b/src/dstack/_internal/core/backends/hotaisle/compute.py index 9a919e0d2a..5e1d798396 100644 --- a/src/dstack/_internal/core/backends/hotaisle/compute.py +++ b/src/dstack/_internal/core/backends/hotaisle/compute.py @@ -67,7 +67,9 @@ def get_all_offers_with_availability( backend=BackendType.HOTAISLE, locations=self.config.regions or None, catalog=self.catalog, - extra_filter=_supported_instances, + extra_filter=lambda o: ( + _supported_instances(o) and (self.config.allow_bare_metal or not _is_bare_metal(o)) + ), ) return [ offer.with_availability(availability=InstanceAvailability.AVAILABLE) @@ -250,6 +252,11 @@ def _supported_instances(offer: InstanceOffer) -> bool: ) +def _is_bare_metal(offer: InstanceOffer) -> bool: + offer_backend_data = validate_extra_ignore(HotAisleOfferBackendData, offer.backend_data) + return offer_backend_data.bare_metal_specs is not None + + class HotAisleInstanceBackendData(CoreModel): ip_address: str bare_metal: bool = False diff --git a/src/dstack/_internal/core/backends/hotaisle/models.py b/src/dstack/_internal/core/backends/hotaisle/models.py index efee6b4e93..7bc730a98e 100644 --- a/src/dstack/_internal/core/backends/hotaisle/models.py +++ b/src/dstack/_internal/core/backends/hotaisle/models.py @@ -4,6 +4,8 @@ from dstack._internal.core.models.common import CoreModel +HOTAISLE_BARE_METAL_DEFAULT = False + class HotAisleAPIKeyCreds(CoreModel): type: Annotated[Literal["api_key"], Field(description="The type of credentials")] = "api_key" @@ -24,6 +26,15 @@ class HotAisleBackendConfig(CoreModel): Optional[List[str]], Field(description="The list of Hot Aisle regions. Omit to use all regions"), ] = None + bare_metal: Annotated[ + Optional[bool], + Field( + description=( + "Whether bare metal offers can be suggested in addition to VMs (experimental)." + f" Defaults to `{str(HOTAISLE_BARE_METAL_DEFAULT).lower()}`" + ) + ), + ] = None class HotAisleBackendConfigWithCreds(HotAisleBackendConfig): @@ -43,3 +54,9 @@ class HotAisleStoredConfig(HotAisleBackendConfig): class HotAisleConfig(HotAisleStoredConfig): creds: AnyHotAisleCreds + + @property + def allow_bare_metal(self) -> bool: + if self.bare_metal is not None: + return self.bare_metal + return HOTAISLE_BARE_METAL_DEFAULT diff --git a/src/dstack/_internal/core/compatibility/backends.py b/src/dstack/_internal/core/compatibility/backends.py new file mode 100644 index 0000000000..a4bb7f21a5 --- /dev/null +++ b/src/dstack/_internal/core/compatibility/backends.py @@ -0,0 +1,15 @@ +from dstack._internal.core.backends.hotaisle.models import HotAisleBackendConfigWithCreds +from dstack._internal.core.backends.models import AnyBackendConfigWithCreds +from dstack._internal.core.models.common import IncludeExcludeDictType + + +def get_backend_config_excludes(config: AnyBackendConfigWithCreds) -> IncludeExcludeDictType: + """ + Returns `config` exclude mapping to exclude certain fields from the create/update backend + request. Use this method to exclude new fields when they are not set to keep + clients backward-compatibility with older servers. + """ + excludes: IncludeExcludeDictType = {} + if isinstance(config, HotAisleBackendConfigWithCreds) and config.bare_metal is None: + excludes["bare_metal"] = True + return excludes diff --git a/src/dstack/api/server/_backends.py b/src/dstack/api/server/_backends.py index 37afa04e05..0b180a42cc 100644 --- a/src/dstack/api/server/_backends.py +++ b/src/dstack/api/server/_backends.py @@ -6,6 +6,7 @@ AnyBackendConfigWithCreds, AnyBackendConfigWithCredsTagged, ) +from dstack._internal.core.compatibility.backends import get_backend_config_excludes from dstack._internal.core.models.backends.base import BackendType from dstack._internal.core.models.common import validate_extra_ignore from dstack._internal.server.schemas.backends import DeleteBackendsRequest @@ -27,7 +28,8 @@ def create( self, project_name: str, config: AnyBackendConfigWithCreds ) -> AnyBackendConfigWithCreds: resp = self._request( - f"/api/project/{project_name}/backends/create", body=config.model_dump_json() + f"/api/project/{project_name}/backends/create", + body=config.model_dump_json(exclude=get_backend_config_excludes(config)), ) return validate_extra_ignore(AnyBackendConfigWithCredsTagged, resp.json()) @@ -35,7 +37,8 @@ def update( self, project_name: str, config: AnyBackendConfigWithCreds ) -> AnyBackendConfigWithCreds: resp = self._request( - f"/api/project/{project_name}/backends/update", body=config.model_dump_json() + f"/api/project/{project_name}/backends/update", + body=config.model_dump_json(exclude=get_backend_config_excludes(config)), ) return validate_extra_ignore(AnyBackendConfigWithCredsTagged, resp.json()) diff --git a/src/tests/_internal/core/backends/hotaisle/test_compute.py b/src/tests/_internal/core/backends/hotaisle/test_compute.py index 2ad0b61e52..562014770a 100644 --- a/src/tests/_internal/core/backends/hotaisle/test_compute.py +++ b/src/tests/_internal/core/backends/hotaisle/test_compute.py @@ -46,9 +46,13 @@ } -def _compute() -> HotAisleCompute: +def _compute(bare_metal: Optional[bool] = None) -> HotAisleCompute: return HotAisleCompute( - HotAisleConfig(team_handle="test-team", creds=HotAisleAPIKeyCreds(api_key="test-key")) + HotAisleConfig( + team_handle="test-team", + creds=HotAisleAPIKeyCreds(api_key="test-key"), + bare_metal=bare_metal, + ) ) @@ -111,7 +115,18 @@ def _provisioning_data( class TestGetAllOffersWithAvailability: - def test_returns_vm_and_bare_metal_offers(self, requests_mock): + @pytest.mark.parametrize( + ("bare_metal", "expected_instance_types"), + [ + (None, ["vm-mi300x-1"]), + (False, ["vm-mi300x-1"]), + (True, ["vm-mi300x-1", "bm-mi300x-8"]), + ], + ids=["default", "disabled", "enabled"], + ) + def test_returns_bare_metal_offers_only_if_enabled( + self, requests_mock, bare_metal, expected_instance_types + ): requests_mock.get( f"{API_URL}/teams/test-team/virtual_machines/available/", json=[{"OnDemandPrice": 199, "Specs": VM_SPECS}], @@ -121,11 +136,29 @@ def test_returns_vm_and_bare_metal_offers(self, requests_mock): json=[{"OnDemandPrice": 2712, "Specs": BARE_METAL_SPECS}], ) - offers = _compute().get_all_offers_with_availability(unallocated_resources=False) + offers = _compute(bare_metal=bare_metal).get_all_offers_with_availability( + unallocated_resources=False + ) + + assert [offer.instance.name for offer in offers] == expected_instance_types + + def test_keeps_specs_in_backend_data(self, requests_mock): + requests_mock.get( + f"{API_URL}/teams/test-team/virtual_machines/available/", + json=[{"OnDemandPrice": 199, "Specs": VM_SPECS}], + ) + requests_mock.get( + f"{API_URL}/teams/test-team/bare_metal/available/", + json=[{"OnDemandPrice": 2712, "Specs": BARE_METAL_SPECS}], + ) + + offers = _compute(bare_metal=True).get_all_offers_with_availability( + unallocated_resources=False + ) - assert [(offer.instance.name, offer.backend_data) for offer in offers] == [ - ("vm-mi300x-1", {"vm_specs": VM_SPECS}), - ("bm-mi300x-8", {"bare_metal_specs": BARE_METAL_SPECS}), + assert [offer.backend_data for offer in offers] == [ + {"vm_specs": VM_SPECS}, + {"bare_metal_specs": BARE_METAL_SPECS}, ] diff --git a/src/tests/_internal/core/compatibility/__init__.py b/src/tests/_internal/core/compatibility/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/src/tests/_internal/core/compatibility/test_backends.py b/src/tests/_internal/core/compatibility/test_backends.py new file mode 100644 index 0000000000..20e55a002c --- /dev/null +++ b/src/tests/_internal/core/compatibility/test_backends.py @@ -0,0 +1,27 @@ +import json + +from dstack._internal.core.backends.hotaisle.models import ( + HotAisleAPIKeyCreds, + HotAisleBackendConfigWithCreds, +) +from dstack._internal.core.compatibility.backends import get_backend_config_excludes + + +def _request_body(config: HotAisleBackendConfigWithCreds) -> dict: + return json.loads(config.model_dump_json(exclude=get_backend_config_excludes(config))) + + +class TestGetBackendConfigExcludes: + def test_excludes_unset_hotaisle_bare_metal(self): + config = HotAisleBackendConfigWithCreds( + team_handle="test-team", creds=HotAisleAPIKeyCreds(api_key="test-key") + ) + + assert "bare_metal" not in _request_body(config) + + def test_keeps_set_hotaisle_bare_metal(self): + config = HotAisleBackendConfigWithCreds( + team_handle="test-team", creds=HotAisleAPIKeyCreds(api_key="test-key"), bare_metal=True + ) + + assert _request_body(config)["bare_metal"] is True