From c221dbadf0402bbcec9c8eabe7275f646ac1538f Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Mon, 5 Oct 2026 13:18:34 +0200 Subject: [PATCH 1/2] Support Vultr VDM GPU plans --- src/gpuhunt/providers/vultr.py | 101 ++++++++++++------- src/tests/providers/test_vultr.py | 157 +++++++++++++++++++++++++++++- 2 files changed, 223 insertions(+), 35 deletions(-) diff --git a/src/gpuhunt/providers/vultr.py b/src/gpuhunt/providers/vultr.py index f547c5c..eec6f92 100644 --- a/src/gpuhunt/providers/vultr.py +++ b/src/gpuhunt/providers/vultr.py @@ -1,6 +1,6 @@ import logging from collections.abc import Iterator -from typing import Any, cast +from typing import Any import requests from requests import Response @@ -38,7 +38,7 @@ def get( def fetch_offers() -> list[CatalogItem]: """Fetch plans with types: - 1. Cloud GPU (vcg), + 1. GPU plans (vcg, vdm), 2. Bare Metal (vbm), 3. and other CPU plans, including: Cloud Compute (vc2), @@ -58,17 +58,18 @@ def _make_offers( bare_metal_plans = bare_metal_plans_response.json()["plans_metal"] other_plans = other_plans_response.json()["plans"] - for plan in bare_metal_plans: - for location in _iter_locations(plan): - catalog_item = get_bare_metal_plans(plan, location) - if catalog_item: - offers.append(catalog_item) - - for plan in other_plans: - for location in _iter_locations(plan): - catalog_item = get_instance_plans(plan, location) - if catalog_item: - offers.append(catalog_item) + for plans, make_offer in ( + (bare_metal_plans, get_bare_metal_plans), + (other_plans, get_instance_plans), + ): + for plan in plans: + # Only publish on-demand offers. Accept responses that omit this flag. + if plan.get("deploy_ondemand") is False: + continue + for location in _iter_locations(plan): + catalog_item = make_offer(plan, location) + if catalog_item: + offers.append(catalog_item) # Vultr's free tier plan is priced at 0, which would rank it above every paid plan. # Offers are expected to carry a real price, so zero-priced plans are not published. @@ -134,32 +135,17 @@ def get_instance_plans(plan: dict, location: str) -> CatalogItem | None: spot=False, disk_size=plan["disk"], ) - elif plan_type == "vcg": - gpu_type = cast(str | None, plan.get("gpu_type")) - if not gpu_type: - logger.warning("Missing gpu_type for plan %s, skipping", plan["id"]) - return None - if "_" not in gpu_type: - logger.warning( - "Failed to parse gpu_type %s for plan %s, skipping", gpu_type, plan["id"] - ) + elif plan_type in ["vcg", "vdm"]: + gpu_details = _get_instance_gpu_details(plan) + if gpu_details is None: return None - gpu_name = gpu_type.split("_")[1] + gpu_count, gpu_name, gpu_memory = gpu_details gpu_vendor = get_gpu_vendor(gpu_name) if not gpu_vendor: logger.warning( - "Failed to detect GPU vendor %s for plan %s, skipping", gpu_type, plan["id"] - ) - return None - gpu_memory = get_gpu_memory(gpu_name) - if not gpu_memory: - logger.warning( - "Failed to detect GPU memory %s for plan %s, skipping", gpu_type, plan["id"] + "Failed to detect GPU vendor %s for plan %s, skipping", gpu_name, plan["id"] ) return None - gpu_memory_total = cast(int, plan["gpu_vram_gb"]) - # For fractional GPU, gpu_count=1 - gpu_count = max(1, gpu_memory_total // gpu_memory) if is_nvidia_superchip(gpu_name): cpu_arch = CPUArchitecture.ARM return CatalogItem( @@ -172,7 +158,7 @@ def get_instance_plans(plan: dict, location: str) -> CatalogItem | None: memory=plan["ram"] / 1024, gpu_count=gpu_count, gpu_name=gpu_name, - gpu_memory=gpu_memory_total / gpu_count, + gpu_memory=gpu_memory, gpu_vendor=gpu_vendor, spot=False, disk_size=plan["disk"], @@ -180,6 +166,34 @@ def get_instance_plans(plan: dict, location: str) -> CatalogItem | None: return None +def _get_instance_gpu_details(plan: dict) -> tuple[int, str, float] | None: + """Return GPU count, model name, and memory per GPU in GB.""" + if plan["type"] == "vdm": + # VDM plans omit GPU metadata, so use known configurations as for bare metal. + gpu_details = VDM_GPU_DETAILS.get(plan["id"]) + if gpu_details is None: + logger.warning("Skipping unknown VDM GPU plan %s", plan["id"]) + return gpu_details + + gpu_type = plan.get("gpu_type") + if not gpu_type or "_" not in gpu_type: + logger.warning( + "Missing or invalid gpu_type %s for plan %s, skipping", gpu_type, plan["id"] + ) + return None + gpu_name = gpu_type.split("_")[1] + full_gpu_memory = get_gpu_memory(gpu_name) + if not full_gpu_memory: + logger.warning( + "Failed to detect GPU memory %s for plan %s, skipping", gpu_type, plan["id"] + ) + return None + total_gpu_memory = plan["gpu_vram_gb"] + # For fractional GPU, gpu_count=1 + gpu_count = max(1, total_gpu_memory // full_gpu_memory) + return gpu_count, gpu_name, total_gpu_memory / gpu_count + + def get_gpu_memory(gpu_name: str) -> int | None: if gpu_name.upper() == "A100": return 80 # VULTR A100 instances have 80GB @@ -209,4 +223,23 @@ def _make_request(method: str, path: str, data: Any = None) -> Response: "vbm-64c-2048gb-8-l40-gpu": (8, "L40S", 48), "vbm-72c-480gb-gh200-gpu": (1, "GH200", 96), "vbm-256c-2048gb-8-mi300x-gpu": (8, "MI300X", 192), + "vbm-256c-3072gb-8-mi325x-gpu": (8, "MI325X", 256), + "vbm-256c-3072gb-8-b200-gpu": (8, "B200", 192), + "vbm-256c-3072gb-8-mi355x-gpu": (8, "MI355X", 288), +} + +VDM_GPU_DETAILS = { + "vcg-a16-6c-64g-16vram": (1, "A16", 16), + "vcg-a16-96c-960g-256vram": (16, "A16", 16), + "vcg-a16-96c-878g-256vram": (16, "A16", 16), + "vcg-a40-24c-120g-48vram": (1, "A40", 48), + "vcg-a40-96c-480g-192vram": (4, "A40", 48), + "vcg-a100-12c-120g-80vram": (1, "A100", 80), + "vcg-a100-96c-960g-640vram": (8, "A100", 80), + "vcg-h100-216c-1914gb-640vram": (8, "H100", 80), + "vcg-b200-248c-2826g-1536vram": (8, "B200", 192), + "vcg-mi355x-252c-2872g-2304vram": (8, "MI355X", 288), + # Use Vultr's documented 8 x 256 GB MI325X configuration. + # The VDM plan's 1536vram suffix conflicts; its allocation is not verified. + "vcg-mi325x-252c-2872g-1536vram": (8, "MI325X", 256), } diff --git a/src/tests/providers/test_vultr.py b/src/tests/providers/test_vultr.py index 2c0f53e..ac26e06 100644 --- a/src/tests/providers/test_vultr.py +++ b/src/tests/providers/test_vultr.py @@ -1,6 +1,14 @@ +import pytest + import gpuhunt._internal.catalog as internal_catalog from gpuhunt import Catalog -from gpuhunt.providers.vultr import VultrProvider, fetch_offers +from gpuhunt._internal.models import AcceleratorVendor +from gpuhunt.providers.vultr import ( + VultrProvider, + fetch_offers, + get_bare_metal_plans, + get_instance_plans, +) bare_metal = { "plans_metal": [ @@ -92,6 +100,36 @@ ] } +vdm_gpu_plan = { + # Public /plans response: a GPU plan whose type differs from its ID prefix, + # and which has none of the gpu_type/gpu_count/gpu_vram_gb fields. + "id": "vcg-a40-24c-120g-48vram", + "vcpu_count": 24, + "ram": 122880, + "disk": 1400, + "disk_type": "DEDICATEDMETAL", + "monthly_cost": 1250, + "hourly_cost": 1.712, + "type": "vdm", + "locations": ["blr"], + "deploy_ondemand": True, + "deploy_preemptible": False, + "gpu_brand": "NVIDIA", +} + +vdm_amd_plan = { + "id": "vcg-mi325x-252c-2872g-1536vram", + "vcpu_count": 252, + "ram": 2940928, + "disk": 14336, + "hourly_cost": 36.92, + "type": "vdm", + "locations": [], + "deploy_ondemand": False, + "deploy_preemptible": True, + "gpu_brand": "AMD", +} + def test_fetch_offers(requests_mock): # Mocking the responses for the API endpoints @@ -121,3 +159,120 @@ def test_fetch_offers_skips_empty_locations(requests_mock): ) assert [offer.location for offer in fetch_offers()] == ["ewr", "ord"] + + +class TestFetchOffers: + def test_vdm_without_gpu_metadata(self, requests_mock): + _mock_plans(requests_mock, plans=[vdm_gpu_plan], plans_metal=[]) + + [offer] = fetch_offers() + + assert offer.instance_name == "vcg-a40-24c-120g-48vram" + assert offer.location == "blr" + assert offer.cpu == 24 + assert offer.memory == 120 + assert offer.disk_size == 1400 + assert offer.price == 1.712 + assert offer.gpu_name == "A40" + assert offer.gpu_count == 1 + assert offer.gpu_memory == 48 + assert offer.gpu_vendor == AcceleratorVendor.NVIDIA + assert not offer.spot + + def test_unknown_vdm_plan(self, requests_mock): + plan = {**vdm_gpu_plan, "id": "vcg-a40-48c-240g-96vram"} + _mock_plans(requests_mock, plans=[plan], plans_metal=[]) + + assert fetch_offers() == [] + + def test_vdm_amd_plan(self, requests_mock): + # Exercise metadata conversion independently of current availability. + plan = { + **vdm_amd_plan, + "locations": ["ord"], + "deploy_ondemand": True, + } + _mock_plans(requests_mock, plans=[plan], plans_metal=[]) + + [offer] = fetch_offers() + + assert offer.gpu_vendor == AcceleratorVendor.AMD + assert offer.gpu_name == "MI325X" + assert offer.gpu_count == 8 + assert offer.gpu_memory == 256 + + @pytest.mark.parametrize( + ("plan", "is_bare_metal"), + [ + (vdm_amd_plan, False), + (vm_instances["plans"][0], False), + (bare_metal["plans_metal"][1], True), + ], + ) + def test_on_demand_plans(self, requests_mock, plan, is_bare_metal): + plan = {k: v for k, v in plan.items() if k != "deploy_ondemand"} + plans = [ + {**plan, "locations": ["ewr"], "deploy_ondemand": True}, + {**plan, "locations": ["ord"], "deploy_ondemand": False, "deploy_preemptible": True}, + {**plan, "locations": ["blr"]}, + ] + _mock_plans( + requests_mock, + plans=[] if is_bare_metal else plans, + plans_metal=plans if is_bare_metal else [], + ) + + assert [offer.location for offer in fetch_offers()] == ["ewr", "blr"] + + +class TestGetInstancePlans: + @pytest.mark.parametrize( + ("gpu_type", "total_vram", "gpu_count", "gpu_memory"), + [ + ("NVIDIA_A100", 4, 1, 4), + ("NVIDIA_A100_SXM", 40, 1, 40), + ("NVIDIA_A100", 160, 2, 80), + ], + ) + def test_vcg_gpu_memory(self, gpu_type, total_vram, gpu_count, gpu_memory): + plan = { + **vm_instances["plans"][0], + "gpu_type": gpu_type, + "gpu_vram_gb": total_vram, + } + + offer = get_instance_plans(plan, "ewr") + + assert offer is not None + assert (offer.gpu_count, offer.gpu_memory) == (gpu_count, gpu_memory) + + +class TestGetBareMetalPlans: + def test_mi325x_memory(self): + # The Vultr plan's memory differs from the generic accelerator catalog. + plan = { + "id": "vbm-256c-3072gb-8-mi325x-gpu", + "cpu_threads": 256, + "ram": 3145728, + "disk": 3576, + "hourly_cost": 36.92, + "gpu_count": 8, + "gpu_vram_gb": 2048, + } + + offer = get_bare_metal_plans(plan, "ord") + + assert offer is not None + assert offer.gpu_vendor == AcceleratorVendor.AMD + assert offer.gpu_name == "MI325X" + assert offer.gpu_count == plan["gpu_count"] + assert offer.gpu_memory == plan["gpu_vram_gb"] / plan["gpu_count"] + + +def _mock_plans(requests_mock, *, plans: list[dict], plans_metal: list[dict]) -> None: + requests_mock.get( + "https://api.vultr.com/v2/plans?type=all&per_page=500", json={"plans": plans} + ) + requests_mock.get( + "https://api.vultr.com/v2/plans-metal?per_page=500", json={"plans_metal": plans_metal} + ) From d31fdfb04484285b971ca9a5512f88628df3ede1 Mon Sep 17 00:00:00 2001 From: Andrey Cheptsov Date: Mon, 5 Oct 2026 13:37:34 +0200 Subject: [PATCH 2/2] Use authenticated Vultr VDM availability --- .github/workflows/catalogs.yml | 2 + README.md | 3 ++ src/gpuhunt/providers/vultr.py | 55 +++++++++++++++++++--- src/tests/providers/test_vultr.py | 76 +++++++++++++++++++++++++++++++ 4 files changed, 130 insertions(+), 6 deletions(-) diff --git a/.github/workflows/catalogs.yml b/.github/workflows/catalogs.yml index 6f412c6..3d5ba5b 100644 --- a/.github/workflows/catalogs.yml +++ b/.github/workflows/catalogs.yml @@ -461,6 +461,8 @@ jobs: - name: Install dependencies run: uv sync --extra dev - name: Run integrity tests + env: + VULTR_API_KEY: ${{ secrets.VULTR_API_KEY }} run: uv run pytest src/integrity_tests/test_vultr.py # The v2 catalog bundles all providers into a single file. It is published for dstack diff --git a/README.md b/README.md index 5505096..f331208 100644 --- a/README.md +++ b/README.md @@ -58,6 +58,9 @@ items = catalog.query() print(*items, sep="\n") ``` +Vultr uses `VULTR_API_KEY`, when set, to query account-specific availability for VDM GPU plans. +Its public plan locations can omit available VDM plans. + ## Supported providers * AWS diff --git a/src/gpuhunt/providers/vultr.py b/src/gpuhunt/providers/vultr.py index eec6f92..6275ef0 100644 --- a/src/gpuhunt/providers/vultr.py +++ b/src/gpuhunt/providers/vultr.py @@ -1,5 +1,7 @@ import logging +import os from collections.abc import Iterator +from concurrent.futures import ThreadPoolExecutor from typing import Any import requests @@ -22,9 +24,13 @@ class VultrProvider(OnlineProvider): NAME = "vultr" + def __init__(self, api_key: str | None = None, regions: list[str] | None = None): + self.api_key = api_key + self.regions = regions + @classmethod def from_env(cls) -> "VultrProvider": - return cls() + return cls(api_key=os.getenv("VULTR_API_KEY")) def get( self, @@ -32,11 +38,13 @@ def get( balance_resources: bool = True, apply_filter: bool = False, ) -> list[CatalogItem]: - offers = fetch_offers() + offers = fetch_offers(api_key=self.api_key, regions=self.regions) return sorted(offers, key=lambda i: i.price) -def fetch_offers() -> list[CatalogItem]: +def fetch_offers( + api_key: str | None = None, regions: list[str] | None = None +) -> list[CatalogItem]: """Fetch plans with types: 1. GPU plans (vcg, vdm), 2. Bare Metal (vbm), @@ -47,11 +55,40 @@ def fetch_offers() -> list[CatalogItem]: All optimized Cloud Types (voc)""" bare_metal_plans_response = _make_request("GET", "/plans-metal?per_page=500") other_plans_response = _make_request("GET", "/plans?type=all&per_page=500") - return _make_offers(bare_metal_plans_response, other_plans_response) + vdm_locations = None + if api_key: + try: + vdm_locations = _get_vdm_locations(api_key, regions) + except requests.RequestException as e: + logger.warning("Failed to fetch Vultr VDM availability: %s", e) + # Preserve existing offers without using unverified public VDM locations. + vdm_locations = {} + return _make_offers(bare_metal_plans_response, other_plans_response, vdm_locations) + + +def _get_vdm_locations(api_key: str, regions: list[str] | None) -> dict[str, list[str]]: + if not regions: + response = _make_request("GET", "/regions?per_page=500", api_key=api_key) + regions = [region["id"] for region in response.json()["regions"]] + + locations: dict[str, list[str]] = {} + with ThreadPoolExecutor(max_workers=5) as executor: + futures = { + region: executor.submit( + _make_request, "GET", f"/regions/{region}/availability?type=vdm", api_key=api_key + ) + for region in regions + } + for region, future in futures.items(): + for plan_id in future.result().json()["available_plans"]: + locations.setdefault(plan_id, []).append(region) + return locations def _make_offers( - bare_metal_plans_response: Response, other_plans_response: Response + bare_metal_plans_response: Response, + other_plans_response: Response, + vdm_locations: dict[str, list[str]] | None = None, ) -> list[CatalogItem]: offers: list[CatalogItem] = [] @@ -66,6 +103,9 @@ def _make_offers( # Only publish on-demand offers. Accept responses that omit this flag. if plan.get("deploy_ondemand") is False: continue + if plan.get("type") == "vdm" and vdm_locations is not None: + # Public VDM locations can be empty despite availability for this account. + plan["locations"] = vdm_locations.get(plan["id"], []) for location in _iter_locations(plan): catalog_item = make_offer(plan, location) if catalog_item: @@ -205,11 +245,14 @@ def get_gpu_memory(gpu_name: str) -> int | None: return None -def _make_request(method: str, path: str, data: Any = None) -> Response: +def _make_request( + method: str, path: str, data: Any = None, *, api_key: str | None = None +) -> Response: response = requests.request( method=method, url=API_URL + path, json=data, + headers={"Authorization": f"Bearer {api_key}"} if api_key else None, timeout=30, ) response.raise_for_status() diff --git a/src/tests/providers/test_vultr.py b/src/tests/providers/test_vultr.py index ac26e06..2ebf5c7 100644 --- a/src/tests/providers/test_vultr.py +++ b/src/tests/providers/test_vultr.py @@ -100,6 +100,16 @@ ] } +cpu_plan = { + "id": "vc2-2c-4gb", + "type": "vc2", + "vcpu_count": 2, + "ram": 4096, + "disk": 80, + "hourly_cost": 0.027, + "locations": ["fra"], +} + vdm_gpu_plan = { # Public /plans response: a GPU plan whose type differs from its ID prefix, # and which has none of the gpu_type/gpu_count/gpu_vram_gb fields. @@ -224,6 +234,72 @@ def test_on_demand_plans(self, requests_mock, plan, is_bare_metal): assert [offer.location for offer in fetch_offers()] == ["ewr", "blr"] + @pytest.mark.parametrize( + ("public_locations", "available_plans", "expected_locations"), + [ + ([], [vdm_gpu_plan["id"]], ["blr"]), + (["ewr"], [], []), + ], + ) + def test_account_availability( + self, requests_mock, monkeypatch, public_locations, available_plans, expected_locations + ): + _mock_plans( + requests_mock, + plans=[ + *vm_instances["plans"], + cpu_plan, + {**vdm_gpu_plan, "locations": public_locations}, + ], + plans_metal=bare_metal["plans_metal"], + ) + # No-key callers retain the public catalog without querying regional availability. + existing_offers = [ + offer for offer in VultrProvider().get() if offer.instance_name != vdm_gpu_plan["id"] + ] + monkeypatch.setenv("VULTR_API_KEY", "test-key") + requests_mock.get( + "https://api.vultr.com/v2/regions?per_page=500", + request_headers={"Authorization": "Bearer test-key"}, + json={"regions": [{"id": "blr"}, {"id": "ord"}]}, + ) + for region in ("blr", "ord"): + requests_mock.get( + f"https://api.vultr.com/v2/regions/{region}/availability?type=vdm", + request_headers={"Authorization": "Bearer test-key"}, + json={ + "available_plans": available_plans if region == "blr" else [], + # VPC-only instances cannot provide the public IP expected by dstack. + "available_vpc_only_plans": [vdm_gpu_plan["id"]], + }, + ) + + offers = VultrProvider.from_env().get() + + assert [ + offer.location for offer in offers if offer.instance_name == vdm_gpu_plan["id"] + ] == expected_locations + assert [ + offer for offer in offers if offer.instance_name != vdm_gpu_plan["id"] + ] == existing_offers + + def test_account_availability_failure_preserves_existing_offers(self, requests_mock): + _mock_plans( + requests_mock, + plans=[*vm_instances["plans"], cpu_plan, vdm_gpu_plan], + plans_metal=bare_metal["plans_metal"], + ) + existing_offers = [ + offer for offer in VultrProvider().get() if offer.instance_name != vdm_gpu_plan["id"] + ] + requests_mock.get( + "https://api.vultr.com/v2/regions/blr/availability?type=vdm", status_code=503 + ) + + offers = VultrProvider(api_key="test-key", regions=["blr"]).get() + + assert offers == existing_offers + class TestGetInstancePlans: @pytest.mark.parametrize(