From 0ba1080dd004c3e5f7a493859d0131c1a9a6b34d Mon Sep 17 00:00:00 2001 From: leejet Date: Sun, 13 Sep 2026 17:34:51 +0800 Subject: [PATCH] feat: preserve explicit backend assignments during auto-fit --- docs/backend.md | 37 ++++--- docs/performance.md | 2 +- examples/common/common.cpp | 4 +- src/core/backend_fit.cpp | 165 +++++++++++++++++++++++------- src/core/ggml_extend_backend.cpp | 2 +- src/core/ggml_extend_backend.h | 1 + src/pipeline/diffusion_engine.cpp | 2 +- 7 files changed, 161 insertions(+), 52 deletions(-) diff --git a/docs/backend.md b/docs/backend.md index 5d7f89d27..c6594a98a 100644 --- a/docs/backend.md +++ b/docs/backend.md @@ -5,7 +5,8 @@ - `--backend` selects the runtime backend used to execute model graphs. - `--params-backend` selects where model parameters are kept. -If `--params-backend` is not set, parameters use the same backend as their module runtime backend. +If `--params-backend` is not set, auto-fit chooses parameter placement. With +`--auto-fit off`, parameters use the same backend as their module runtime backend. ## Syntax @@ -129,17 +130,21 @@ warning. ## Automatic placement (`--auto-fit on|off`) `--auto-fit` requires `on` or `off` and defaults to `on` when omitted. -Explicit `--backend` or `--params-backend` assignments disable auto-fit, +Explicit `--params-backend` assignments disable auto-fit, regardless of argument order, even with `--auto-fit on`. -When enabled, auto-fit uses one GPU for `diffusion` / `te` / `vae` computation. It chooses -the GPU with the largest available memory budget (the first device on a tie), -then derives parameter placements from the model metadata and the remaining -memory budgets. The chosen backend specifications are printed. +Auto-fit preserves explicit `--backend` assignments, including per-module +assignments and device lists. For modules without a runtime assignment, it chooses +the GPU with the largest available memory budget (the first device on a tie). +It then derives parameter placements from the model metadata, each module's +compute devices, and the remaining memory budgets. The chosen backend +specifications are printed. ```shell sd-cli -m model.safetensors -p "a cat" --auto-fit on sd-cli -m model.safetensors -p "a cat" --auto-fit on --max-vram cuda0=8,cuda1=14 +sd-cli -m model.safetensors -p "a cat" --backend cuda0 +sd-cli -m model.safetensors -p "a cat" --backend diffusion=cuda0,te=cpu,vae=cuda1 sd-cli -m model.safetensors -p "a cat" --auto-fit off ``` @@ -153,7 +158,7 @@ Components are considered in `diffusion`, `te`, `vae` order so that repeatedly used diffusion weights have priority. Each component's weights use the first storage location with enough remaining budget: -1. The main GPU, leaving estimated space for computation and weight staging. +1. The component's compute GPU, leaving estimated space for computation and weight staging. 2. CPU RAM, reserving the larger of 2 GiB or 10% of available RAM for other work. 3. Another GPU, choosing the one with the largest remaining budget that fits. 4. Disk, reloading weights on demand. @@ -170,10 +175,17 @@ weight to be copied again at every step. RAM and GPU budgets are shared across components. Each component uses a single parameter backend; several other GPUs' capacities are not combined to store one component. If available RAM cannot be queried, RAM residency is skipped. -Other GPUs store weights only: weights are copied to the main GPU for execution. -Auto-fit does not select multi-GPU layer/row computation, so `--split-mode` does -not change its placements. Use explicit backend assignments for multi-GPU -computation. +Weights stored on another GPU are copied to the component's compute devices for +execution. CPU modules use RAM or disk. Compute reserves and cache priority are +accounted for separately on each device, so a CPU module does not reserve GPU +space. Storage on another module's GPU also leaves room for that module's work. + +Auto-fit does not select multi-GPU layer/row computation itself. Explicit device +lists and `--split-mode` still control that computation. Before the runners have +built their split plans, auto-fit conservatively counts the full component size +on each listed GPU when checking residency and cache space. This can offload +parameters even when a split layout would fit; use `--auto-fit off` to keep the +default split-device parameter placement. For example, a diffusion model whose full weights exceed the main GPU's budget can use `--backend diffusion=cuda0 --params-backend diffusion=cpu` when RAM is @@ -292,6 +304,7 @@ The example CLI/server still accepts these older CPU placement flags as compatib Because this default is inserted first, later explicit `--params-backend` entries can still override it, for example `--offload-to-cpu --params-backend te=disk` keeps non-TE parameters on CPU and reloads TE parameters from disk. Library callers should set `backend` and `params_backend` directly. `sd_ctx_params_init()` -enables `auto_fit` by default; nonempty `backend` or `params_backend` assignments disable it. +enables `auto_fit` by default; a nonempty `params_backend` assignment disables it. +The `backend` assignment constrains auto-fit's compute placement. The old CPU/offload fields are no longer part of the C API. Explicit `--backend` and `--params-backend` assignments are preferred for new commands. diff --git a/docs/performance.md b/docs/performance.md index 0aa0afbed..216672020 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -27,7 +27,7 @@ Using `--offload-to-cpu` allows you to offload weights to the CPU, saving VRAM w ## Use params backend to reduce VRAM or RAM usage. -`--params-backend` controls where model parameters are kept. If it is not set, parameters use the same backend as `--backend`, so a GPU runtime backend also keeps parameters in VRAM. +`--params-backend` controls where model parameters are kept. If it is not set, auto-fit chooses parameter placement while preserving `--backend`. With `--auto-fit off`, parameters use the same backend as `--backend`, so a GPU runtime backend also keeps parameters in VRAM. Use CPU params to reduce VRAM usage: diff --git a/examples/common/common.cpp b/examples/common/common.cpp index cdf21bc7c..2357dcbb1 100644 --- a/examples/common/common.cpp +++ b/examples/common/common.cpp @@ -720,9 +720,9 @@ ArgOptions SDContextParams::get_options() { }}, {"", "--auto-fit", - "on|off (default: on). Use one GPU for diffusion/te/vae computation and place weights on that GPU, " + "on|off (default: on). Preserve --backend (otherwise select one GPU) and place weights on the compute GPU, " "RAM, another GPU, or disk in that order, according to available memory (--max-vram limits GPU budgets). " - "Disabled by explicit --backend or --params-backend; uses automatic graph segmentation when needed", + "Disabled by explicit --params-backend; uses automatic graph segmentation when needed", on_auto_fit_arg}, {"", "--type", diff --git a/src/core/backend_fit.cpp b/src/core/backend_fit.cpp index 82150ab91..4eb61704e 100644 --- a/src/core/backend_fit.cpp +++ b/src/core/backend_fit.cpp @@ -58,9 +58,15 @@ namespace sd::backend_fit { size_t params_device = SIZE_MAX; }; + struct Runtime { + std::string name; + std::vector devices; + }; + struct Plan { bool valid = false; size_t main_device = SIZE_MAX; + std::vector runtimes; std::vector decisions; }; @@ -121,11 +127,14 @@ namespace sd::backend_fit { return name; } - static std::vector enumerate_gpu_devices(const sd::ggml_graph_cut::MaxVramAssignment& budgets) { + static std::vector enumerate_gpu_devices(const sd::ggml_graph_cut::MaxVramAssignment& budgets, + bool include_other_devices) { std::vector out; for (size_t i = 0; i < ggml_backend_dev_count(); ++i) { ggml_backend_dev_t dev = ggml_backend_dev_get(i); - if (ggml_backend_dev_type(dev) != GGML_BACKEND_DEVICE_TYPE_GPU) { + const auto type = ggml_backend_dev_type(dev); + if (type != GGML_BACKEND_DEVICE_TYPE_GPU && + (!include_other_devices || type == GGML_BACKEND_DEVICE_TYPE_CPU)) { continue; } Device device; @@ -183,18 +192,29 @@ namespace sd::backend_fit { return -1; } - static Plan compute_plan(const std::vector& components, - const std::vector& devices, - int64_t ram_budget_bytes) { - Plan plan; + static size_t select_main_device(const std::vector& devices) { + size_t main_device = SIZE_MAX; for (size_t di = 0; di < devices.size(); ++di) { if (devices[di].budget_bytes > 0 && - (plan.main_device == SIZE_MAX || devices[di].budget_bytes > devices[plan.main_device].budget_bytes)) { - plan.main_device = di; + (main_device == SIZE_MAX || devices[di].budget_bytes > devices[main_device].budget_bytes)) { + main_device = di; } } - if (plan.main_device == SIZE_MAX) { - return plan; + return main_device; + } + + static Plan compute_plan(const std::vector& components, + const std::vector& devices, + int64_t ram_budget_bytes, + const std::vector& runtimes = {}) { + Plan plan; + plan.main_device = select_main_device(devices); + plan.runtimes = runtimes; + if (plan.runtimes.empty()) { + if (plan.main_device == SIZE_MAX) { + return plan; + } + plan.runtimes.resize(components.size(), {devices[plan.main_device].name, {plan.main_device}}); } std::vector order(components.size()); @@ -212,31 +232,47 @@ namespace sd::backend_fit { ram_budget_bytes = std::max(ram_budget_bytes, 0); plan.decisions.resize(components.size()); - for (size_t ci : order) { - const Component& comp = components[ci]; - Decision& decision = plan.decisions[ci]; - if (comp.params_bytes == 0) { - continue; - } - - // Higher-priority offloaded weights need GPU cache space across graph runs. + auto uses_device = [&](size_t ci, size_t di) { + const auto& runtime_devices = plan.runtimes[ci].devices; + return std::find(runtime_devices.begin(), runtime_devices.end(), di) != runtime_devices.end(); + }; + auto headroom_for = [&](size_t ci, size_t di) { + // Higher-priority offloaded weights need cache space on their compute devices. int64_t headroom = 0; for (size_t other = 0; other < components.size(); ++other) { - if (components[other].params_bytes == 0) { + if (components[other].params_bytes == 0 || !uses_device(other, di)) { continue; } const bool resident = other == ci || plan.decisions[other].params_location == ParamsLocation::MAIN_GPU; - const int64_t cached_weights = components[other].kind < comp.kind + const int64_t cached_weights = components[other].kind < components[ci].kind ? components[other].params_bytes : components[other].staging_bytes; headroom = std::max(headroom, components[other].reserve_bytes + (resident ? 0 : cached_weights)); } - int64_t& main_remaining = remaining[plan.main_device]; - if (headroom <= main_remaining && comp.params_bytes <= main_remaining - headroom) { + return headroom; + }; + + for (size_t ci : order) { + const Component& comp = components[ci]; + Decision& decision = plan.decisions[ci]; + if (comp.params_bytes == 0) { + continue; + } + + const auto& runtime_devices = plan.runtimes[ci].devices; + const bool fits_runtime = !runtime_devices.empty() && + std::all_of(runtime_devices.begin(), runtime_devices.end(), [&](size_t di) { + const int64_t headroom = headroom_for(ci, di); + return headroom <= remaining[di] && comp.params_bytes <= remaining[di] - headroom; + }); + if (fits_runtime) { decision.params_location = ParamsLocation::MAIN_GPU; - decision.params_device = plan.main_device; - main_remaining -= comp.params_bytes; + decision.params_device = runtime_devices.front(); + // Exact split allocations are unavailable until the runners build their plans. + for (size_t di : runtime_devices) { + remaining[di] -= comp.params_bytes; + } continue; } if (comp.params_bytes <= ram_budget_bytes) { @@ -244,10 +280,14 @@ namespace sd::backend_fit { ram_budget_bytes -= comp.params_bytes; continue; } + if (runtime_devices.empty()) { + continue; + } size_t best = SIZE_MAX; for (size_t di = 0; di < devices.size(); ++di) { - if (di != plan.main_device && comp.params_bytes <= remaining[di] && + const int64_t headroom = headroom_for(ci, di); + if (!uses_device(ci, di) && headroom <= remaining[di] && comp.params_bytes <= remaining[di] - headroom && (best == SIZE_MAX || remaining[di] > remaining[best])) { best = di; } @@ -280,7 +320,7 @@ namespace sd::backend_fit { const std::vector& devices, int64_t free_ram, int64_t ram_budget) { - LOG_INFO("auto-fit plan (single-GPU compute on %s):", devices[plan.main_device].name.c_str()); + LOG_INFO("auto-fit plan:"); LOG_INFO(" devices:"); for (const Device& device : devices) { LOG_INFO(" %-12s %-32s free %6lld MiB, budget %6lld MiB", @@ -293,17 +333,19 @@ namespace sd::backend_fit { LOG_INFO(" RAM free %6lld MiB, params budget %6lld MiB", (long long)(free_ram / MiB), (long long)(ram_budget / MiB)); } - LOG_INFO(" main-GPU weight cache priority: diffusion > te > vae"); - LOG_INFO(" components (params: main GPU -> RAM -> other GPU -> disk):"); + LOG_INFO(" compute-device weight cache priority: diffusion > te > vae"); + LOG_INFO(" components (params: compute device -> RAM -> other GPU -> disk):"); for (size_t ci = 0; ci < components.size(); ++ci) { const Component& comp = components[ci]; if (comp.params_bytes == 0) { continue; } - const std::string params = params_backend_name(plan.decisions[ci], devices); + const std::string params = plan.decisions[ci].params_location == ParamsLocation::MAIN_GPU + ? plan.runtimes[ci].name + : params_backend_name(plan.decisions[ci], devices); LOG_INFO(" %-12s params %6lld MiB, compute reserve %5lld MiB -> compute %s, params %s", comp.name, (long long)(comp.params_bytes / MiB), (long long)(comp.reserve_bytes / MiB), - devices[plan.main_device].name.c_str(), params.c_str()); + plan.runtimes[ci].name.c_str(), params.c_str()); } } @@ -328,6 +370,51 @@ namespace sd::backend_fit { return ""; } + static bool resolve_runtimes(const std::vector& components, + const std::vector& devices, + std::string& runtime_spec, + std::vector& runtimes, + std::string& error) { + SDBackendAssignment assignment; + if (!sd_parse_backend_assignment(runtime_spec, &assignment, &error)) { + return false; + } + const size_t main_device = select_main_device(devices); + const SDBackendModule modules[] = {SDBackendModule::DIFFUSION, SDBackendModule::TE, SDBackendModule::VAE}; + for (const Component& comp : components) { + std::string name = assignment.get(modules[int(comp.kind)]); + if (name.empty()) { + name = main_device == SIZE_MAX ? "cpu" : devices[main_device].name; + if (comp.params_bytes > 0) { + append_assignment(runtime_spec, module_key(comp.kind), name); + } + } + Runtime runtime; + for (const std::string& part : split_string(name, '&')) { + if (trim(part).empty()) { + continue; + } + const std::string resolved = sd_backend_resolve_name(part); + if (resolved.empty()) { + error = "backend '" + part + "' was not found"; + return false; + } + if (!runtime.name.empty()) { + runtime.name += "&"; + } + runtime.name += resolved; + for (size_t di = 0; di < devices.size(); ++di) { + if (devices[di].name == resolved && + std::find(runtime.devices.begin(), runtime.devices.end(), di) == runtime.devices.end()) { + runtime.devices.push_back(di); + } + } + } + runtimes.push_back(std::move(runtime)); + } + return true; + } + bool derive_backend_specs(ModelLoader& loader, ggml_type override_wtype, sd::ggml_graph_cut::MaxVramAssignment& budgets, @@ -339,12 +426,18 @@ namespace sd::backend_fit { return false; } - const auto components = estimate_components(loader, override_wtype); - const auto devices = enumerate_gpu_devices(budgets); + // Resolve once to ensure dynamic backends are loaded before enumerating devices. + sd_backend_resolve_name(""); + const auto components = estimate_components(loader, override_wtype); + const auto devices = enumerate_gpu_devices(budgets, !runtime_spec.empty()); + std::vector runtimes; + if (!runtime_spec.empty() && !resolve_runtimes(components, devices, runtime_spec, runtimes, error)) { + LOG_ERROR("%s", error.c_str()); + return false; + } const int64_t free_ram = available_ram_bytes(); const int64_t ram_budget = std::max(free_ram - std::max(2048 * MiB, free_ram / 10), 0); - const auto plan = compute_plan(components, devices, ram_budget); - runtime_spec.clear(); + const auto plan = compute_plan(components, devices, ram_budget, runtimes); params_spec.clear(); if (!plan.valid) { if (devices.empty()) { @@ -362,7 +455,9 @@ namespace sd::backend_fit { continue; } const char* key = module_key(components[ci].kind); - append_assignment(runtime_spec, key, devices[plan.main_device].name); + if (runtimes.empty()) { + append_assignment(runtime_spec, key, plan.runtimes[ci].name); + } if (plan.decisions[ci].params_location != ParamsLocation::MAIN_GPU) { append_assignment(params_spec, key, params_backend_name(plan.decisions[ci], devices)); } diff --git a/src/core/ggml_extend_backend.cpp b/src/core/ggml_extend_backend.cpp index 9c4cfa42a..9072c1560 100644 --- a/src/core/ggml_extend_backend.cpp +++ b/src/core/ggml_extend_backend.cpp @@ -593,7 +593,7 @@ static ggml_backend_t sd_get_default_backend() { return backend; } -static bool sd_parse_backend_assignment(const std::string& spec, SDBackendAssignment* assignment, std::string* error) { +bool sd_parse_backend_assignment(const std::string& spec, SDBackendAssignment* assignment, std::string* error) { if (assignment == nullptr) { return false; } diff --git a/src/core/ggml_extend_backend.h b/src/core/ggml_extend_backend.h index 01652cbfb..3430ccc5b 100644 --- a/src/core/ggml_extend_backend.h +++ b/src/core/ggml_extend_backend.h @@ -93,6 +93,7 @@ ggml_status sd_backend_graph_compute_with_eval_callback(ggml_backend_t backend, sd_graph_eval_callback_t callback_eval, void* callback_eval_user_data); std::string sd_backend_resolve_name(const std::string& name); +bool sd_parse_backend_assignment(const std::string& spec, SDBackendAssignment* assignment, std::string* error); const char* sd_backend_module_name(SDBackendModule module); void ggml_ext_im_set_f32_1d(const struct ggml_tensor* tensor, int i, float value); bool add_rpc_devices(const std::string& servers); diff --git a/src/pipeline/diffusion_engine.cpp b/src/pipeline/diffusion_engine.cpp index 5d51e2417..18414017f 100644 --- a/src/pipeline/diffusion_engine.cpp +++ b/src/pipeline/diffusion_engine.cpp @@ -862,7 +862,7 @@ bool StableDiffusionGGML::init(const sd_ctx_params_t* sd_ctx_params) { backend_spec = SAFE_STR(sd_ctx_params->backend); params_backend_spec = SAFE_STR(sd_ctx_params->params_backend); split_mode_spec = SAFE_STR(sd_ctx_params->split_mode); - auto_fit_enabled = sd_ctx_params->auto_fit && backend_spec.empty() && params_backend_spec.empty(); + auto_fit_enabled = sd_ctx_params->auto_fit && params_backend_spec.empty(); max_vram_assignment.reset(0.f); { std::string error;